Trifecta-Lab / trifecta_bro /model /feature_engine.py
Brettapps's picture
Upload folder using huggingface_hub (part 21)
e23172f verified
Raw History Blame Contribute Delete
3.17 kB
"""Feature engineering.
Form-string parsing follows the FormFav convention:
- digits 1..9 are finishing positions (1st..9th)
- '0' means 10th or worse
- 'x' (or 'X') means a spell / no result (NOT a finishing position)
- '-' / '.' separators are ignored
Recent results weigh more than old ones. Quality improves/declines tracked via
slope of recent positions.
"""
from __future__ import annotations
import math
import re
from typing import Optional
# Map a finishing token to a numeric score (higher = better). x/spell = None.
_FINISH_SCORE = {1: 100, 2: 80, 3: 65, 4: 50, 5: 40, 6: 32, 7: 26, 8: 20, 9: 15, 0: 8}
def parse_form_positions(form: str) -> list[Optional[int]]:
"""Return finishing positions oldest->newest; x/spell -> None."""
if not form:
return []
tokens = re.findall(r"[0-9xX]", form)
out: list[Optional[int]] = []
for t in tokens:
if t in ("x", "X"):
out.append(None) # spell / no result
else:
out.append(int(t))
return out
def form_quality_score(form: str, max_len: int = 12) -> tuple[float, dict]:
"""Return (0..100 score, diagnostics). Recent results weighted exponentially."""
positions = parse_form_positions(form)
# normalise to latest `max_len`
recent = positions[-max_len:]
if not recent:
return 0.0, {"reason": "no form data", "recent": []}
# weight recency: newest gets highest weight
n = len(recent)
total_w = 0.0
weighted = 0.0
for i, pos in enumerate(recent):
w = 2 ** (i) # newest (last) has largest exponent
if pos is None:
# spell: mildly negative but not catastrophic; treat as 0-score
total_w += w
continue
total_w += w
weighted += _FINISH_SCORE.get(pos, 8) * w
base = (weighted / total_w) if total_w else 0.0
# consistency: fraction of recent runs that placed (1..3) — ignoring spells
placed = [p for p in recent if p is not None and p <= 3]
consistency = len(placed) / max(1, len([p for p in recent if p is not None]))
# trend: compare mean of last 3 (non-spell) vs prior 3 (non-spell)
valid = [p for p in recent if p is not None]
trend = 0.0
if len(valid) >= 4:
last3 = valid[-3:]
prev3 = valid[-6:-3]
if prev3:
m_last = sum(last3) / len(last3)
m_prev = sum(prev3) / len(prev3)
# lower position number = better; improvement => positive
trend = max(-1.0, min(1.0, (m_prev - m_last) / 3.0))
score = 0.75 * base + 20 * consistency + 15 * max(0.0, trend)
score = max(0.0, min(100.0, score))
return round(score, 1), {
"recent": recent,
"consistency": round(consistency, 2),
"trend": round(trend, 2),
"base": round(base, 1),
}
def spell_count(form: str) -> int:
"""Number of spell markers (x) — indicates layoffs."""
return len(re.findall(r"[xX]", form or ""))
def is_fresh_after_spell(form: str) -> bool:
"""True if the most recent result is a spell (just resumed)."""
positions = parse_form_positions(form)
return bool(positions) and positions[-1] is None