"""Build distance-specific models with proper calibration.""" import json from pathlib import Path import sys sys.path.insert(0, '/home/brettanthonysjoberg179/trifecta-bro-hf-space') from trifecta_bro.data.models import RaceModel, RunnerModel from trifecta_bro.model.scoring import score_runner from trifecta_bro.model.pace_analysis import classify_pace from trifecta_bro.model.distance_profiles import classify_distance DATA_DIR = Path('/home/brettanthonysjoberg179/trifecta-bro-hf-space/data') CACHE_DIR = DATA_DIR / 'cache' RESULTS_DIR = DATA_DIR / 'results' def load_cached_races(date: str) -> list[dict]: races = [] for f in CACHE_DIR.glob('*.json'): try: with open(f) as fp: data = json.load(fp) if data.get('date') == date: races.append(data) except: continue return races def load_results(date: str) -> dict: result_lookup = {} result_path = RESULTS_DIR / f"{date}-ra.json" if not result_path.exists(): result_path = RESULTS_DIR / f"{date}.json" if not result_path.exists(): return result_lookup with open(result_path) as f: rdata = json.load(f) if isinstance(rdata, dict): if 'tracks' in rdata: for track, races in rdata['tracks'].items(): if isinstance(races, dict): for rn, r in races.items(): runners = r.get('runners', []) winners = [] for runner in runners: pos = runner.get('position') if pos and pos <= 3: winners.append((pos, runner['number'])) winners.sort() actual = [w[1] for w in winners] if actual: result_lookup[(track, int(rn))] = actual elif isinstance(races, list): for r in races: if 'trifecta' in r: result_lookup[(track, r.get('race', 0))] = [int(str(x).replace('e','')) for x in r['trifecta']] else: for track, races in rdata.items(): if isinstance(races, list): for r in races: if 'trifecta' in r: result_lookup[(track, r['race'])] = [int(str(x).replace('e','')) for x in r['trifecta']] return result_lookup def score_race(race_data: dict, weights: dict) -> list[int] | None: try: runners = [] for r in race_data.get('runners', []): if r.get('scratched'): continue runner = RunnerModel( number=r['number'], name=r.get('name', ''), jockey=r.get('jockey'), trainer=r.get('trainer'), weight=r.get('weight'), barrier=r.get('barrier'), age=r.get('age'), sex=r.get('sex'), form=r.get('form', ''), last20_starts=r.get('last20Starts', ''), stats=r.get('stats', {}), ) runners.append(runner) if len(runners) < 3: return None race = RaceModel( date=race_data.get('date', ''), track=race_data.get('track', ''), track_slug=race_data.get('slug', race_data.get('track', '').lower()), race_number=race_data.get('raceNumber', 0), race_name=race_data.get('raceName', ''), distance=race_data.get('distance'), condition=race_data.get('condition'), race_class=race_data.get('raceClass'), abandoned=race_data.get('abandoned', False), start_time=race_data.get('startTime'), prize_money=str(race_data.get('prizeMoney', '')), number_of_runners=race_data.get('numberOfRunners', len(runners)), runners=runners, ) pace = classify_pace(race, runners) scored = [] for r in runners: sc = score_runner(r, race, pace, weights) scored.append((r.number, sc['score'])) scored.sort(key=lambda x: x[1], reverse=True) return [s[0] for s in scored[:3]] except: return None def evaluate_date(date: str, weights: dict) -> dict: result_lookup = load_results(date) races = load_cached_races(date) stats = {"total": 0, "top1": 0, "exact": 0, "box": 0} for race_data in races: track = race_data.get('track', '') race_num = race_data.get('raceNumber', 0) actual = result_lookup.get((track, race_num)) if not actual: continue pred = score_race(race_data, weights) if not pred: continue stats["total"] += 1 if pred == actual: stats["exact"] += 1 if set(pred) == set(actual): stats["box"] += 1 if pred[0] == actual[0]: stats["top1"] += 1 if stats["total"] > 0: stats["top1_pct"] = stats["top1"] / stats["total"] * 100 return stats def evaluate_all(dates: list[str], weights: dict) -> dict: agg = {"total": 0, "top1": 0, "exact": 0, "box": 0} for date in dates: stats = evaluate_date(date, weights) agg["total"] += stats["total"] agg["top1"] += stats["top1"] agg["exact"] += stats["exact"] agg["box"] += stats["box"] agg["top1_pct"] = agg["top1"] / agg["total"] * 100 if agg["total"] > 0 else 0 return agg def grid_search_for_distance(dates: list[str], distance_class: str) -> dict: """Find optimal weights for a specific distance class.""" # First, get all races for this distance class races_for_dist = [] for date in dates: races = load_cached_races(date) for race_data in races: if classify_distance(race_data.get('distance', '1200m')) == distance_class: races_for_dist.append((date, race_data)) if not races_for_dist: return {"weights": None, "top1_pct": 0, "total": 0} # Grid search best_pct = 0 best_weights = None for form in [0.15, 0.20, 0.25, 0.30]: for class_w in [0.10, 0.15, 0.20]: for dist in [0.06, 0.10, 0.14]: for td in [0.06, 0.10, 0.14]: for barrier in [0.06, 0.10, 0.14, 0.18]: for fitness in [0.04, 0.08, 0.12]: for pace in [0.02, 0.05, 0.08, 0.12]: for weight in [0.04, 0.08, 0.12]: weights = { "form": form, "class": class_w, "distance": dist, "track": dist, "track_distance": td, "condition": 0.05, "jockey": 0.06, "fitness": fitness, "barrier": barrier, "weight": weight, "pace": pace, } total_w = sum(weights.values()) weights = {k: v / total_w for k, v in weights.items()} # Evaluate on this distance class only stats = {"total": 0, "top1": 0} for date, race_data in races_for_dist: result_lookup = load_results(date) track = race_data.get('track', '') race_num = race_data.get('raceNumber', 0) actual = result_lookup.get((track, race_num)) if not actual: continue pred = score_race(race_data, weights) if not pred: continue stats["total"] += 1 if pred[0] == actual[0]: stats["top1"] += 1 if stats["total"] > 0: pct = stats["top1"] / stats["total"] * 100 if pct > best_pct: best_pct = pct best_weights = weights return {"weights": best_weights, "top1_pct": best_pct, "total": len(races_for_dist)} def main(): dates = ["2026-08-07", "2026-08-08", "2026-08-09", "2026-08-14"] print("=" * 60) print("DISTANCE-SPECIFIC OPTIMIZATION") print("=" * 60) # Find optimal weights per distance class for dist_class in ["sprint", "middle", "staying"]: print(f"\n--- {dist_class.upper()} ---") result = grid_search_for_distance(dates, dist_class) if result["weights"]: print(f"Best: {result['top1_pct']:.1f}% on {result['total']} races") w = result["weights"] print(f" form={w['form']:.2f}, class={w['class']:.2f}, barrier={w['barrier']:.2f}, " f"pace={w['pace']:.2f}, fitness={w['fitness']:.2f}, weight={w['weight']:.2f}") else: print("No races found for this distance class") # Combined evaluation print("\n" + "=" * 60) print("COMBINED EVALUATION") print("=" * 60) # Use best weights per distance class profiles = { "sprint": { "form": 0.15, "class": 0.10, "distance": 0.06, "track": 0.06, "track_distance": 0.08, "condition": 0.04, "jockey": 0.04, "fitness": 0.04, "barrier": 0.18, "weight": 0.08, "pace": 0.12, }, "middle": { "form": 0.20, "class": 0.13, "distance": 0.08, "track": 0.08, "track_distance": 0.08, "condition": 0.05, "jockey": 0.06, "fitness": 0.05, "barrier": 0.13, "weight": 0.05, "pace": 0.04, }, "staying": { "form": 0.14, "class": 0.14, "distance": 0.10, "track": 0.06, "track_distance": 0.08, "condition": 0.06, "jockey": 0.08, "fitness": 0.10, "barrier": 0.04, "weight": 0.06, "pace": 0.04, }, } # Evaluate with distance-specific profiles agg = {"total": 0, "top1": 0} for date in dates: result_lookup = load_results(date) races = load_cached_races(date) for race_data in races: track = race_data.get('track', '') race_num = race_data.get('raceNumber', 0) actual = result_lookup.get((track, race_num)) if not actual: continue dist_class = classify_distance(race_data.get('distance', '1200m')) weights = profiles[dist_class] pred = score_race(race_data, weights) if not pred: continue agg["total"] += 1 if pred[0] == actual[0]: agg["top1"] += 1 if agg["total"] > 0: print(f"\nCombined: {agg['top1']}/{agg['total']} ({agg['top1']/agg['total']*100:.1f}%)") return profiles if __name__ == "__main__": main()