Spaces:
Configuration error
Configuration error
File size: 5,491 Bytes
e23172f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 | """Run backtesting and weight optimization."""
import json
import sys
from pathlib import Path
from collections import defaultdict
# Add project to path
sys.path.insert(0, '/home/brettanthonysjoberg179/trifecta-bro-hf-space')
from trifecta_bro.model.scoring import score_runner
from trifecta_bro.model.distance_profiles import get_profile, classify_distance
from trifecta_bro.model.pace_analysis import classify_pace
from trifecta_bro.data.models import RaceModel, RunnerModel
from trifecta_bro.evaluation.backtest import load_day_data, evaluate_day
def load_all_data(dates):
"""Load all predictions and results for multiple dates."""
all_data = {}
for date in dates:
predictions, results = load_day_data(date)
if predictions and results:
all_data[date] = {"predictions": predictions, "results": results}
return all_data
def run_backtest(dates, weight_profile, label=""):
"""Run backtest with given weights."""
total_races = 0
total_top1 = 0
total_exact = 0
day_stats = {}
for date in dates:
predictions, results = load_day_data(date)
if not predictions or not results:
continue
stats = evaluate_day(predictions, results)
day_stats[date] = stats
total_races += stats["total"]
total_top1 += stats["top1"]
total_exact += stats["exact"]
top1_pct = total_top1 / total_races * 100 if total_races > 0 else 0
exact_pct = total_exact / total_races * 100 if total_races > 0 else 0
return {
"label": label,
"total_races": total_races,
"total_top1": total_top1,
"total_exact": total_exact,
"top1_pct": top1_pct,
"exact_pct": exact_pct,
"day_stats": day_stats,
}
def grid_search_weights(dates, param_grid, base_weights, n_top=5):
"""Grid search over weight combinations."""
import itertools
param_names = list(param_grid.keys())
param_values = [param_grid[name] for name in param_names]
results = []
for combo in itertools.product(*param_values):
weights = dict(base_weights)
for name, value in zip(param_names, combo):
weights[name] = value
# Normalize
total_w = sum(weights.values())
weights = {k: v / total_w for k, v in weights.items()}
# Run backtest
bt = run_backtest(dates, weights)
bt["weights"] = weights
bt["params"] = dict(zip(param_names, combo))
results.append(bt)
results.sort(key=lambda x: x["top1_pct"], reverse=True)
return results[:n_top]
def main():
dates = ["2026-08-07", "2026-08-08", "2026-08-09", "2026-08-12", "2026-08-13", "2026-08-14"]
print("Loading data...")
all_data = load_all_data(dates)
print(f"Loaded data for {len(all_data)} dates: {list(all_data.keys())}")
# Current v2.0 weights
v2_weights = {
"form": 0.18, "class": 0.14, "distance": 0.10, "track": 0.10,
"track_distance": 0.10, "condition": 0.05, "jockey": 0.06,
"fitness": 0.08, "barrier": 0.08, "weight": 0.08, "pace": 0.03,
}
# Run backtest with current weights
print("\n" + "="*60)
print("CURRENT v2.0 WEIGHTS")
print("="*60)
bt_current = run_backtest(dates, v2_weights, "v2.0 current")
print(f"Races: {bt_current['total_races']}")
print(f"Top 1: {bt_current['total_top1']}/{bt_current['total_races']} ({bt_current['top1_pct']:.1f}%)")
print(f"Exact: {bt_current['total_exact']}/{bt_current['total_races']} ({bt_current['exact_pct']:.1f}%)")
# Grid search
print("\n" + "="*60)
print("GRID SEARCH OPTIMIZATION")
print("="*60)
param_grid = {
"form": [0.15, 0.20, 0.25, 0.30],
"class": [0.10, 0.15, 0.20],
"distance": [0.06, 0.10, 0.14],
"track": [0.06, 0.10, 0.14],
"track_distance": [0.06, 0.10, 0.14],
"fitness": [0.04, 0.08, 0.12],
"barrier": [0.06, 0.10, 0.14],
"weight": [0.04, 0.08, 0.12],
}
# Fix some params to reduce search space
fixed = {"condition": 0.05, "jockey": 0.06, "pace": 0.03}
top_results = grid_search_weights(dates, param_grid, {**v2_weights, **fixed}, n_top=10)
print("\nTop 10 weight combinations:")
for i, r in enumerate(top_results):
print(f"\n{i+1}. Top1: {r['top1_pct']:.1f}% | Exact: {r['exact_pct']:.1f}%")
print(f" form={r['params']['form']:.2f}, class={r['params']['class']:.2f}, "
f"distance={r['params']['distance']:.2f}, track={r['params']['track']:.2f}, "
f"track_dist={r['params']['track_distance']:.2f}, fitness={r['params']['fitness']:.2f}, "
f"barrier={r['params']['barrier']:.2f}, weight={r['params']['weight']:.2f}")
# Best weights
if top_results:
best = top_results[0]
print("\n" + "="*60)
print("OPTIMIZED WEIGHTS")
print("="*60)
print(f"Top 1: {best['top1_pct']:.1f}% (was {bt_current['top1_pct']:.1f}%)")
print(f"Exact: {best['exact_pct']:.1f}% (was {bt_current['exact_pct']:.1f}%)")
# Per-day breakdown
print("\nPer-day performance:")
for date, stats in sorted(best["day_stats"].items()):
print(f" {date}: {stats['total']} races, {stats['top1']} top1 ({stats['top1']/stats['total']*100:.0f}%)")
return top_results
if __name__ == "__main__":
main()
|