MYTHOSLIVE / backtest /backtester.py
3VVM's picture
Fix evaluation identities, model failures, causal execution and reproducibility
b678b3f verified
Raw History Blame Contribute Delete
31.9 kB
import json
import traceback
import numpy as np
import pandas as pd
from backtest.evaluation import evaluation_identity, forecast_metrics, new_run_id, score_forecasts
from data.validation import validate_ohlcv
from features.feature_pipeline import slice_features
from risk_management import (
CostModel, TradeLevels, average_true_range, build_trade_levels, simulate_trade,
summarize_trades,
)
from utils.helpers import MINUTES_PER_CANDLE, forex_market_open, infer_step, price_direction
def backtest_model(
df: pd.DataFrame,
model,
window: int = 100,
horizon: int = 1,
step: int = 1,
features_df: pd.DataFrame = None,
max_evaluations: int = None,
timeframe: str = None,
market: str = 'Crypto',
symbol: str = '',
costs: CostModel = None,
origin_start: int = None,
origin_end: int = None,
minimum_edge_atr: float = 0.20,
minimum_risk_reward: float = 1.20,
atr_stop_multiplier: float = 1.25,
prevent_overlap: bool = True,
run_id: str = None,
model_id: str = None,
phase: str = 'UNSPLIT',
initialization_error: Exception = None,
) -> pd.DataFrame:
if not isinstance(df, pd.DataFrame):
raise TypeError('Backtest candles must be provided as a pandas DataFrame.')
if any(isinstance(value, (bool, np.bool_)) or not isinstance(value, (int, np.integer)) or value < 1 for value in (window, horizon, step)):
raise ValueError('Window, horizon, and step must be positive integers.')
if max_evaluations is not None and (isinstance(max_evaluations, (bool, np.bool_)) or not isinstance(max_evaluations, (int, np.integer)) or max_evaluations < 1):
raise ValueError('max_evaluations must be a positive integer.')
for name, value in (('origin_start', origin_start), ('origin_end', origin_end)):
if value is not None and (isinstance(value, (bool, np.bool_)) or not isinstance(value, (int, np.integer)) or value < 0):
raise ValueError(f'{name} must be a non-negative integer when supplied.')
if features_df is not None and len(features_df) != len(df):
raise ValueError('Features must align with every OHLCV row.')
if features_df is not None and not isinstance(features_df, pd.DataFrame):
raise TypeError('features_df must be a pandas DataFrame when supplied.')
if features_df is not None and isinstance(features_df.index, pd.DatetimeIndex) and not features_df.index.equals(df.index):
raise ValueError('Feature timestamps do not match OHLCV timestamps.')
if timeframe is not None and timeframe not in MINUTES_PER_CANDLE:
raise ValueError(f'Unknown timeframe {timeframe!r}.')
if df.empty or len(df) < window + horizon or not df.index.is_monotonic_increasing or df.index.duplicated().any():
raise ValueError('Backtest candles must be non-empty, chronologically ordered, and unique.')
required = {'Open', 'High', 'Low', 'Close'}
if not required.issubset(df.columns):
raise ValueError(f'Backtest candles are missing columns: {sorted(required - set(df.columns))}.')
ohlc = df[['Open', 'High', 'Low', 'Close']].apply(pd.to_numeric, errors='coerce')
if (not np.isfinite(ohlc.to_numpy(dtype=float)).all() or (ohlc <= 0).any().any() or
(ohlc['High'] < ohlc[['Open', 'Close', 'Low']].max(axis=1)).any() or
(ohlc['Low'] > ohlc[['Open', 'Close', 'High']].min(axis=1)).any()):
raise ValueError('Backtest candles contain invalid/non-finite OHLC observations.')
df = validate_ohlcv(df)
if not np.isfinite(pd.to_numeric(df['Close'], errors='coerce').to_numpy(dtype=float)).all():
raise ValueError('Backtest Close prices must be finite real observations.')
close = df["Close"]
if costs is None:
costs = CostModel()
elif not isinstance(costs, CostModel):
raise TypeError('costs must be a CostModel instance.')
timestamps = [
ts.isoformat() if hasattr(ts, "isoformat") else ts for ts in df.index
]
last_i = len(close) - horizon
rows = []
run_id = run_id or new_run_id()
model_id = model_id or getattr(model, 'name', type(model).__name__)
expected_step = pd.Timedelta(minutes=MINUTES_PER_CANDLE[timeframe]) if timeframe else infer_step(df.index)
positions = list(range(window, last_i + 1, step))
if origin_start is not None:
positions = [position for position in positions if position >= origin_start]
if origin_end is not None:
positions = [position for position in positions if position < origin_end]
if max_evaluations is not None:
if max_evaluations < 1:
raise ValueError('max_evaluations must be positive.')
positions = positions[-max_evaluations:]
next_entry_index = window
for i in positions:
history = close.iloc[i - window : i].copy(deep=True)
feature_window = (
slice_features(features_df, i - window, i)
if features_df is not None
else None
)
current_close = float(close.iloc[i - 1])
target_idx = i + horizon - 1
actual_next_close = float(close.iloc[target_idx])
origin = (df.index[i - 1] + expected_step).isoformat()
context = {
'run_id': run_id, 'Model': model_id, 'Phase': phase,
'evaluation_id': evaluation_identity(run_id, model_id, phase, origin, horizon),
'evaluation_origin': origin, 'origin_index': i,
'origin_candle_open_utc': timestamps[i - 1],
'training_start_utc': timestamps[i - window], 'forecast_horizon': horizon,
'entry_timestamp': timestamps[i], 'target_timestamp': timestamps[target_idx],
}
continuous = bool((df.index[i - 1:target_idx + 1].to_series().diff().dropna() == expected_step).all())
continuous_history = bool((history.index.to_series().diff().dropna() == expected_step).all())
evaluation_status = 'FAILED'
if hasattr(model, 'last_diagnostics'):
model.last_diagnostics = {}
try:
if not continuous or not continuous_history:
evaluation_status = 'GAP_SKIPPED'
raise ValueError('Target or training window crosses a missing candle or market closure; no candles were filled.')
if market=='Forex' and not all(forex_market_open(timestamp) for timestamp in df.index[i:target_idx+1]):
evaluation_status = 'MARKET_CLOSED'
raise ValueError('Target includes FX market-closed candles; execution is blocked.')
if feature_window is not None and not np.isfinite(feature_window.apply(pd.to_numeric, errors='coerce').to_numpy()).all():
evaluation_status = 'DATA_SKIPPED'
raise ValueError('Causal features are unavailable or still warming up at this origin.')
if initialization_error is not None:
raise initialization_error
forecast = model.predict(history, horizon=horizon, features=feature_window)
forecast_values = np.asarray(forecast, dtype=float)
if forecast_values.shape != (horizon,):
raise ValueError('Model returned an incorrect forecast horizon/shape.')
if not np.isfinite(forecast_values).all() or (forecast_values <= 0).any():
raise ValueError(
f"forecast contains a non-finite or non-positive value: {forecast!r}"
)
predicted_next_close = float(forecast_values[-1])
except Exception as e:
rows.append(
{
**context, 'evaluation_status': evaluation_status,
'continuous_history': continuous_history,
'error_type': type(e).__name__,
'error_traceback': ''.join(traceback.format_exception(e)),
'model_diagnostics': json.dumps(getattr(model, 'last_diagnostics', {}) or {}, default=str),
"datetime": timestamps[target_idx],
"current_close": current_close,
"predicted_close": np.nan,
"actual_close": actual_next_close,
"predicted_direction": None,
"actual_direction": None,
"correct": None,
"continuous_target": continuous,
"trade_taken": False,
"entry": np.nan, "stop_loss": np.nan, "take_profit": np.nan,
"risk_reward": np.nan, "atr": np.nan, "outcome": evaluation_status,
"exit_price": np.nan, "net_r": np.nan, "bars_held": 0,
"error": str(e),
}
)
continue
predicted_direction = price_direction(predicted_next_close, current_close)
actual_direction = price_direction(actual_next_close, current_close)
correct = predicted_direction == actual_direction
trade_direction = {'up': 'BUY', 'down': 'SELL'}.get(predicted_direction)
trade_taken = False
try:
levels = build_trade_levels(
df.iloc[:i], trade_direction, market, symbol,
forecast_target=predicted_next_close,
# The signal is formed on close[i - 1]. The order is filled on
# the next candle's real open, while all levels still use only
# history through the closed signal candle.
entry_price=float(df['Open'].iloc[i]),
# Only trade a forecast that clears volatility noise and has a
# target the model actually predicted. Previously a tiny forecast
# was replaced by a distant swing target, creating timeout losses
# unrelated to the forecast signal.
forecast_only=True,
minimum_edge_atr=minimum_edge_atr,
minimum_risk_reward=minimum_risk_reward,
use_structure_stop=False,
atr_stop_multiplier=atr_stop_multiplier,
minimum_edge_bps=(
costs.spread_bps + costs.fee_bps + 2.0 * costs.slippage_bps
) * 1.25,
)
trade_taken = levels is not None and (not prevent_overlap or i >= next_entry_index)
trade = {'outcome': 'NO_LEVELS', 'exit_price': None, 'net_r': None, 'bars_held': 0}
if levels is not None and not trade_taken:
trade = {**trade, 'outcome': 'OVERLAP_SKIPPED'}
elif levels is not None:
trade = {**trade, **simulate_trade(levels, df.iloc[i:target_idx + 1], costs)}
if prevent_overlap:
next_entry_index = i + max(1, int(trade.get('bars_held', 0) or 0))
trade_error = None
except Exception as error:
trade_taken = False
levels = None
trade = {'outcome': 'TRADE_FAILED', 'exit_price': None, 'net_r': None, 'bars_held': 0}
trade_error = f'trade execution failed: {error}'
rows.append(
{
**context, 'evaluation_status': 'SUCCESS',
'continuous_history': continuous_history,
'model_diagnostics': json.dumps(getattr(model, 'last_diagnostics', {}) or {}, default=str),
'forecast_path': json.dumps(forecast_values.tolist()),
"datetime": timestamps[target_idx],
"current_close": current_close,
"predicted_close": predicted_next_close,
"actual_close": actual_next_close,
"predicted_direction": predicted_direction,
"actual_direction": actual_direction,
"correct": correct,
"continuous_target": continuous,
"trade_taken": trade_taken,
"entry": levels.entry if levels and trade_taken else np.nan,
"stop_loss": levels.stop_loss if levels and trade_taken else np.nan,
"take_profit": levels.take_profit if levels and trade_taken else np.nan,
"risk_reward": levels.risk_reward if levels and trade_taken else np.nan,
"atr": levels.atr if levels and trade_taken else np.nan,
"swing_support": levels.swing_support if levels and trade_taken else np.nan,
"swing_resistance": levels.swing_resistance if levels and trade_taken else np.nan,
"outcome": trade.get('outcome'),
"exit_price": trade.get('exit_price'),
"net_r": trade.get('net_r'),
"bars_held": trade.get('bars_held', 0),
'exit_timestamp': trade.get('exit_timestamp'),
'entry_fill': trade.get('entry_fill'), 'exit_fill': trade.get('exit_fill'),
'fees_per_unit': trade.get('fees_per_unit'),
'gross_pnl_per_unit': trade.get('gross_pnl_per_unit'),
'risk_per_unit': levels.risk_per_unit if levels and trade_taken else None,
'reward_per_unit': levels.reward_per_unit if levels and trade_taken else None,
'error_type': 'TradeExecutionError' if trade_error else None,
"error": trade_error,
}
)
return score_forecasts(pd.DataFrame(rows))
def backtest_random_signals(
df: pd.DataFrame,
window: int = 100,
horizon: int = 1,
step: int = 1,
max_evaluations: int = None,
timeframe: str = None,
market: str = 'Crypto',
symbol: str = '',
costs: CostModel = None,
seed: int = 42,
origin_start: int = None,
origin_end: int = None,
) -> pd.DataFrame:
"""Run a deterministic random-direction control using only causal levels.
The direction is random and independently reproducible for each origin.
Its target is two historical ATRs from the signal close, so the control
has the same conservative level construction and explicit costs as model
trades without using a realized future price.
"""
if not isinstance(df, pd.DataFrame):
raise TypeError('Backtest candles must be provided as a pandas DataFrame.')
if any(isinstance(value, (bool, np.bool_)) or not isinstance(value, (int, np.integer)) or value < 1 for value in (window, horizon, step)):
raise ValueError('Window, horizon, and step must be positive integers.')
if max_evaluations is not None and (isinstance(max_evaluations, (bool, np.bool_)) or not isinstance(max_evaluations, (int, np.integer)) or max_evaluations < 1):
raise ValueError('max_evaluations must be a positive integer.')
if timeframe is not None and timeframe not in MINUTES_PER_CANDLE:
raise ValueError(f'Unknown timeframe {timeframe!r}.')
for name, value in (('origin_start', origin_start), ('origin_end', origin_end)):
if value is not None and (isinstance(value, (bool, np.bool_)) or not isinstance(value, (int, np.integer)) or value < 0):
raise ValueError(f'{name} must be a non-negative integer when supplied.')
if df.empty or len(df) < window + horizon or not df.index.is_monotonic_increasing or df.index.duplicated().any():
raise ValueError('Backtest candles must be non-empty, chronologically ordered, and unique.')
required = {'Open', 'High', 'Low', 'Close'}
if not required.issubset(df.columns):
raise ValueError(f'Backtest candles are missing columns: {sorted(required - set(df.columns))}.')
ohlc = df[['Open', 'High', 'Low', 'Close']].apply(pd.to_numeric, errors='coerce')
if (not np.isfinite(ohlc.to_numpy(dtype=float)).all() or (ohlc <= 0).any().any() or
(ohlc['High'] < ohlc[['Open', 'Close', 'Low']].max(axis=1)).any() or
(ohlc['Low'] > ohlc[['Open', 'Close', 'High']].min(axis=1)).any()):
raise ValueError('Backtest candles contain invalid/non-finite OHLC observations.')
if costs is None:
costs = CostModel()
elif not isinstance(costs, CostModel):
raise TypeError('costs must be a CostModel instance.')
df = df.copy()
df[['Open', 'High', 'Low', 'Close']] = ohlc.astype(float)
close = df['Close']
timestamps = [ts.isoformat() if hasattr(ts, 'isoformat') else ts for ts in df.index]
last_i = len(close) - horizon
positions = list(range(window, last_i + 1, step))
if origin_start is not None:
positions = [position for position in positions if position >= origin_start]
if origin_end is not None:
positions = [position for position in positions if position < origin_end]
if max_evaluations is not None:
positions = positions[-max_evaluations:]
rows = []
for i in positions:
current_close = float(close.iloc[i - 1])
target_idx = i + horizon - 1
actual_next_close = float(close.iloc[target_idx])
continuous = True
try:
if timeframe:
expected_step = pd.Timedelta(minutes=MINUTES_PER_CANDLE[timeframe])
target_times = df.index[i - 1:target_idx + 1]
continuous = bool((target_times.to_series().diff().dropna() == expected_step).all())
if not continuous:
raise ValueError('Target crosses a missing candle or market closure; not a continuous timeframe horizon.')
atr = average_true_range(df.iloc[:i][['High', 'Low', 'Close']])
if not np.isfinite(atr) or atr <= 0:
raise ValueError('Causal ATR is unavailable for the random control.')
direction = 'BUY' if np.random.default_rng(int(seed) + i * 1009).integers(0, 2) == 1 else 'SELL'
sign = 1.0 if direction == 'BUY' else -1.0
random_target = current_close + sign * 2.0 * float(atr)
levels = build_trade_levels(
df.iloc[:i], direction, market, symbol,
forecast_target=random_target,
entry_price=float(df['Open'].iloc[i]),
forecast_only=True,
minimum_edge_atr=0.20,
minimum_risk_reward=1.20,
use_structure_stop=False,
minimum_edge_bps=(
costs.spread_bps + costs.fee_bps + 2.0 * costs.slippage_bps
) * 1.25,
)
if levels is None:
raise ValueError('Causal random-control levels did not clear the execution gate.')
trade = simulate_trade(levels, df.iloc[i:target_idx + 1], costs)
error = None
except Exception as exc:
direction = None
random_target = np.nan
levels = None
trade = {'outcome': 'RANDOM_FAILED', 'exit_price': np.nan, 'net_r': np.nan, 'bars_held': 0}
error = str(exc)
actual_direction = price_direction(actual_next_close, current_close) if direction else None
predicted_direction = {'BUY': 'up', 'SELL': 'down'}.get(direction)
rows.append({
'datetime': timestamps[target_idx],
'current_close': current_close,
'predicted_close': random_target,
'actual_close': actual_next_close,
'predicted_direction': predicted_direction,
'actual_direction': actual_direction,
'correct': predicted_direction == actual_direction if predicted_direction else None,
'continuous_target': continuous,
'trade_taken': levels is not None,
'entry': levels.entry if levels else np.nan,
'stop_loss': levels.stop_loss if levels else np.nan,
'take_profit': levels.take_profit if levels else np.nan,
'risk_reward': levels.risk_reward if levels else np.nan,
'atr': levels.atr if levels else np.nan,
'outcome': trade.get('outcome'),
'exit_price': trade.get('exit_price'),
'net_r': trade.get('net_r'),
'bars_held': trade.get('bars_held', 0),
'error': error,
})
return pd.DataFrame(rows)
MATCHED_RANDOM_RUNS = 200
def _matched_levels(row, direction: str) -> TradeLevels | None:
"""Reuse a model trade's causal distances for a direction control.
The control never inspects a future price. It receives only the model's
already-eligible entry, stop/target distances, and the same future path
length used by the model trade. Random directions are then permuted while
preserving the model trade count and long/short ratio.
"""
try:
entry = float(row['entry'])
stop = float(row['stop_loss'])
target = float(row['take_profit'])
risk = abs(entry - stop)
reward = abs(target - entry)
atr = float(row.get('atr', 1.0))
support = float(row.get('swing_support', entry))
resistance = float(row.get('swing_resistance', entry))
except (TypeError, ValueError, KeyError):
return None
if any(not np.isfinite(value) or value <= 0 for value in (entry, risk, reward, atr, support, resistance)):
return None
if direction == 'BUY':
stop_loss = entry - risk
take_profit = entry + reward
elif direction == 'SELL':
stop_loss = entry + risk
take_profit = entry - reward
else:
return None
if stop_loss <= 0 or take_profit <= 0:
return None
return TradeLevels(
direction=direction,
entry=entry,
stop_loss=stop_loss,
take_profit=take_profit,
risk_per_unit=risk,
reward_per_unit=reward,
risk_reward=reward / risk,
atr=atr,
swing_support=support,
swing_resistance=resistance,
tick_size=1.0,
precision_digits=0,
level_basis='matched causal model-trade distances; control direction only',
)
def matched_direction_control(
model_results: pd.DataFrame,
df: pd.DataFrame,
horizon: int,
costs: CostModel,
*,
inverted: bool = False,
runs: int = MATCHED_RANDOM_RUNS,
seed: int = 42,
market: str = 'Crypto',
symbol: str = '',
minimum_edge_atr: float = .20,
minimum_risk_reward: float = 1.20,
atr_stop_multiplier: float = 1.25,
):
"""Independent direction controls with full exits and no overlap.
Directions preserve the original trade-side multiset. Random controls
execute at eligible causal model origins with their own positions/exits;
they never inherit a realized original holding duration.
"""
def no_model_trades(label, control_runs):
summary = summarize_trades(pd.DataFrame())
summary.update({
'model': label, 'matched_trade_count': 0,
'control_runs': int(control_runs), 'random_mean_net_profit_r': None,
'random_5pct_net_profit_r': None, 'random_95pct_net_profit_r': None,
'matched_runs': 0, 'status': 'NO_MODEL_TRADES',
})
return summary, pd.DataFrame()
if not isinstance(model_results, pd.DataFrame) or model_results.empty:
return no_model_trades('Inverted signals' if inverted else 'Random baseline', 1 if inverted else int(runs))
eligible = model_results[
model_results.get('trade_taken', False).astype(bool) &
model_results['net_r'].notna()
].copy()
eligible = eligible[eligible['predicted_direction'].isin(('up', 'down'))]
if eligible.empty:
return no_model_trades('Inverted signals' if inverted else 'Random baseline', 1 if inverted else int(runs))
if isinstance(runs, (bool, np.bool_)) or int(runs) < 1:
raise ValueError('Matched control runs must be a positive integer.')
runs = int(runs)
original = np.array(['BUY' if value == 'up' else 'SELL' for value in eligible['predicted_direction']], dtype=object)
table = {}
origins = []
expected_step = infer_step(df.index)
# Controls are matched to the model's *eligible* origins only. Building a
# table from every successful forecast and then walking that larger origin
# set allowed a random control to replace an ineligible model row with a
# different opportunity, breaking the promised trade-count/side match.
for _, source_row in eligible.iterrows():
timestamp=str(source_row['datetime'])
try:
target_idx=int(df.index.get_loc(pd.Timestamp(timestamp)))
except (KeyError,TypeError,ValueError):
continue
i=target_idx-int(horizon)+1
if i<1:
continue
origins.append(i); table[i]={}
move=abs(float(source_row['predicted_close'])-float(source_row['current_close']))
for direction in ('BUY','SELL'):
sign=1 if direction=='BUY' else -1
levels=build_trade_levels(df.iloc[:i],direction,market,symbol,
forecast_target=float(source_row['current_close'])+sign*move,
entry_price=float(df.Open.iloc[i]),forecast_only=True,
minimum_edge_atr=minimum_edge_atr,minimum_risk_reward=minimum_risk_reward,
use_structure_stop=False,atr_stop_multiplier=atr_stop_multiplier,
minimum_edge_bps=(costs.spread_bps+costs.fee_bps+2*costs.slippage_bps)*1.25)
if levels is not None:
trade=simulate_trade(levels,df.iloc[i:target_idx+1],costs)
table[i][direction]={**trade,'entry':levels.entry,'stop_loss':levels.stop_loss,
'take_profit':levels.take_profit,'datetime':timestamp,
'origin_index': i,
'evaluation_origin': (df.index[i - 1] + expected_step).isoformat(),
'target_timestamp': df.index[target_idx].isoformat(),
'forecast_horizon': int(horizon),
'current_close': float(df['Close'].iloc[i - 1]),
'predicted_close': float(source_row['current_close']) + sign * move}
control_summaries = []
all_rows = []
actual_runs = 1 if inverted else runs
for run_number in range(actual_runs):
if inverted:
assigned = np.array(['SELL' if value == 'BUY' else 'BUY' for value in original], dtype=object)
else:
rng = np.random.default_rng(int(seed) + run_number)
assigned = rng.permutation(original)
trade_rows = []
next_available=0
control_origins=origins
for assignment_index, start_idx in enumerate(control_origins):
if assignment_index >= len(assigned):
break
# Consume the side assignment for this origin even when the
# opposite-side causal levels are unavailable or overlap blocks
# the trade. Reusing assigned[len(trade_rows)] after a skip moved
# a later origin's direction onto the wrong evaluation.
direction=str(assigned[assignment_index])
if start_idx<next_available or direction not in table.get(start_idx,{}):
continue
trade=table[start_idx][direction]
trade_rows.append({
**trade,
'trade_taken': True,
'net_r': trade.get('net_r'),
'outcome': trade.get('outcome'),
'predicted_direction': 'up' if direction == 'BUY' else 'down',
'control_run': run_number + 1,
'error': None,
})
next_available=start_idx+int(trade['bars_held'])
details = pd.DataFrame(trade_rows)
control_summaries.append(summarize_trades(details))
all_rows.extend(trade_rows)
net_values = np.array([
float(summary['net_profit_r']) for summary in control_summaries
if summary.get('net_profit_r') is not None and np.isfinite(summary['net_profit_r'])
], dtype=float)
if not len(net_values):
net_values = np.array([np.nan])
aggregate = {
'model': 'Inverted signals' if inverted else 'Random baseline',
'aggregation': 'mean per independent control run; details contain every completed control trade',
'trades': float(np.mean([s['trades'] for s in control_summaries])),
'matched_trade_count': int(len(eligible)),
'control_runs': actual_runs,
'random_mean_net_profit_r': float(np.mean(net_values)) if np.isfinite(net_values).any() else None,
'random_5pct_net_profit_r': float(np.quantile(net_values, 0.05)) if np.isfinite(net_values).any() else None,
'random_95pct_net_profit_r': float(np.quantile(net_values, 0.95)) if np.isfinite(net_values).any() else None,
'matched_runs':sum(summary['trades']==len(original) for summary in control_summaries),
'status': 'OK' if all(summary['trades']==len(original) for summary in control_summaries) else 'INCONCLUSIVE_COUNT',
}
for key in ('wins', 'losses', 'breakeven', 'timeouts', 'tp_exits', 'sl_exits',
'win_rate_pct', 'profit_factor', 'expectancy_r', 'max_drawdown_r',
'average_r', 'net_profit_r', 'top3_removed_profit_r'):
values = [s[key] for s in control_summaries if s.get(key) is not None]
aggregate[key] = float(np.mean(values)) if values else None
return aggregate, pd.DataFrame(all_rows)
def summarize_backtest(results: pd.DataFrame, model_name: str) -> dict:
return {
"model": model_name,
**forecast_metrics(results),
**summarize_trades(results),
}
def run_all_models_backtest(
df: pd.DataFrame,
models: dict,
window: int = 100,
horizon: int = 1,
step: int = 1,
features_df: pd.DataFrame = None,
origin_start: int = None,
origin_end: int = None,
max_evaluations: int = None,
timeframe: str = None,
market: str = 'Crypto',
symbol: str = '',
costs: CostModel = None,
):
per_model_results = {}
per_model_summary = {}
for name, model in models.items():
results = backtest_model(
df, model, window=window, horizon=horizon, step=step, features_df=features_df,
origin_start=origin_start, origin_end=origin_end,
max_evaluations=max_evaluations, timeframe=timeframe,
market=market, symbol=symbol, costs=costs,
)
per_model_results[name] = results
per_model_summary[name] = summarize_backtest(results, name)
return per_model_results, per_model_summary
BASELINE_LABEL = "Close-only"
FEATURED_LABEL = "+14 Features"
def run_comparison_backtest(
df: pd.DataFrame,
models: dict,
features_df: pd.DataFrame,
window: int = 100,
horizon: int = 1,
step: int = 1,
origin_start: int = None,
origin_end: int = None,
max_evaluations: int = None,
timeframe: str = None,
market: str = 'Crypto',
symbol: str = '',
costs: CostModel = None,
):
per_model_results = {}
per_model_summary = {}
for name, model_pair in models.items():
baseline_model, featured_model = model_pair
results_list, summary_list = [], []
for label, model, feats in (
(BASELINE_LABEL, baseline_model, None),
(FEATURED_LABEL, featured_model, features_df),
):
results = backtest_model(
df, model, window=window, horizon=horizon, step=step, features_df=feats,
origin_start=origin_start, origin_end=origin_end,
max_evaluations=max_evaluations, timeframe=timeframe,
market=market, symbol=symbol, costs=costs,
)
summary = summarize_backtest(results, name)
summary["mode"] = label
results = results.copy()
results.insert(0, "mode", label)
results_list.append(results)
summary_list.append(summary)
per_model_results[name] = pd.concat(results_list, ignore_index=True)
per_model_summary[name] = summary_list
return per_model_results, per_model_summary