""" The ten standardized causal indicator families (spec sections 19-20, 22) plus warm-up handling (section 24). Every computation here uses only rolling/expanding-style pandas ops seeded from row <=t; nothing looks forward. This is verified directly in tests/test_features_and_leakage.py by re-computing on a truncated frame and checking the value is unchanged. """ from __future__ import annotations import numpy as np import pandas as pd import config as cfg def _rsi(close: pd.Series, period: int) -> pd.Series: delta = close.diff() gain = delta.clip(lower=0) loss = -delta.clip(upper=0) avg_gain = gain.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() avg_loss = loss.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() rs = avg_gain / avg_loss.replace(0, np.nan) rsi = 100 - (100 / (1 + rs)) # avg_loss==0 (pure uptrend) -> RSI is 100, not NaN. Warm-up rows # (avg_loss itself NaN) fall through unchanged since NaN != 0 is True, # so they correctly stay NaN rather than being fabricated as 50 or 100. return rsi.where(avg_loss != 0, 100.0) def _macd(close: pd.Series, fast: int, slow: int, signal: int): ema_fast = close.ewm(span=fast, adjust=False, min_periods=fast).mean() ema_slow = close.ewm(span=slow, adjust=False, min_periods=slow).mean() macd = ema_fast - ema_slow macd_signal = macd.ewm(span=signal, adjust=False, min_periods=signal).mean() return macd, macd_signal, macd - macd_signal def _bollinger(close: pd.Series, period: int, k: float): mid = close.rolling(period, min_periods=period).mean() std = close.rolling(period, min_periods=period).std(ddof=0) return mid, mid + k * std, mid - k * std def _stochastic(high, low, close, k_period, d_period): lowest = low.rolling(k_period, min_periods=k_period).min() highest = high.rolling(k_period, min_periods=k_period).max() rng = highest - lowest k = 100 * (close - lowest) / rng.replace(0, np.nan) # rng==0 means price hasn't moved at all over the lookback window # (real, observed data -- e.g. a stale/illiquid stretch of identical # ticks, not warm-up). There's no range to place `close` within, so # %K is defined as the neutral midpoint (50) rather than left as a # raw 0/0 NaN. Warm-up rows are unaffected: rng itself is NaN there # (not 0), so `rng != 0` stays True and the NaN passes through -- # same convention as RSI's avg_loss==0 case above. k = k.where(rng != 0, 50.0) d = k.rolling(d_period, min_periods=d_period).mean() return k, d def _atr(high, low, close, period): prev_close = close.shift(1) tr = pd.concat([ high - low, (high - prev_close).abs(), (low - prev_close).abs(), ], axis=1).max(axis=1) return tr.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() def _adx(high, low, close, period): up_move = high.diff() down_move = -low.diff() plus_dm = np.where((up_move > down_move) & (up_move > 0), up_move, 0.0) minus_dm = np.where((down_move > up_move) & (down_move > 0), down_move, 0.0) atr = _atr(high, low, close, period) plus_dm_ewm = pd.Series(plus_dm, index=high.index).ewm( alpha=1 / period, adjust=False, min_periods=period).mean() minus_dm_ewm = pd.Series(minus_dm, index=high.index).ewm( alpha=1 / period, adjust=False, min_periods=period).mean() # atr==0 is a real, reachable state, not just theoretical: a long # enough run of identical high/low/close (illiquid or stale ticks -- # exactly the kind of flat stretch visible in live 1m forex data) # decays the ATR ewm to an exact float 0.0. +DM/-DM are necessarily # 0 too whenever atr==0 (no high/low movement to measure), so the # correct DI reading is 0 (no directional movement), not an # undefined 0/0. Warm-up rows stay NaN because atr itself is NaN # there (not 0), so `atr != 0` stays True and the NaN passes through. plus_di = (100 * plus_dm_ewm / atr.replace(0, np.nan)).where(atr != 0, 0.0) minus_di = (100 * minus_dm_ewm / atr.replace(0, np.nan)).where(atr != 0, 0.0) di_sum = plus_di + minus_di # Same logic one level up: di_sum==0 (both DIs flat-zero, or a # perfectly balanced up/down) means there's no dominant direction -- # dx is defined as 0 (no trend), not NaN. di_sum is NaN (not 0) # during genuine warm-up, so that case is still preserved as NaN. dx = (100 * (plus_di - minus_di).abs() / di_sum.replace(0, np.nan)).where(di_sum != 0, 0.0) return dx.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() def _roc(close, period): return 100 * (close / close.shift(period) - 1) def _obv(close, volume): direction = np.sign(close.diff().fillna(0)) return (direction * volume.fillna(0)).cumsum() def compute_indicators(df: pd.DataFrame, ic: cfg.IndicatorConfig = None) -> pd.DataFrame: """df must have columns open/high/low/close/volume, ascending index. Returns df with indicator columns appended; warm-up rows are NaN (never forward-filled from the future — section 24).""" ic = ic or cfg.IndicatorConfig() out = df.copy() out["rsi"] = _rsi(out["close"], ic.rsi_period) out["macd"], out["macd_signal"], out["macd_hist"] = _macd( out["close"], ic.macd_fast, ic.macd_slow, ic.macd_signal) out["bb_mid"], out["bb_upper"], out["bb_lower"] = _bollinger(out["close"], ic.bb_period, ic.bb_std) out["sma_20"] = out["close"].rolling(ic.sma_periods[0], min_periods=ic.sma_periods[0]).mean() out["sma_50"] = out["close"].rolling(ic.sma_periods[1], min_periods=ic.sma_periods[1]).mean() out["ema_12"] = out["close"].ewm(span=ic.ema_periods[0], adjust=False, min_periods=ic.ema_periods[0]).mean() out["ema_26"] = out["close"].ewm(span=ic.ema_periods[1], adjust=False, min_periods=ic.ema_periods[1]).mean() out["stoch_k"], out["stoch_d"] = _stochastic(out["high"], out["low"], out["close"], ic.stoch_k, ic.stoch_d) out["atr"] = _atr(out["high"], out["low"], out["close"], ic.atr_period) out["adx"] = _adx(out["high"], out["low"], out["close"], ic.adx_period) out["roc"] = _roc(out["close"], ic.roc_period) out["obv"] = _obv(out["close"], out["volume"].fillna(0)) return out def warm_up_valid_from(df: pd.DataFrame) -> int: """First integer index where ALL configured feature columns are simultaneously non-NaN. Raises if that never happens (section 88: 'do NOT manufacture data'). On failure, the message names the slowest-to-warm-up column(s) and how many bars they actually need vs. how many are available -- diagnosed empirically from df itself (each column's real first non-NaN position), never from a hand-derived "expected" formula per indicator. A hand-derived number for something like MACD or ADX (ewm-of-an-ewm, with its own min_periods layered on top) is exactly the kind of guess this project's own spec section 6 (real execution over static assumption) argues against -- so this reads it back off the real computed columns instead (section 98: errors must say what failed, not just that something did).""" cols = [c for c in cfg.PAST_DYNAMIC_FEATURES if c in df.columns] valid = df[cols].notna().all(axis=1) if not bool(valid.any()): first_valid: dict = {} never_valid = [] for c in cols: nn = df[c].notna().to_numpy() if nn.any(): first_valid[c] = int(np.argmax(nn)) else: never_valid.append(c) if never_valid: blockers = sorted(never_valid) detail = ( f"never produced a single valid value anywhere in the " f"{len(df)} bar(s) available" ) else: worst = max(first_valid.values()) blockers = sorted(c for c, idx in first_valid.items() if idx == worst) detail = ( f"needs {worst + 1} bar(s) before its first valid value " f"(only {len(df)} available)" ) raise ValueError( f"No fully warmed-up row found — insufficient historical data " f"for this indicator configuration (spec section 88). Slowest " f"to warm up: {', '.join(blockers)} — {detail}. Widen the " f"History window (more bars) or switch to a coarser/finer " f"timeframe so these indicators have enough lookback before " f"the backtest/forecast starts." ) return int(np.argmax(valid.to_numpy()))