Download features.py from 3VVM/Lk: direct link, hf CLI and curl.
- Browser
- Download file 8.59 kB
-
https://huggingface.co/spaces/3VVM/Lk/resolve/main/features.py
- Command line
-
hf download hf://spaces/3VVM/Lk/features.py
-
curl -L -o features.py https://huggingface.co/spaces/3VVM/Lk/resolve/main/features.py
8.59 kB
| """ | |
| The ten standardized causal indicator families (spec sections 19-20, 22) | |
| plus warm-up handling (section 24). Every computation here uses only | |
| rolling/expanding-style pandas ops seeded from row <=t; nothing looks | |
| forward. This is verified directly in tests/test_features_and_leakage.py | |
| by re-computing on a truncated frame and checking the value is unchanged. | |
| """ | |
| from __future__ import annotations | |
| import numpy as np | |
| import pandas as pd | |
| import config as cfg | |
| def _rsi(close: pd.Series, period: int) -> pd.Series: | |
| delta = close.diff() | |
| gain = delta.clip(lower=0) | |
| loss = -delta.clip(upper=0) | |
| avg_gain = gain.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() | |
| avg_loss = loss.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() | |
| rs = avg_gain / avg_loss.replace(0, np.nan) | |
| rsi = 100 - (100 / (1 + rs)) | |
| # avg_loss==0 (pure uptrend) -> RSI is 100, not NaN. Warm-up rows | |
| # (avg_loss itself NaN) fall through unchanged since NaN != 0 is True, | |
| # so they correctly stay NaN rather than being fabricated as 50 or 100. | |
| return rsi.where(avg_loss != 0, 100.0) | |
| def _macd(close: pd.Series, fast: int, slow: int, signal: int): | |
| ema_fast = close.ewm(span=fast, adjust=False, min_periods=fast).mean() | |
| ema_slow = close.ewm(span=slow, adjust=False, min_periods=slow).mean() | |
| macd = ema_fast - ema_slow | |
| macd_signal = macd.ewm(span=signal, adjust=False, min_periods=signal).mean() | |
| return macd, macd_signal, macd - macd_signal | |
| def _bollinger(close: pd.Series, period: int, k: float): | |
| mid = close.rolling(period, min_periods=period).mean() | |
| std = close.rolling(period, min_periods=period).std(ddof=0) | |
| return mid, mid + k * std, mid - k * std | |
| def _stochastic(high, low, close, k_period, d_period): | |
| lowest = low.rolling(k_period, min_periods=k_period).min() | |
| highest = high.rolling(k_period, min_periods=k_period).max() | |
| rng = highest - lowest | |
| k = 100 * (close - lowest) / rng.replace(0, np.nan) | |
| # rng==0 means price hasn't moved at all over the lookback window | |
| # (real, observed data -- e.g. a stale/illiquid stretch of identical | |
| # ticks, not warm-up). There's no range to place `close` within, so | |
| # %K is defined as the neutral midpoint (50) rather than left as a | |
| # raw 0/0 NaN. Warm-up rows are unaffected: rng itself is NaN there | |
| # (not 0), so `rng != 0` stays True and the NaN passes through -- | |
| # same convention as RSI's avg_loss==0 case above. | |
| k = k.where(rng != 0, 50.0) | |
| d = k.rolling(d_period, min_periods=d_period).mean() | |
| return k, d | |
| def _atr(high, low, close, period): | |
| prev_close = close.shift(1) | |
| tr = pd.concat([ | |
| high - low, | |
| (high - prev_close).abs(), | |
| (low - prev_close).abs(), | |
| ], axis=1).max(axis=1) | |
| return tr.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() | |
| def _adx(high, low, close, period): | |
| up_move = high.diff() | |
| down_move = -low.diff() | |
| plus_dm = np.where((up_move > down_move) & (up_move > 0), up_move, 0.0) | |
| minus_dm = np.where((down_move > up_move) & (down_move > 0), down_move, 0.0) | |
| atr = _atr(high, low, close, period) | |
| plus_dm_ewm = pd.Series(plus_dm, index=high.index).ewm( | |
| alpha=1 / period, adjust=False, min_periods=period).mean() | |
| minus_dm_ewm = pd.Series(minus_dm, index=high.index).ewm( | |
| alpha=1 / period, adjust=False, min_periods=period).mean() | |
| # atr==0 is a real, reachable state, not just theoretical: a long | |
| # enough run of identical high/low/close (illiquid or stale ticks -- | |
| # exactly the kind of flat stretch visible in live 1m forex data) | |
| # decays the ATR ewm to an exact float 0.0. +DM/-DM are necessarily | |
| # 0 too whenever atr==0 (no high/low movement to measure), so the | |
| # correct DI reading is 0 (no directional movement), not an | |
| # undefined 0/0. Warm-up rows stay NaN because atr itself is NaN | |
| # there (not 0), so `atr != 0` stays True and the NaN passes through. | |
| plus_di = (100 * plus_dm_ewm / atr.replace(0, np.nan)).where(atr != 0, 0.0) | |
| minus_di = (100 * minus_dm_ewm / atr.replace(0, np.nan)).where(atr != 0, 0.0) | |
| di_sum = plus_di + minus_di | |
| # Same logic one level up: di_sum==0 (both DIs flat-zero, or a | |
| # perfectly balanced up/down) means there's no dominant direction -- | |
| # dx is defined as 0 (no trend), not NaN. di_sum is NaN (not 0) | |
| # during genuine warm-up, so that case is still preserved as NaN. | |
| dx = (100 * (plus_di - minus_di).abs() / di_sum.replace(0, np.nan)).where(di_sum != 0, 0.0) | |
| return dx.ewm(alpha=1 / period, adjust=False, min_periods=period).mean() | |
| def _roc(close, period): | |
| return 100 * (close / close.shift(period) - 1) | |
| def _obv(close, volume): | |
| direction = np.sign(close.diff().fillna(0)) | |
| return (direction * volume.fillna(0)).cumsum() | |
| def compute_indicators(df: pd.DataFrame, ic: cfg.IndicatorConfig = None) -> pd.DataFrame: | |
| """df must have columns open/high/low/close/volume, ascending index. | |
| Returns df with indicator columns appended; warm-up rows are NaN | |
| (never forward-filled from the future — section 24).""" | |
| ic = ic or cfg.IndicatorConfig() | |
| out = df.copy() | |
| out["rsi"] = _rsi(out["close"], ic.rsi_period) | |
| out["macd"], out["macd_signal"], out["macd_hist"] = _macd( | |
| out["close"], ic.macd_fast, ic.macd_slow, ic.macd_signal) | |
| out["bb_mid"], out["bb_upper"], out["bb_lower"] = _bollinger(out["close"], ic.bb_period, ic.bb_std) | |
| out["sma_20"] = out["close"].rolling(ic.sma_periods[0], min_periods=ic.sma_periods[0]).mean() | |
| out["sma_50"] = out["close"].rolling(ic.sma_periods[1], min_periods=ic.sma_periods[1]).mean() | |
| out["ema_12"] = out["close"].ewm(span=ic.ema_periods[0], adjust=False, min_periods=ic.ema_periods[0]).mean() | |
| out["ema_26"] = out["close"].ewm(span=ic.ema_periods[1], adjust=False, min_periods=ic.ema_periods[1]).mean() | |
| out["stoch_k"], out["stoch_d"] = _stochastic(out["high"], out["low"], out["close"], ic.stoch_k, ic.stoch_d) | |
| out["atr"] = _atr(out["high"], out["low"], out["close"], ic.atr_period) | |
| out["adx"] = _adx(out["high"], out["low"], out["close"], ic.adx_period) | |
| out["roc"] = _roc(out["close"], ic.roc_period) | |
| out["obv"] = _obv(out["close"], out["volume"].fillna(0)) | |
| return out | |
| def warm_up_valid_from(df: pd.DataFrame) -> int: | |
| """First integer index where ALL configured feature columns are | |
| simultaneously non-NaN. Raises if that never happens (section 88: | |
| 'do NOT manufacture data'). | |
| On failure, the message names the slowest-to-warm-up column(s) and | |
| how many bars they actually need vs. how many are available -- | |
| diagnosed empirically from df itself (each column's real first | |
| non-NaN position), never from a hand-derived "expected" formula per | |
| indicator. A hand-derived number for something like MACD or ADX | |
| (ewm-of-an-ewm, with its own min_periods layered on top) is exactly | |
| the kind of guess this project's own spec section 6 (real execution | |
| over static assumption) argues against -- so this reads it back off | |
| the real computed columns instead (section 98: errors must say what | |
| failed, not just that something did).""" | |
| cols = [c for c in cfg.PAST_DYNAMIC_FEATURES if c in df.columns] | |
| valid = df[cols].notna().all(axis=1) | |
| if not bool(valid.any()): | |
| first_valid: dict = {} | |
| never_valid = [] | |
| for c in cols: | |
| nn = df[c].notna().to_numpy() | |
| if nn.any(): | |
| first_valid[c] = int(np.argmax(nn)) | |
| else: | |
| never_valid.append(c) | |
| if never_valid: | |
| blockers = sorted(never_valid) | |
| detail = ( | |
| f"never produced a single valid value anywhere in the " | |
| f"{len(df)} bar(s) available" | |
| ) | |
| else: | |
| worst = max(first_valid.values()) | |
| blockers = sorted(c for c, idx in first_valid.items() if idx == worst) | |
| detail = ( | |
| f"needs {worst + 1} bar(s) before its first valid value " | |
| f"(only {len(df)} available)" | |
| ) | |
| raise ValueError( | |
| f"No fully warmed-up row found — insufficient historical data " | |
| f"for this indicator configuration (spec section 88). Slowest " | |
| f"to warm up: {', '.join(blockers)} — {detail}. Widen the " | |
| f"History window (more bars) or switch to a coarser/finer " | |
| f"timeframe so these indicators have enough lookback before " | |
| f"the backtest/forecast starts." | |
| ) | |
| return int(np.argmax(valid.to_numpy())) | |