Download ml/features.py from raghava4u/Trading-Bot-M20: direct link, hf CLI and curl.
- Browser
- Download file 10.4 kB
-
https://huggingface.co/raghava4u/Trading-Bot-M20/resolve/main/ml/features.py
- Command line
-
hf download hf://raghava4u/Trading-Bot-M20/ml/features.py
-
curl -L -o features.py https://huggingface.co/raghava4u/Trading-Bot-M20/resolve/main/ml/features.py
10.4 kB
| """ | |
| ml/features.py β Multi-timeframe feature engineering for ML models. | |
| Produces a flat feature vector per bar from multiple timeframes. | |
| All features are causal (no lookahead) and normalised for tree models. | |
| """ | |
| from __future__ import annotations | |
| import numpy as np | |
| import pandas as pd | |
| from signals.technical import ( | |
| compute_rsi, compute_macd, compute_bollinger_bands, | |
| compute_vwap, compute_atr, compute_ema, | |
| ) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # SINGLE-TIMEFRAME FEATURES | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def _safe_pct(a: pd.Series, b: pd.Series) -> pd.Series: | |
| """(a - b) / b, with NaN-safe division.""" | |
| return (a - b) / b.replace(0, np.nan) | |
| def compute_bar_features(df: pd.DataFrame, prefix: str = "") -> pd.DataFrame: | |
| """Compute technical features from OHLCV bars. | |
| Returns DataFrame with ~30 features, aligned to df.index. | |
| All features use only past data (no lookahead). | |
| """ | |
| close = df["close"].astype(float) | |
| high = df["high"].astype(float) | |
| low = df["low"].astype(float) | |
| volume = df["volume"].astype(float) | |
| feats = pd.DataFrame(index=df.index) | |
| p = prefix | |
| # ββ Returns (multiple lookbacks) ββ | |
| for lb in [1, 3, 5, 10, 20]: | |
| feats[f"{p}ret_{lb}"] = close.pct_change(lb) | |
| # ββ Volatility ββ | |
| atr = compute_atr(df, 14) | |
| feats[f"{p}atr_pct"] = atr / close # ATR as % of price | |
| feats[f"{p}atr_ratio"] = atr / atr.rolling(50).mean() # current vs avg ATR | |
| feats[f"{p}realised_vol_20"] = close.pct_change().rolling(20).std() | |
| # ββ RSI ββ | |
| rsi = compute_rsi(close, 14) | |
| feats[f"{p}rsi"] = rsi / 100.0 # normalise to [0, 1] | |
| feats[f"{p}rsi_slope"] = (rsi - rsi.shift(3)) / 100.0 | |
| # ββ MACD ββ | |
| macd_line, macd_signal, macd_hist = compute_macd(close) | |
| feats[f"{p}macd_hist_norm"] = macd_hist / close # normalised | |
| feats[f"{p}macd_crossover"] = np.sign(macd_hist) - np.sign(macd_hist.shift(1)) | |
| # ββ Bollinger Bands ββ | |
| bb_upper, bb_middle, bb_lower = compute_bollinger_bands(close) | |
| bb_range = (bb_upper - bb_lower).replace(0, np.nan) | |
| feats[f"{p}bb_pct_b"] = (close - bb_lower) / bb_range # %B | |
| feats[f"{p}bb_width"] = bb_range / bb_middle # bandwidth | |
| # ββ VWAP distance (skip if not intraday) ββ | |
| try: | |
| vwap = compute_vwap(df) | |
| feats[f"{p}vwap_dist"] = _safe_pct(close, vwap) | |
| except Exception: | |
| feats[f"{p}vwap_dist"] = 0.0 | |
| # ββ EMAs ββ | |
| ema20 = compute_ema(close, 20) | |
| ema50 = compute_ema(close, 50) | |
| feats[f"{p}ema20_dist"] = _safe_pct(close, ema20) | |
| feats[f"{p}ema50_dist"] = _safe_pct(close, ema50) | |
| feats[f"{p}ema_spread"] = _safe_pct(ema20, ema50) | |
| # ββ Volume features ββ | |
| vol_ma = volume.rolling(50).mean() | |
| feats[f"{p}vol_ratio"] = volume / vol_ma.replace(0, np.nan) | |
| feats[f"{p}vol_trend"] = (volume.rolling(5).mean() | |
| / volume.rolling(20).mean().replace(0, np.nan)) | |
| # ββ Candle patterns (normalised) ββ | |
| bar_range = (high - low).replace(0, np.nan) | |
| feats[f"{p}body_pct"] = (close - df["open"].astype(float)) / bar_range | |
| feats[f"{p}upper_wick"] = (high - close.clip(upper=high)) / bar_range | |
| feats[f"{p}lower_wick"] = (close.clip(lower=low) - low) / bar_range | |
| return feats | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # DAILY REGIME FEATURES (for regime classifier) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def compute_regime_features(daily_df: pd.DataFrame) -> pd.DataFrame: | |
| """Compute features specifically designed for regime classification. | |
| Returns DataFrame indexed by daily_df.index with ~20 features. | |
| """ | |
| close = daily_df["close"].astype(float) | |
| high = daily_df["high"].astype(float) | |
| low = daily_df["low"].astype(float) | |
| volume = daily_df["volume"].astype(float) | |
| feats = pd.DataFrame(index=daily_df.index) | |
| # ββ Trend features ββ | |
| for w in [5, 10, 20, 50]: | |
| feats[f"ret_{w}d"] = close.pct_change(w) | |
| ema = compute_ema(close, w) | |
| feats[f"ema{w}_dist"] = _safe_pct(close, ema) | |
| # ββ Trend slope (linear regression slope over window) ββ | |
| for w in [10, 20]: | |
| x = np.arange(w, dtype=float) | |
| x_mean = x.mean() | |
| x_var = ((x - x_mean) ** 2).sum() | |
| slopes = close.rolling(w).apply( | |
| lambda y: np.sum((x - x_mean) * (y - y.mean())) / x_var | |
| if len(y) == w else 0.0, raw=True | |
| ) | |
| feats[f"slope_{w}d"] = slopes / close # normalised | |
| # ββ Volatility ββ | |
| atr = compute_atr(daily_df, 14) | |
| feats["atr_pct"] = atr / close | |
| feats["vol_20d"] = close.pct_change().rolling(20).std() | |
| feats["vol_ratio"] = (close.pct_change().rolling(5).std() | |
| / close.pct_change().rolling(20).std().replace(0, np.nan)) | |
| # ββ Momentum ββ | |
| rsi = compute_rsi(close, 14) | |
| feats["rsi"] = rsi / 100.0 | |
| _, _, macd_hist = compute_macd(close) | |
| feats["macd_hist_norm"] = macd_hist / close | |
| # ββ Volume ββ | |
| vol_ma = volume.rolling(20).mean() | |
| feats["vol_ratio_20d"] = volume / vol_ma.replace(0, np.nan) | |
| # ββ Range ββ | |
| feats["daily_range_pct"] = (high - low) / close | |
| return feats | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # MULTI-TIMEFRAME FEATURE MERGE | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def merge_multi_tf_features( | |
| intraday_df: pd.DataFrame, | |
| daily_df: pd.DataFrame | None = None, | |
| hourly_df: pd.DataFrame | None = None, | |
| ) -> pd.DataFrame: | |
| """Build feature matrix by computing features on each timeframe | |
| and forward-filling higher-TF features onto intraday bars. | |
| Returns: DataFrame indexed like intraday_df with all features. | |
| """ | |
| # Intraday features | |
| feats = compute_bar_features(intraday_df, prefix="intra_") | |
| # Daily features β forward-fill onto intraday | |
| if daily_df is not None and len(daily_df) >= 50: | |
| daily_feats = compute_bar_features(daily_df, prefix="daily_") | |
| # Map daily features to intraday by date | |
| if hasattr(daily_feats.index, 'date'): | |
| daily_feats.index = daily_feats.index.date | |
| daily_feats = daily_feats[~daily_feats.index.duplicated(keep='last')] | |
| intra_dates = (intraday_df.index.date if hasattr(intraday_df.index, 'date') | |
| else pd.to_datetime(intraday_df.index).date) | |
| for col in daily_feats.columns: | |
| feats[col] = daily_feats[col].reindex(intra_dates).values | |
| # Hourly features β forward-fill onto intraday | |
| if hourly_df is not None and len(hourly_df) >= 50: | |
| hourly_feats = compute_bar_features(hourly_df, prefix="hourly_") | |
| # Reindex to intraday via asof merge (latest hourly bar <= intraday time) | |
| hourly_feats = hourly_feats.sort_index() | |
| feats_sorted = feats.sort_index() | |
| for col in hourly_feats.columns: | |
| merged = pd.merge_asof( | |
| feats_sorted[[feats_sorted.columns[0]]], | |
| hourly_feats[[col]], | |
| left_index=True, right_index=True, | |
| direction="backward", | |
| ) | |
| feats[col] = merged[col].values | |
| return feats | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # LABEL GENERATION (for supervised learning) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def make_return_labels( | |
| df: pd.DataFrame, | |
| horizon: int = 6, | |
| threshold_pct: float = 0.15, | |
| ) -> pd.Series: | |
| """Create classification labels from forward returns. | |
| Args: | |
| horizon: bars to look forward (6 bars = 30min on 5m TF) | |
| threshold_pct: minimum move to count as UP/DOWN (in %) | |
| Returns: Series of {1 = up, -1 = down, 0 = flat} aligned to df.index. | |
| """ | |
| close = df["close"].astype(float) | |
| fwd_ret = close.shift(-horizon) / close - 1.0 | |
| fwd_pct = fwd_ret * 100 | |
| labels = pd.Series(0, index=df.index, dtype=int) | |
| labels[fwd_pct > threshold_pct] = 1 | |
| labels[fwd_pct < -threshold_pct] = -1 | |
| return labels | |
| def make_regime_labels(daily_df: pd.DataFrame, window: int = 20) -> pd.Series: | |
| """Create regime labels for daily bars (0=SIDEWAYS, 1=BULL, 2=BEAR). | |
| Uses same logic as rule-based detector for ground truth. | |
| """ | |
| close = daily_df["close"].astype(float) | |
| rolling_ret = close.pct_change(window).fillna(0) * 100 | |
| labels = pd.Series(0, index=daily_df.index, dtype=int) # SIDEWAYS | |
| labels[rolling_ret > 2.0] = 1 # BULL | |
| labels[rolling_ret < -2.0] = 2 # BEAR | |
| return labels | |
| REGIME_INT_TO_STR = {0: "SIDEWAYS", 1: "BULL", 2: "BEAR"} | |
| REGIME_STR_TO_INT = {"SIDEWAYS": 0, "BULL": 1, "BEAR": 2} | |