File size: 2,131 Bytes
8bf6b71
 
 
 
 
9473607
8bf6b71
 
9473607
 
8bf6b71
 
 
9473607
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8bf6b71
 
 
 
 
9473607
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
import numpy as np
import pandas as pd
from abc import ABC, abstractmethod

def validate_history_and_horizon(history, horizon: int, model_name: str) -> None:
    if isinstance(horizon, bool) or not isinstance(horizon, (int, np.integer)) or horizon < 1:
        raise ValueError(f'{model_name}: horizon must be >= 1, got {horizon}.')
    values = np.asarray(pd.Series(history).astype(float), dtype=float)
    if values.ndim != 1 or values.size == 0:
        raise ValueError(f'{model_name}: price history must contain at least one candle.')
    if not np.all(np.isfinite(values)):
        raise ValueError(f'{model_name}: price history contains NaN/inf. This usually means a data glitch left a missing or malformed candle -- re-fetch the history and try again.')


def standardize_exogenous(features: pd.DataFrame, model_name: str) -> pd.DataFrame:
    """Scale causal regressors using statistics from the supplied history only.

    OHLC prices and oscillator values have very different magnitudes. Passing
    them unscaled to statsmodels frequently creates singular matrices on short
    windows, especially for FX. Scaling here is causal (the future is never
    included) and keeps the same columns so the feature contract remains
    auditable.
    """
    if not isinstance(features, pd.DataFrame) or features.empty:
        raise ValueError(f'{model_name}: features must be a non-empty DataFrame.')
    numeric = features.apply(pd.to_numeric, errors='coerce').astype(float)
    if not np.isfinite(numeric.to_numpy()).all():
        raise ValueError(f'{model_name}: features contain non-finite values.')
    mean = numeric.mean(axis=0)
    scale = numeric.std(axis=0, ddof=0).replace(0.0, 1.0).fillna(1.0)
    scaled = (numeric - mean) / scale
    # A single outlier should not make the optimizer overflow. This is a
    # bounded transform, not an imputation or a fabricated observation.
    return scaled.clip(-12.0, 12.0)

class BaseForecastModel(ABC):
    name = 'base'

    @abstractmethod
    def predict(self, history: pd.Series, horizon: int=1, features: pd.DataFrame=None) -> list:
        raise NotImplementedError