File size: 4,321 Bytes
9831ced
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
import numpy as np
import pandas as pd
from abc import ABC, abstractmethod
import numbers

def validate_history_and_horizon(history, horizon: int, model_name: str) -> None:
    if isinstance(horizon, bool) or not isinstance(horizon, (int, np.integer)) or horizon < 1:
        raise ValueError(f'{model_name}: horizon must be >= 1, got {horizon}.')
    values = np.asarray(pd.Series(history).astype(float), dtype=float)
    if values.ndim != 1 or values.size == 0:
        raise ValueError(f'{model_name}: price history must contain at least one candle.')
    if not np.all(np.isfinite(values)):
        raise ValueError(f'{model_name}: price history contains NaN/inf. This usually means a data glitch left a missing or malformed candle -- re-fetch the history and try again.')
    if (values <= 0).any():
        raise ValueError(f'{model_name}: price history must be strictly positive.')
    if isinstance(history, pd.Series) and isinstance(history.index, pd.DatetimeIndex):
        if history.index.isna().any() or history.index.duplicated().any() or not history.index.is_monotonic_increasing:
            raise ValueError(f'{model_name}: history timestamps must be ordered and unique.')


def standardize_exogenous(features: pd.DataFrame, model_name: str) -> pd.DataFrame:
    """Scale causal regressors using statistics from the supplied history only.

    OHLC prices and oscillator values have very different magnitudes. Passing
    them unscaled to statsmodels frequently creates singular matrices on short
    windows, especially for FX. Scaling here is causal (the future is never
    included) and keeps the same columns so the feature contract remains
    auditable.
    """
    if not isinstance(features, pd.DataFrame) or features.empty:
        raise ValueError(f'{model_name}: features must be a non-empty DataFrame.')
    numeric = features.apply(pd.to_numeric, errors='coerce').astype(float)
    if not np.isfinite(numeric.to_numpy()).all():
        raise ValueError(f'{model_name}: features contain non-finite values.')
    mean = numeric.mean(axis=0)
    scale = numeric.std(axis=0, ddof=0).replace(0.0, 1.0).fillna(1.0)
    scaled = (numeric - mean) / scale
    # A single outlier should not make the optimizer overflow. This is a
    # bounded transform, not an imputation or a fabricated observation.
    return scaled.clip(-12.0, 12.0)


def causal_exogenous(features, horizon, model_name):
    """Pair target at t with covariates at t-1, using training-only scaling.

    OHLC(t) and indicators(t) contain information about Close(t). They cannot
    be used as exogenous predictors of that same target. The last observed
    covariate is instead the legitimate first future regressor; later unknown
    covariates explicitly use persistence, not realized future candles.
    """
    if isinstance(horizon, (bool, np.bool_)) or not isinstance(horizon, numbers.Integral) or horizon < 1:
        raise ValueError(f'{model_name}: horizon must be a positive integer.')
    if not isinstance(features, pd.DataFrame) or features.empty or features.columns.duplicated().any():
        raise ValueError(f'{model_name}: causal features require a non-empty DataFrame with unique columns.')
    numeric = features.apply(pd.to_numeric, errors='coerce').astype(float).reset_index(drop=True)
    if len(numeric) < 2 or not np.isfinite(numeric.to_numpy()).all():
        raise ValueError(f'{model_name}: causal features need at least two finite rows.')
    train = numeric.iloc[:-1].reset_index(drop=True)
    mean = train.mean()
    scale = train.std(ddof=0).replace(0, 1).fillna(1)
    # Constant columns cannot identify a regression coefficient and may conflict
    # with the ARIMA trend. Their future value cannot affect a learned fit.
    columns = train.columns[train.nunique() > 1]
    if len(columns) == 0:
        return None, None
    scaled = ((train[columns] - mean[columns]) / scale[columns]).clip(-12, 12)
    future_row = ((numeric.iloc[[-1]][columns] - mean[columns]) / scale[columns]).clip(-12, 12)
    future = pd.concat([future_row] * int(horizon), ignore_index=True)
    return scaled, future

class BaseForecastModel(ABC):
    name = 'base'

    @abstractmethod
    def predict(self, history: pd.Series, horizon: int=1, features: pd.DataFrame=None) -> list:
        raise NotImplementedError