File size: 15,974 Bytes
b0d9ebf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8d1eb7b
b0d9ebf
 
 
 
 
 
 
 
 
 
8af7f2e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b0d9ebf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8d1eb7b
 
8af7f2e
b0d9ebf
 
 
 
 
 
 
8d1eb7b
 
 
 
 
 
b0d9ebf
8af7f2e
b0d9ebf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8af7f2e
 
b0d9ebf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8d1eb7b
 
b0d9ebf
 
 
 
8d1eb7b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b0d9ebf
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8af7f2e
 
8d1eb7b
 
b0d9ebf
8af7f2e
b0d9ebf
8d1eb7b
 
b0d9ebf
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
#!/usr/bin/env python3
"""research/ml_selection_backtest.py — "give it one day, can it pick winners?" backtest.

Answers the practical question: on a given day, if the ML model RANKS the universe and we
buy its top-N BULLISH picks, does the basket make money over the next 1 / 3 / 5 days —
net of costs, and does it beat just buying the whole market that day?

This is a stock-SELECTION / portfolio test (distinct from research/ml_backtest.py, which
measures per-prediction accuracy). It uses the trained 3D model to rank, then measures the
picks' REALIZED forward close-to-close returns straight from the OHLCV cache — including the
5-day horizon the model isn't trained on, so we can honestly see the 5-day outcome.

For each out-of-sample date:
  1. Predict every ticker with a row that day (batch), keep BULLISH picks.
  2. Rank by expected up-move (up_q50), tie-break by confidence; take top-N.
  3. Realized fwd return at 1/3/5 trading days for each pick, minus 0.30% round-trip cost.
  4. Basket return = equal-weight mean; baseline = equal-weight ALL stocks that day (market).

Usage:
    python research/ml_selection_backtest.py                 # top-10, all holdout dates
    python research/ml_selection_backtest.py --top 5 --step 2
    python research/ml_selection_backtest.py --date 2026-03-15   # inspect one day's picks
"""
from __future__ import annotations

import argparse
import os
import sys

import numpy as np
import pandas as pd

_PROJ_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
if _PROJ_ROOT not in sys.path:
    sys.path.insert(0, _PROJ_ROOT)

from ml_predictor.features import FEATURE_COLUMNS  # noqa: E402
from ml_predictor.infer import MLPredictor  # noqa: E402
from ml_predictor.dataset import _cached_tickers, _load_ticker, DEFAULT_STEP  # noqa: E402
from research.strategy_validation_funnel import _sharpe, _max_drawdown  # noqa: E402

DEFAULT_CSV = os.path.join(_PROJ_ROOT, "ml_predictor", "training_data.csv")
OUT_CSV = os.path.join(os.path.dirname(os.path.abspath(__file__)), "ml_selection_results.csv")

ROUND_TRIP_COST_PCT = 0.30
HORIZONS = [1, 3, 5]
_CONF_RANK = {"HIGH": 0, "MEDIUM": 1, "LOW": 2}
_SELECTOR_TF = "3D"   # use the longest-horizon model to rank for a multi-day hold


def _passes_filters(r, filters: set) -> bool:
    """Strategy/quality gates (the docs' 'Filter' stage) applied to a BULLISH candidate.
    Each is an AND gate computed from the features already in the row."""
    def g(k, d=float("nan")):
        v = r.get(k, d)
        return float(v) if v is not None else d
    if "trend" in filters:      # confirmed uptrend: above EMA50 AND EMA200
        if not (g("price_vs_ema50") > 0 and g("price_vs_ema200") > 0):
            return False
    if "momentum" in filters:   # positive momentum: MACD>0 AND 10-day return>0
        if not (g("macd_hist") > 0 and g("return_10d") > 0):
            return False
    if "adx" in filters:        # trending market only
        if not (g("adx14") > 20):
            return False
    if "trigger" in filters:    # ≥1 bullish strategy trigger fired (S-signal confirmation)
        if sum(g(f"trigger_T{n}", 0) for n in range(1, 8)) < 1:
            return False
    if "lowvol" in filters:     # avoid extreme-volatility small-caps (noise)
        v = g("atr_pct")
        if not (v == v and v <= 4.0):
            return False
    if "notob" in filters:      # not overbought
        if not (g("rsi14") < 70):
            return False
    return True


def _load_close_series() -> dict:
    """Preload every cached ticker's close Series (for realized forward returns)."""
    out = {}
    for tk in _cached_tickers():
        loaded = _load_ticker(tk)
        if loaded is not None:
            out[tk] = loaded[0].dropna()  # close series
    return out


def _fwd_ret(close: pd.Series, date: pd.Timestamp, h: int) -> float:
    """Close-to-close % return h trading days after `date` (NaN if not enough future bars)."""
    try:
        idx = close.index.searchsorted(pd.Timestamp(date))
        if idx >= len(close):
            return float("nan")
        p0 = float(close.iloc[idx])
        j = idx + h
        if j >= len(close) or p0 <= 0:
            return float("nan")
        return (float(close.iloc[j]) / p0 - 1.0) * 100.0
    except Exception:
        return float("nan")


def run(csv_path: str = DEFAULT_CSV, top_n: int = 10, step: int = 1,
        one_date: str | None = None, rank_mode: str = "expmove",
        min_conf: str = "LOW", filters: set | None = None,
        six_filter: bool = False) -> pd.DataFrame:
    filters = filters or set()
    predictor = MLPredictor()
    if not predictor.available:
        raise SystemExit("ml_predictor model not loaded — run `python ml_predictor/train.py` first.")

    df = pd.read_csv(csv_path)
    df["date"] = pd.to_datetime(df["date"])
    holdout_start = predictor.manifest.get("holdout_start")
    # --six-filter validates the ML SELECTOR itself through the doc's funnel, which needs an
    # in-sample (IS) leg too — so evaluate the FULL date range and split at holdout_start.
    if six_filter:
        oos = df.copy()
    else:
        oos = df[df["date"] >= pd.to_datetime(holdout_start)].copy() if holdout_start else df
    print(f"  Model: rank by {_SELECTOR_TF} BULLISH · mode={rank_mode} · min_conf={min_conf} · "
          f"filters={sorted(filters) or 'none'} · top-{top_n} picks/day · cost {ROUND_TRIP_COST_PCT}% round-trip")
    print(f"  Out-of-sample rows: {len(oos):,} · loading close series for realized fwd returns…")
    closes = _load_close_series()

    dates = sorted(oos["date"].unique())
    if one_date:
        dates = [pd.Timestamp(one_date)]
    else:
        dates = dates[::step]

    pick_rows = []          # per-pick detail
    day_rows = []           # per-day basket vs market
    for d in dates:
        day = oos[oos["date"] == d]
        if len(day) < top_n:
            continue
        X = day[FEATURE_COLUMNS].to_numpy(dtype=float)
        q, proba_m, classes = predictor._raw_predict(_SELECTOR_TF, X)
        median_w = float(predictor.manifest.get("tf", {}).get(_SELECTOR_TF, {}).get("median_train_width", 1.5)) or 1.5
        cand = []
        for i, (_, r) in enumerate(day.iterrows()):
            row_q = {k: float(v[i]) for k, v in q.items()}
            pred = predictor._derive(row_q, proba_m[i], classes, _SELECTOR_TF, price=100.0,
                                     atr14=float(r["atr_pct"]) if np.isfinite(r["atr_pct"]) else None,
                                     median_w=median_w, live_price=None, today_high=None,
                                     news_score=0, anchor_close=100.0)
            if pred["direction"] != "BULLISH":
                continue
            crank = _CONF_RANK.get(pred["confidence"], 3)
            if crank > _CONF_RANK.get(min_conf, 2):     # confidence filter
                continue
            if filters and not _passes_filters(r, filters):   # strategy/quality gates
                continue
            atrp = float(r["atr_pct"]) if np.isfinite(r["atr_pct"]) else 2.0
            up50 = row_q["up50"]
            proba = proba_m[i]
            bull_p = float(proba[list(classes).index("BULLISH")]) if "BULLISH" in classes else 0.0
            cand.append({"ticker": r["ticker"], "up50": up50, "crank": crank, "conf": pred["confidence"],
                         "atr_pct": max(atrp, 0.1), "bull_p": bull_p})
        if not cand:
            day_rows.append({"date": d.strftime("%Y-%m-%d"), "n_picks": 0})
            continue
        # Ranking strategy (higher score = picked first):
        if rank_mode == "riskadj":       # expected move per unit volatility (Sharpe-like)
            keyfn = lambda c: (-(c["up50"] / c["atr_pct"]), c["crank"])
        elif rank_mode == "conf":        # confidence first (bull prob), then expected move
            keyfn = lambda c: (c["crank"], -c["bull_p"], -c["up50"])
        else:                            # "expmove" (default): raw expected up-move
            keyfn = lambda c: (-c["up50"], c["crank"])
        cand.sort(key=keyfn)
        picks = [(c["ticker"], c["up50"], c["crank"], c["conf"]) for c in cand[:top_n]]

        # Realized forward returns (net of cost) for picks and for the whole market that day.
        basket = {h: [] for h in HORIZONS}
        for tk, up50, _, conf in picks:
            cs = closes.get(tk)
            rets = {h: (_fwd_ret(cs, d, h) - ROUND_TRIP_COST_PCT) if cs is not None else float("nan")
                    for h in HORIZONS}
            for h in HORIZONS:
                if not np.isnan(rets[h]):
                    basket[h].append(rets[h])
            pick_rows.append({"date": d.strftime("%Y-%m-%d"), "ticker": tk, "exp_up_q50": round(up50, 2),
                              "confidence": conf, **{f"ret_{h}d_net": round(rets[h], 3) for h in HORIZONS}})
        market = {h: [] for h in HORIZONS}
        for tk in day["ticker"]:
            cs = closes.get(tk)
            for h in HORIZONS:
                v = _fwd_ret(cs, d, h) - ROUND_TRIP_COST_PCT if cs is not None else float("nan")
                if not np.isnan(v):
                    market[h].append(v)

        row = {"date": d.strftime("%Y-%m-%d"), "n_picks": len(picks)}
        for h in HORIZONS:
            row[f"basket_{h}d"] = float(np.mean(basket[h])) if basket[h] else float("nan")
            row[f"market_{h}d"] = float(np.mean(market[h])) if market[h] else float("nan")
        day_rows.append(row)

    picks_df = pd.DataFrame(pick_rows)
    days_df = pd.DataFrame(day_rows)
    picks_df.to_csv(OUT_CSV, index=False)

    if one_date:
        _one_day_report(one_date, picks_df, days_df)
    else:
        _summary(days_df, picks_df, top_n)
    if six_filter and not one_date:
        _six_filter_verdict(days_df, pd.to_datetime(holdout_start) if holdout_start else None)
    print(f"\n  ✓ Wrote per-pick detail → {OUT_CSV}")
    return days_df


def _six_filter_verdict(days_df: pd.DataFrame, split):
    """Run the '9,120-backtest' doc's 6-filter funnel on the ML SELECTION basket itself.
    Treats each decision day's top-N basket 3-day return as one trade; splits IS/OOS at the
    model's holdout_start. NOTE: the OOS leg is only as long as the manifest holdout window —
    if that is a handful of days the verdict is directional, not conclusive."""
    HOLD = 3
    d = days_df.copy()
    d = d[d["n_picks"] > 0]
    d["_dt"] = pd.to_datetime(d["date"])
    col = "basket_3d"
    if split is None:
        is_r = []
        oos_r = list(d[col].dropna())
    else:
        is_r = list(d[d["_dt"] < split][col].dropna())
        oos_r = list(d[d["_dt"] >= split][col].dropna())
    is_s, oos_s = _sharpe(is_r, HOLD), _sharpe(oos_r, HOLD)
    mdd = _max_drawdown(oos_r)
    n_oos = len(oos_r)
    checks = [
        ("[01] OOS Sharpe > 0.5", oos_s > 0.5, f"{oos_s:+.2f}"),
        ("[02] Max DD better than -35%", mdd > -35.0, f"{mdd:.1f}%"),
        ("[03] OOS Sharpe < 2.5 (not absurd)", oos_s < 2.5, f"{oos_s:+.2f}"),
        ("[04] OOS <= IS*1.3 + 0.5 (not overfit)", oos_s <= is_s * 1.3 + 0.5, f"OOS {oos_s:+.2f} / IS {is_s:+.2f}"),
        ("[05] At least 30 OOS trades", n_oos >= 30, f"{n_oos}"),
        ("[06] IS Sharpe > 0", is_s > 0, f"{is_s:+.2f}"),
    ]
    print("\n" + "=" * 78)
    print("  6-FILTER VALIDATION — is the ML top-N SELECTION strategy a real OOS edge?")
    print(f"  IS trades={len(is_r)}  OOS trades={n_oos}  (3-day basket return per decision day)")
    print("=" * 78)
    for label, ok, val in checks:
        print(f"  {'PASS' if ok else 'FAIL'}  {label:<42} {val}")
    verdict = "SURVIVES all 6 filters" if all(c[1] for c in checks) else "does NOT survive"
    print(f"\n  → The ML selection strategy {verdict}.")
    if n_oos < 30:
        print("  ⚠ OOS window is short (manifest holdout is small) — treat as directional only.")


def _summary(days_df: pd.DataFrame, picks_df: pd.DataFrame, top_n: int):
    active = days_df[days_df["n_picks"] > 0]
    print("\n" + "=" * 78)
    print(f"  STOCK-SELECTION BACKTEST — top-{top_n} BULLISH picks, equal-weight, held N days")
    print(f"  {len(active)} decision days · {len(picks_df):,} total picks · net of {ROUND_TRIP_COST_PCT}% cost")
    print("=" * 78)
    print(f"  {'Hold':<7}{'BasketAvg':>11}{'MarketAvg':>11}{'Edge':>9}{'PickWin%':>10}{'DaysBeatMkt':>13}{'BookRet':>9}")
    print("  " + "-" * 70)
    for h in HORIZONS:
        b = active[f"basket_{h}d"].dropna()
        m = active[f"market_{h}d"].dropna()
        pick_win = float((picks_df[f"ret_{h}d_net"] > 0).mean()) if len(picks_df) else float("nan")
        beat = float((active[f"basket_{h}d"] > active[f"market_{h}d"]).mean())
        # Equal-weight daily-rebalanced book return: compound each day's basket mean across days.
        book = 1.0
        for v in active[f"basket_{h}d"].dropna():
            book *= (1 + v / 100.0)
        book_ret = (book - 1) * 100.0
        print(f"  {str(h)+'d':<7}{b.mean():>+11.2f}{m.mean():>+11.2f}{(b.mean()-m.mean()):>+9.2f}"
              f"{pick_win:>10.0%}{beat:>13.0%}{book_ret:>+9.1f}")
    print("\n  BasketAvg/MarketAvg = mean per-pick vs per-stock net return over the hold.")
    print("  Edge = how much the picks beat buying the whole market. PickWin% = picks that closed green.")
    print("  DaysBeatMkt = fraction of decision days the basket beat the market.")
    print(f"  BookRet compounds each day's top-{top_n} basket across all decision days (rebalanced).")
    # Direct answer to the 5-day question
    b5 = active["basket_5d"].dropna()
    m5 = active["market_5d"].dropna()
    win5 = float((picks_df["ret_5d_net"] > 0).mean()) if len(picks_df) else float("nan")
    verdict = "YES" if (b5.mean() > 0 and b5.mean() > m5.mean()) else ("MARGINAL" if b5.mean() > 0 else "NO")
    print("\n  ── 5-DAY VERDICT ──")
    print(f"  Avg pick makes {b5.mean():+.2f}% over 5 days (net) vs market {m5.mean():+.2f}% · "
          f"{win5:.0%} of picks close green · edge {b5.mean()-m5.mean():+.2f}%  →  profitable? {verdict}")


def _one_day_report(date: str, picks_df: pd.DataFrame, days_df: pd.DataFrame):
    print("\n" + "=" * 78)
    print(f"  SINGLE-DAY PICKS — {date}")
    print("=" * 78)
    if picks_df.empty:
        print("  No BULLISH picks that day.")
        return
    cols = ["ticker", "confidence", "exp_up_q50", "ret_1d_net", "ret_3d_net", "ret_5d_net"]
    print(picks_df[cols].to_string(index=False))
    if not days_df.empty and days_df.iloc[0].get("n_picks", 0) > 0:
        r = days_df.iloc[0]
        print(f"\n  Basket avg — 1d {r['basket_1d']:+.2f}%  3d {r['basket_3d']:+.2f}%  5d {r['basket_5d']:+.2f}%  "
              f"(net) vs market 5d {r['market_5d']:+.2f}%")


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--csv", default=DEFAULT_CSV)
    ap.add_argument("--top", type=int, default=10)
    ap.add_argument("--step", type=int, default=1, help="sample every Nth decision day")
    ap.add_argument("--date", default=None, help="inspect a single date (YYYY-MM-DD)")
    ap.add_argument("--rank", default="expmove", choices=["expmove", "riskadj", "conf"],
                    help="ranking strategy for picks")
    ap.add_argument("--min-conf", default="LOW", choices=["LOW", "MEDIUM", "HIGH"],
                    help="only pick stocks at/above this confidence")
    ap.add_argument("--filters", default="", help="comma-separated quality gates: "
                    "trend,momentum,adx,trigger,lowvol,notob")
    ap.add_argument("--six-filter", action="store_true",
                    help="validate the ML selection basket through the doc's 6-filter funnel (IS vs OOS)")
    args = ap.parse_args()
    filters = {f.strip() for f in args.filters.split(",") if f.strip()}
    run(args.csv, top_n=args.top, step=args.step, one_date=args.date,
        rank_mode=args.rank, min_conf=args.min_conf, filters=filters,
        six_filter=args.six_filter)


if __name__ == "__main__":
    main()