#!/usr/bin/env python3 """C23: Drop XGB — RF+LGBM only with weight search. C22 showed XGB optimal weight is 0.25 on 12-stock (vs 0.5 on 5-stock), suggesting XGB adds noise across diverse sectors. Remove it entirely and re-search the RF/LGBM blend. """ import sys from pathlib import Path ROOT = Path(__file__).resolve().parent.parent sys.path.insert(0, str(ROOT)) import warnings; warnings.filterwarnings("ignore") import json import argparse import numpy as np from sklearn.ensemble import RandomForestClassifier from sklearn.preprocessing import StandardScaler from lightgbm import LGBMClassifier from scripts.improvement_harness import ( fetch_df, build_triple_barrier_labels, compute_metrics, CURRENT_FEATURES, DEFAULT_STOCKS, EXTENDED_STOCKS, PASS_DIR_ACC, PASS_UP_PREC, MIN_TRAIN, STEP, LABEL_HORIZON, RF_PARAMS, ) from models.predictor import _build_features CLASSES = [-1, 0, 1] LGBM_PARAMS = dict( n_estimators=150, max_depth=6, learning_rate=0.05, class_weight="balanced", random_state=42, n_jobs=1, verbose=-1, ) WEIGHT_GRID = [0.25, 0.5, 0.75, 1.0, 1.25, 1.5, 1.75, 2.0] DIR_ACC_FLOOR = 40.0 def _collect_proba(feat_df, label_arr, cols): avail = [c for c in cols if c in feat_df.columns] X_all = feat_df[avail].fillna(0).values n = len(feat_df) y_true_all, rf_all, lgbm_all = [], [], [] cutoff = MIN_TRAIN while cutoff + STEP + LABEL_HORIZON <= n: train_end = cutoff - LABEL_HORIZON if train_end < MIN_TRAIN - LABEL_HORIZON: cutoff += STEP; continue y_tr = label_arr[:train_end] valid = ~np.isnan(y_tr) y_v = y_tr[valid].astype(int) if len(y_v) < 10 or len(np.unique(y_v)) < 2: cutoff += STEP; continue X_tr = X_all[:train_end][valid] scaler = StandardScaler() X_tr_s = scaler.fit_transform(X_tr) rf = RandomForestClassifier(**RF_PARAMS) rf.fit(X_tr_s, y_v) lgbm = LGBMClassifier(**LGBM_PARAMS) lgbm.fit(X_tr_s, y_v) test_end = min(cutoff + STEP, n - LABEL_HORIZON) y_te = label_arr[cutoff:test_end] valid_te = ~np.isnan(y_te) if valid_te.sum() == 0: cutoff += STEP; continue X_te_s = scaler.transform(X_all[cutoff:test_end][valid_te]) y_true = y_te[valid_te].astype(int) rf_p = np.zeros((len(y_true), 3)) for col_i, cls in enumerate(CLASSES): if cls in rf.classes_: rf_p[:, col_i] = rf.predict_proba(X_te_s)[:, list(rf.classes_).index(cls)] lgbm_p = np.zeros((len(y_true), 3)) lgbm_raw = lgbm.predict_proba(X_te_s) for col_i, cls in enumerate(CLASSES): if cls in lgbm.classes_: lgbm_p[:, col_i] = lgbm_raw[:, list(lgbm.classes_).index(cls)] y_true_all.extend(y_true.tolist()) rf_all.append(rf_p) lgbm_all.append(lgbm_p) cutoff += STEP if not y_true_all: return None return { "y_true": np.array(y_true_all), "rf_proba": np.vstack(rf_all), "lgbm_proba": np.vstack(lgbm_all), } def _blend_metrics(c, w_rf, w_lgbm): total = w_rf + w_lgbm blended = (w_rf * c["rf_proba"] + w_lgbm * c["lgbm_proba"]) / total y_pred = np.array([CLASSES[i] for i in blended.argmax(axis=1)]) return compute_metrics(c["y_true"], y_pred) def _search_weights(all_collected): best_score, best_w, best_avg = -1.0, (1.0, 1.0), {} for w_lgbm in WEIGHT_GRID: up_precs, dir_accs = [], [] for c in all_collected: m = _blend_metrics(c, 1.0, w_lgbm) if m and not np.isnan(m.get("up_precision", float("nan"))): up_precs.append(m["up_precision"]) dir_accs.append(m["dir_accuracy"]) if not up_precs: continue avg_up = float(np.mean(up_precs)) avg_dir = float(np.mean(dir_accs)) if avg_dir < DIR_ACC_FLOOR: continue if avg_up > best_score: best_score = avg_up best_w = (1.0, w_lgbm) best_avg = {"dir_accuracy": round(avg_dir, 1), "up_precision": round(avg_up, 1)} return best_w, best_avg def main(): parser = argparse.ArgumentParser() parser.add_argument("--extended", action="store_true") args = parser.parse_args() stocks = EXTENDED_STOCKS if args.extended else DEFAULT_STOCKS tag = "12-stock" if args.extended else "5-stock" print(f"\n=== C23: RF+LGBM only [{tag}] ===") all_collected, per_stock = [], {} for stock_no in stocks: print(f" {stock_no} collecting...", end=" ", flush=True) df = fetch_df(stock_no) if df is None or df.empty: print("no data"); continue feat = _build_features(df) close = (df.set_index("date")["close"] if "date" in df.columns else df["close"]).values labels = build_triple_barrier_labels(close) c = _collect_proba(feat, labels, CURRENT_FEATURES) if c is None: print("insufficient data"); continue baseline = _blend_metrics(c, 1.0, 1.0) per_stock[stock_no] = {"baseline_equal": baseline} all_collected.append(c) print(f"dir={baseline['dir_accuracy']}% ↑prec={baseline['up_precision']}%") if not all_collected: print("No data."); return print("\n Searching RF/LGBM weights...", flush=True) best_w, best_avg = _search_weights(all_collected) w_rf, w_lgbm = best_w for i, (stock_no, info) in enumerate(per_stock.items()): if i < len(all_collected): info["optimized"] = _blend_metrics(all_collected[i], w_rf, w_lgbm) base_dir = np.mean([v["baseline_equal"]["dir_accuracy"] for v in per_stock.values()]) base_prec = np.mean([v["baseline_equal"]["up_precision"] for v in per_stock.values()]) passed = (best_avg.get("dir_accuracy", 0) >= PASS_DIR_ACC and best_avg.get("up_precision", 0) >= PASS_UP_PREC) print(f"\n Best weights: RF={w_rf}, LGBM={w_lgbm}") print(f" Baseline RF+LGBM (1/1): dir={base_dir:.1f}% ↑prec={base_prec:.1f}%") print(f" Optimized : dir={best_avg.get('dir_accuracy')}% ↑prec={best_avg.get('up_precision')}%") print(f" Gate (dir≥{PASS_DIR_ACC}% AND ↑prec≥{PASS_UP_PREC}%): {'PASS ✓' if passed else 'FAIL ✗'}") print(f"\n Per-stock (optimized w_lgbm={w_lgbm}):") for s, v in per_stock.items(): opt = v.get("optimized", {}) base = v["baseline_equal"] delta = f"{opt.get('up_precision',0)-base['up_precision']:+.1f}pp" print(f" {s}: ↑prec {base['up_precision']}% → {opt.get('up_precision')}% ({delta})") result = { "experiment": "C23", "description": "Drop XGB; RF+LGBM only with weight search", "stocks": stocks, "best_weights": {"RF": w_rf, "LGBM": w_lgbm}, "baseline_aggregate": {"dir_accuracy": round(base_dir, 1), "up_precision": round(base_prec, 1)}, "optimized_aggregate": best_avg, "per_stock": per_stock, "passed": passed, "pass_gate": {"dir_accuracy": PASS_DIR_ACC, "up_precision": PASS_UP_PREC}, } suffix = "_12stock" if args.extended else "" out = ROOT / f"docs/c23_result{suffix}.json" out.parent.mkdir(exist_ok=True) out.write_text(json.dumps(result, indent=2)) print(f"\n Saved: {out}") if __name__ == "__main__": main()