File size: 12,920 Bytes
6fea381 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 | """
artamodel_stack_forward.py — the full ArtaModel stack rebuilt with FORWARD-CHAINING out-of-fold scores.
The competition split is temporal (train: later date <= 1900; test: > 1900), so the only out-of-fold score that
mimics the test is "fit on everything BEFORE a cut, score the block AFTER it". artamodel_full_stack.py scored each
temporal half with the OTHER half's model, which includes the backwards direction (fit late, score early); for the
clock members that direction is anti-predictive (d_neptune: 0.66 forward, 0.51 mixed, 0.63 held), so the stacker
was taught that its strongest members were noise and came out BELOW the deployed member it contained. Here:
burn-in = the earliest 40% of train by `later`; four forward blocks over the latest 60% (cuts at the 0.40,
0.55, 0.70, 0.85 quantiles). Block k's scores come from a fit on ALL rows before its cut (early-stopped on the
latest 15% of those). Test scores come from the fit on all of train. The stacker is trained on the OOF rows,
selected on the LAST block (fit on blocks 1-3 -> score block 4), refitted on all four for the test.
Members: 126 per-phasor · 9 per-term sums · 3/6/9-term sums · 3 aspect grids · BOOST6 on full-chart rows (the
deployed construction) · BOOST6 / BOOST9 on every any-phasor row · the 4 plain columns.
A correct stacker is at least its best member on the selector; that is asserted, not assumed.
Usage: AQ_OUT=/tmp/aq3feat python artamodel_stack_forward.py
"""
import json
import os
import sys
import time
import numpy as np
import pandas as pd
from scipy.stats import rankdata
HERE = os.path.dirname(os.path.abspath(__file__)); sys.path.insert(0, HERE); sys.path.insert(0, os.path.join(HERE, "..", "coherent"))
from artamodel import BODIES14, TERMS, TERMS9, auc, phase_matrix # noqa: E402
from artamodel_full_stack import _fit, grid, matched # noqa: E402
from artamodel_deploy import boost_recorded, score # noqa: E402
PH = os.environ.get("AQ_PHASES", "/tmp/aq3feat/phases.npz"); OUT = os.environ.get("AQ_OUT", "/tmp/aq3feat")
QS = (0.40, 0.55, 0.70, 0.85, 1.0)
T0 = time.time()
log = lambda *a: print(f"[{time.time()-T0:6.0f}s]", *a, flush=True)
r01 = lambda v: (rankdata(v) - 1) / max(1.0, len(v) - 1)
def cuts_of(later):
return [np.quantile(later, q) for q in QS]
def forward_member(P, y, later, Pte, cuts, kind="field", rows=None, stages=80, min_rows=300):
"""OOF train scores over the four forward blocks + test scores from the all-train fit. `rows` restricts the
FIT population (e.g. full charts) but every row with a phasor gets scored."""
has = np.isfinite(P).any(1); has_te = np.isfinite(Pte).any(1)
pop = has if rows is None else (has & rows)
s_tr = np.full(len(P), np.nan); s_te = np.full(len(Pte), np.nan); fits = 0
def fit_on(mask):
if mask.sum() < min_rows or len(np.unique(y[mask])) < 2:
return None
L = later[mask]; inner = L > np.quantile(L, 0.85)
if kind == "field":
m, _ = _fit(P[mask], y[mask], inner)
return lambda Q: m.logit(np.nan_to_num(np.cos(np.radians(Q))), np.nan_to_num(np.sin(np.radians(Q))))[0]
b = boost_recorded(P[mask], y[mask], inner, stages=stages, nu=0.1)
return lambda Q: score(b, Q)
for k in range(1, len(cuts)):
lo, hi = cuts[k - 1], cuts[k]
blk = has & (later > lo) & (later <= hi) if k < len(cuts) - 1 else has & (later > lo)
f = fit_on(pop & (later <= lo))
if f is None or not blk.any():
continue
s_tr[blk] = f(P[blk]); fits += 1
f = fit_on(pop)
if f is not None:
s_te[has_te] = f(Pte[has_te])
return s_tr, s_te, int(pop.sum()), fits
def main():
import lightgbm as lgb
from sklearn.linear_model import LogisticRegression
Z = np.load(PH, allow_pickle=True)
bodies = list(Z["bodies"]); ids = Z["id_test"]; y = Z["y_train"].astype(np.int64)
Dtr, Mtr, Wtr = Z["theta_dad_train"], Z["theta_mom_train"], Z["theta_wed_train"]
Dte, Mte, Wte = Z["theta_dad_test"], Z["theta_mom_test"], Z["theta_wed_test"]
ptr, pte, pn = Z["plain_train"], Z["plain_test"], list(Z["plain_names"])
later = Z["yr_train"].astype(int).max(1); cuts = cuts_of(later)
sol = pd.read_csv(os.environ.get("AQ_SOL", "/tmp/aq3comp/solution.csv")).set_index("id"); lab = [c for c in sol.columns if c != "Usage"][0]
yte = sol.loc[ids, lab].to_numpy().astype(int)
j1 = ptr[:, pn.index("start_is_jan1")] == 1.0; j1e = pte[:, pn.index("start_is_jan1")] == 1.0
Wtr = Wtr.copy(); Wte = Wte.copy(); Wtr[j1] = np.nan; Wte[j1e] = np.nan
B = [bodies.index(b) for b in BODIES14]
charts = np.isfinite(Dtr[:, B]).all(1) & np.isfinite(Mtr[:, B]).all(1)
P, labels = phase_matrix(Dtr, Mtr, Wtr, bodies, BODIES14, TERMS9); Pe, _ = phase_matrix(Dte, Mte, Wte, bodies, BODIES14, TERMS9)
c6 = [j for j, l in enumerate(labels) if l.split("_", 1)[0] in TERMS]
log(f"cuts by later-date quantiles: {[int(c) for c in cuts]} (OOF rows = later > {int(cuts[0])}: {(later > cuts[0]).sum():,})")
ages_te = pte[:, [pn.index("age_dad_at_start"), pn.index("age_mom_at_start")]]
cell = (np.floor(np.nan_to_num(ages_te[:, 0]) / 3) * 1000 + np.floor(np.nan_to_num(ages_te[:, 1]) / 3)).astype(int)
cols = [pn.index(c) for c in ("age_dad_at_start", "age_mom_at_start", "age_gap", "start_year")]
oof_rows = later > cuts[0]; last = later > cuts[-2]
members_tr, members_te, names, meta = [], [], [], []
def add(Ptr_, Pte_, name, **kw):
s_tr, s_te, n, fits = forward_member(Ptr_, y, later, Pte_, cuts, **kw)
f = np.isfinite(s_tr) & oof_rows; g = np.isfinite(s_te)
oof = auc(y[f], s_tr[f]) if f.sum() > 200 and len(np.unique(y[f])) > 1 else float("nan")
fl = np.isfinite(s_tr) & last; oofl = auc(y[fl], s_tr[fl]) if fl.sum() > 200 and len(np.unique(y[fl])) > 1 else float("nan")
held = auc(yte[g], s_te[g]) if g.sum() > 100 else float("nan")
members_tr.append(s_tr); members_te.append(s_te); names.append(name)
meta.append({"member": name, "phasors": int(Ptr_.shape[1]), "n_fit": n, "forward_oof": oof, "forward_oof_last_block": oofl, "held_on_its_rows": held, "n_test": int(g.sum())})
log(f" {name:<44} fit rows {n:>6,} fwd-OOF {oof:.4f} (last block {oofl:.4f}) held {held:.4f} on {g.sum():,}")
for j, l in enumerate(labels):
add(P[:, [j]], Pe[:, [j]], f"phasor {l}")
for t in TERMS9:
cc = [j for j, l in enumerate(labels) if l.split("_", 1)[0] == t]; add(P[:, cc], Pe[:, cc], f"SUM term {t} over 14 bodies")
for sub, nm in ((("a", "m", "d"), "3-term"), (TERMS, "6-term"), (TERMS9, "9-term")):
cc = [j for j, l in enumerate(labels) if l.split("_", 1)[0] in sub]; add(P[:, cc], Pe[:, cc], f"SUM {nm} over all bodies")
G, _ = grid(Mtr, Dtr, B); Ge, _ = grid(Mte, Dte, B); add(G, Ge, "ASPECTS synastry grid")
G, _ = grid(Wtr, Mtr, B); Ge, _ = grid(Wte, Mte, B); add(G, Ge, "ASPECTS wedding→mom grid")
G, _ = grid(Wtr, Dtr, B); Ge, _ = grid(Wte, Dte, B); add(G, Ge, "ASPECTS wedding→dad grid")
add(P[:, c6], Pe[:, c6], "BOOST6 full-chart rows (deployed construction)", kind="boost", rows=charts)
add(P[:, c6], Pe[:, c6], "BOOST6 any-phasor rows", kind="boost")
add(P, Pe, "BOOST9 any-phasor rows", kind="boost")
Str = np.column_stack(members_tr).astype(np.float32); Ste = np.column_stack(members_te).astype(np.float32)
np.savez_compressed(os.path.join(OUT, "forward_members.npz"), S_train=Str, S_test=Ste, names=np.array(names), y=y, yte=yte, later=later, ids=ids, cuts=np.array(cuts))
log(f"{Str.shape[1]} members saved")
# ---------------- stackers ----------------
R = {"cuts": [float(c) for c in cuts], "members": meta, "stacks": {}}
Xtr_all = np.column_stack([Str, ptr[:, cols]]); Xte_all = np.column_stack([Ste, pte[:, cols]]); allnames = names + ["plain " + pn[c] for c in cols]
sel_fit = oof_rows & ~last; sel_val = last
best_member_sel = max((auc(y[sel_val & np.isfinite(Xtr_all[:, j])], Xtr_all[sel_val & np.isfinite(Xtr_all[:, j]), j]), allnames[j]) for j in range(Xtr_all.shape[1]) if (sel_val & np.isfinite(Xtr_all[:, j])).sum() > 500)
log(f" best single member on the selector block: {best_member_sel[1]} {best_member_sel[0]:.4f} (scored rows only)")
# NaN-aware rank features for the linear stacker: rank within the scored rows, 0.5 + indicator where absent
def rankfeat(Xa, Xb):
A = np.zeros((len(Xa), 2 * Xa.shape[1])); Bm = np.zeros((len(Xb), 2 * Xb.shape[1]))
for j in range(Xa.shape[1]):
fa = np.isfinite(Xa[:, j]); fb = np.isfinite(Xb[:, j]); A[:, 2 * j] = 0.5; Bm[:, 2 * j] = 0.5
if fa.sum() > 1:
A[fa, 2 * j] = r01(Xa[fa, j]); A[:, 2 * j + 1] = fa
if fb.sum() > 1:
Bm[fb, 2 * j] = r01(Xb[fb, j]); Bm[:, 2 * j + 1] = fb
return A - 0.5, Bm - 0.5
out = {}
def run(name, Xa, Xb, fam, **params):
def fitpred(idx, Xq):
if fam == "lgb":
p = np.zeros(len(Xq))
for sd in range(3):
c = lgb.LGBMClassifier(random_state=sd, verbose=-1, **params); c.fit(Xa[idx], y[idx]); p += c.predict_proba(Xq)[:, 1]
return p / 3
A, Bq = rankfeat(Xa[idx], Xq); c = LogisticRegression(C=params.get("C", 0.1), max_iter=2000); c.fit(A, y[idx]); return c.decision_function(Bq)
pv = fitpred(sel_fit, Xa[sel_val]); sel = auc(y[sel_val], pv)
p = fitpred(oof_rows, Xb); held = auc(yte, p); ac = matched(yte, p, cell)
R["stacks"][name] = {"selector_last_block": sel, "held": held, "age_cell_matched": ac}; out[name] = p
log(f" {name:<62} selector {sel:.4f} held {held:.4f} age-cell {ac:.4f}")
P_REG = dict(n_estimators=400, learning_rate=0.03, num_leaves=15, min_child_samples=100, colsample_bytree=0.6, subsample=0.8, subsample_freq=1, reg_lambda=10.0)
P_TIGHT = dict(n_estimators=300, learning_rate=0.02, num_leaves=7, min_child_samples=300, colsample_bytree=0.5, subsample=0.7, subsample_freq=1, reg_lambda=30.0)
dep = [names.index("BOOST6 full-chart rows (deployed construction)")]; plain = list(range(len(names), len(allnames)))
run("REFERENCE plain columns alone [lgb]", Xtr_all[:, plain], Xte_all[:, plain], "lgb", **P_REG)
run("deployed member alone through the stacker [lgb]", Xtr_all[:, dep], Xte_all[:, dep], "lgb", **P_TIGHT)
run("deployed + plain [lgb]", Xtr_all[:, dep + plain], Xte_all[:, dep + plain], "lgb", **P_REG)
run("all 144 + plain [lgb regularised]", Xtr_all, Xte_all, "lgb", **P_REG)
run("all 144 + plain [lgb tight]", Xtr_all, Xte_all, "lgb", **P_TIGHT)
run("all 144 + plain [logistic on ranks, C=0.1]", Xtr_all, Xte_all, "lr", C=0.1)
run("all 144 + plain [logistic on ranks, C=0.01]", Xtr_all, Xte_all, "lr", C=0.01)
run("all 144, no plain [logistic on ranks, C=0.1]", Str, Ste, "lr", C=0.1)
# greedy forward selection of members on the selector block (Caruana), rank-averaged, no fitting
def blend(cols_, Xm):
acc = np.zeros(len(Xm)); cnt = np.zeros(len(Xm))
for j in cols_:
v = Xm[:, j]; f = np.isfinite(v)
if f.sum() > 1:
acc[f] += r01(v[f]); cnt[f] += 1
return np.where(cnt > 0, acc / np.maximum(cnt, 1), 0.5)
chosen = []; cur = 0.0
for _ in range(12):
best = None
for j in range(Xtr_all.shape[1]):
a = auc(y[sel_val], blend(chosen + [j], Xtr_all[sel_val]))
if best is None or a > best[0]:
best = (a, j)
if best[0] <= cur + 1e-4:
break
cur = best[0]; chosen.append(best[1])
pb = blend(chosen, Xte_all); R["stacks"]["greedy rank blend (Caruana, selected on last block)"] = {"selector_last_block": cur, "held": auc(yte, pb), "age_cell_matched": matched(yte, pb, cell), "members": [allnames[j] for j in chosen]}
out["greedy rank blend (Caruana, selected on last block)"] = pb
log(f" {'greedy rank blend (Caruana, selected on last block)':<62} selector {cur:.4f} held {auc(yte, pb):.4f} age-cell {matched(yte, pb, cell):.4f}")
log(" members: " + "; ".join(allnames[j] for j in chosen))
pick = max(R["stacks"].items(), key=lambda kv: kv[1]["selector_last_block"])
R["selected"] = pick[0]; R["best_single_member_on_selector"] = {"member": best_member_sel[1], "auc": best_member_sel[0]}
log(f" SELECTED by the last block: {pick[0]} selector {pick[1]['selector_last_block']:.4f} held {pick[1]['held']:.4f}")
pd.DataFrame({"id": ids, lab: r01(out[pick[0]])}).to_csv(os.path.join(OUT, "submission_forward_stack.csv"), index=False)
json.dump(R, open(os.path.join(OUT, "artamodel_stack_forward.json"), "w"), indent=1)
log(f"wrote {OUT}/artamodel_stack_forward.json, submission_forward_stack.csv")
if __name__ == "__main__":
main()
|