draftfm / frozen_battery.json
brianward92's picture
DraftFM v0.1 artifacts: 14 pinned checkpoints (sha256-verified against run_manifest), ONNX exports, battery + ledger manifests
14a6c3d verified
Raw
History Blame Contribute Delete
7.58 kB
{
"protocol_tag": "eval-protocol-v1.1",
"_comment": "Pre-registered battery (docs/eval_protocol.md section 4). Freeze artifact-backed members with their file sha256 (path points at the artifact FILE, e.g. runs/<id>/best.pt) BEFORE the T0 snapshot download; frozen_snapshot is filled at T0 with the raw MSH csv.gz path, sha256, and S3 ETag. run_frozen_eval.py refuses real mode while any of these are null. Member paths are DATA_ROOT-relative (resolved by run_frozen_eval.resolve_artifact_path against <MTGA_DATA_ROOT>/foundation; the pinned checkpoints are published in the brianward92/draftfm Hugging Face model repo, local mirror /opt/brianward/dat/mtga/foundation/runs/). The sha256 is the authoritative anchor, so the battery is portable across boxes -- set MTGA_DATA_ROOT to the unpacked-weights root (T2.7). draftfm members carry condition={wr_id:33, games_id:6} for deployment-mode scoring (matches scripts/export_draftfm.py's serving default, ~0.66 win rate / 1000-games bucket); member_frames also always computes 'human' mode (no override) for the same member.",
"_pending": "Post-day-1 rows (per-set MSH ceiling, F-full fine-tuned on MSH) are pre-registered in docs/eval_protocol.md section 4.6 but their artifacts don't exist yet by design (trained only after this zero-shot battery runs against the frozen MSH snapshot) -- they get appended in a follow-up commit, never added or iterated before T0.",
"calibration": {
"temperature": 1.28,
"fit_run": "20260704_135822_f_dev",
"fit_script": "scripts/fit_dev_temperature.py",
"selection_rule": "argmin mean dev-trio (BRO/TMT/SOS) log-loss over a 0.30:3.00:0.02 grid; never fitted on MSH",
"dev_mean_ece_at_t1": 0.0891,
"dev_mean_ece_at_frozen_t": 0.0250,
"frozen_before_t0": true,
"applied_to": ["deployment", "human"]
},
"frozen_snapshot": {
"path": "raw/draft_data_public.MSH.PremierDraft.csv.gz",
"sha256": "013df16b8994534f69ed63c87ab684acafc5f4cbe82982264b0fc111dbb2183a",
"etag": "\"252007adfbca6f026766527823ffa6d5-8\""
},
"models": [
{
"name": "baseline-random",
"kind": "baseline-random"
},
{
"name": "baseline-rarity",
"kind": "baseline-rarity"
},
{
"name": "f-full",
"kind": "draftfm",
"run": "20260705_110743_f_full",
"path": "runs/20260705_110743_f_full/best.pt",
"sha256": "61f87c1a0f84daa1baf254f1386deb819edd59d7f175aa3621a9cb620eee6167",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "headline model: final recipe (set_ctx=False, winning ablation), all 31 sets, no dev holdout. Its own dev numbers are meaningless and never quoted."
},
{
"name": "f-dev",
"kind": "draftfm",
"run": "20260704_135822_f_dev",
"path": "runs/20260704_135822_f_dev/best.pt",
"sha256": "c10e93e6e5022c217d03f41ba3f0e5975a2d460b96b59337d33ac422a314741d",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "universe minus dev trio (BRO/TMT/SOS); doubles as scaling rung S27 (F-dev universe)."
},
{
"name": "s1",
"kind": "draftfm",
"run": "20260704_154741_s1",
"path": "runs/20260704_154741_s1/best.pt",
"sha256": "8b2b8cdf737fc5863e951563ddc11630b4c2b9ad78d8180fc225ee15cc261448",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S1={NEO} (Bertram anchor)."
},
{
"name": "s2",
"kind": "draftfm",
"run": "20260704_161450_s2",
"path": "runs/20260704_161450_s2/best.pt",
"sha256": "0be4890dc410025f1873aadaf9b9ffc5d353c20f356c2819a7abbd1e280797e4",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S2=S1+DSK."
},
{
"name": "s2b",
"kind": "draftfm",
"run": "20260704_171827_s2b",
"path": "runs/20260704_171827_s2b/best.pt",
"sha256": "79e2f953788f97449e00a322f11e76c11f914a908ece60e7562a801d47231481",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling composition probe S2b={MOM,TDM}."
},
{
"name": "s4",
"kind": "draftfm",
"run": "20260704_175554_s4",
"path": "runs/20260704_175554_s4/best.pt",
"sha256": "5b35bb209b3daaa03e4723e7d56e4ed2a222b89a130acba0da3622cf8f100068",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S4=S2+DMU,FIN."
},
{
"name": "s4b",
"kind": "draftfm",
"run": "20260704_191659_s4b",
"path": "runs/20260704_191659_s4b/best.pt",
"sha256": "2382932cb470ff98d575e37a6ebd9963bf83355261ebdd56520ade4ad1aeb176",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling composition probe S4b={STX,SNC,OTJ,TLA}."
},
{
"name": "s8",
"kind": "draftfm",
"run": "20260704_202620_s8",
"path": "runs/20260704_202620_s8/best.pt",
"sha256": "ef23bff70621a8278d1dbeb9387fa90ec6656bcc3dcf22c4e969b1f3df416d1a",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S8=S4+STX,MOM,BLB,TLA."
},
{
"name": "s16",
"kind": "draftfm",
"run": "20260704_213310_s16",
"path": "runs/20260704_213310_s16/best.pt",
"sha256": "87c9e3e9ea36281c5c8e5cf582ac513fc24f7a942740442b45bffa67a89ff2e4",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S16=S8+AFR,SNC,ONE,LTR,WOE,MKM,OTJ,EOE."
},
{
"name": "a-notext",
"kind": "draftfm",
"run": "20260704_135710_a_notext",
"path": "runs/20260704_135710_a_notext/best.pt",
"sha256": "824623d0aed62f8f1d039d8d195cfcd3db7a86f4da7ca06bc2d123dd32e50861",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: no oracle-text embedding (structured 391-d only). Pre-registered for the UB-shift analysis."
},
{
"name": "a-noctx",
"kind": "draftfm",
"run": "20260704_160422_a_noctx",
"path": "runs/20260704_160422_a_noctx/best.pt",
"sha256": "d3eb5e65b7a6b2446877c18c741138e0122c6cd1fef1543c4d557a8b419ea804",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: set_ctx=False (no set-context cross-attention). Won the dev-trio ablation sweep; F-full uses this recipe (F-dev keeps the set-context tower -- see method.tex sec:arch's disclosure paragraph)."
},
{
"name": "a-proportional",
"kind": "draftfm",
"run": "20260704_201946_a_proportional",
"path": "runs/20260704_201946_a_proportional/best.pt",
"sha256": "5463d3c9262148cc2127ae1d1e60716bc7a36f251c71dd2a4115dd873a038658",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: proportional (not sqrt) per-set sampling."
},
{
"name": "a-topfilter",
"kind": "draftfm",
"run": "20260705_030215_a_topfilter",
"path": "runs/20260705_030215_a_topfilter/best.pt",
"sha256": "f0240ba209eda57094b8c3afafd982fd1280afac3542b9bcc350c77b34449fd8",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: train on expert (top-skill) picks only, no skill conditioning."
},
{
"name": "a-noub",
"kind": "draftfm",
"run": "20260705_164029_a_noUB",
"path": "runs/20260705_164029_a_noUB/best.pt",
"sha256": "9d9107563b234d3715567f844be2309e13435c368a69dc2a19eb0d2b2ff8d790",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: three licensed-IP training sets (LTR/FIN/TLA) removed from A-noctx (the winning architecture). Pre-registered for the UB-shift analysis; clean comparator is the A-noctx row, not Full."
}
]
}