File size: 7,575 Bytes
14a6c3d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 | {
"protocol_tag": "eval-protocol-v1.1",
"_comment": "Pre-registered battery (docs/eval_protocol.md section 4). Freeze artifact-backed members with their file sha256 (path points at the artifact FILE, e.g. runs/<id>/best.pt) BEFORE the T0 snapshot download; frozen_snapshot is filled at T0 with the raw MSH csv.gz path, sha256, and S3 ETag. run_frozen_eval.py refuses real mode while any of these are null. Member paths are DATA_ROOT-relative (resolved by run_frozen_eval.resolve_artifact_path against <MTGA_DATA_ROOT>/foundation; the pinned checkpoints are published in the brianward92/draftfm Hugging Face model repo, local mirror /opt/brianward/dat/mtga/foundation/runs/). The sha256 is the authoritative anchor, so the battery is portable across boxes -- set MTGA_DATA_ROOT to the unpacked-weights root (T2.7). draftfm members carry condition={wr_id:33, games_id:6} for deployment-mode scoring (matches scripts/export_draftfm.py's serving default, ~0.66 win rate / 1000-games bucket); member_frames also always computes 'human' mode (no override) for the same member.",
"_pending": "Post-day-1 rows (per-set MSH ceiling, F-full fine-tuned on MSH) are pre-registered in docs/eval_protocol.md section 4.6 but their artifacts don't exist yet by design (trained only after this zero-shot battery runs against the frozen MSH snapshot) -- they get appended in a follow-up commit, never added or iterated before T0.",
"calibration": {
"temperature": 1.28,
"fit_run": "20260704_135822_f_dev",
"fit_script": "scripts/fit_dev_temperature.py",
"selection_rule": "argmin mean dev-trio (BRO/TMT/SOS) log-loss over a 0.30:3.00:0.02 grid; never fitted on MSH",
"dev_mean_ece_at_t1": 0.0891,
"dev_mean_ece_at_frozen_t": 0.0250,
"frozen_before_t0": true,
"applied_to": ["deployment", "human"]
},
"frozen_snapshot": {
"path": "raw/draft_data_public.MSH.PremierDraft.csv.gz",
"sha256": "013df16b8994534f69ed63c87ab684acafc5f4cbe82982264b0fc111dbb2183a",
"etag": "\"252007adfbca6f026766527823ffa6d5-8\""
},
"models": [
{
"name": "baseline-random",
"kind": "baseline-random"
},
{
"name": "baseline-rarity",
"kind": "baseline-rarity"
},
{
"name": "f-full",
"kind": "draftfm",
"run": "20260705_110743_f_full",
"path": "runs/20260705_110743_f_full/best.pt",
"sha256": "61f87c1a0f84daa1baf254f1386deb819edd59d7f175aa3621a9cb620eee6167",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "headline model: final recipe (set_ctx=False, winning ablation), all 31 sets, no dev holdout. Its own dev numbers are meaningless and never quoted."
},
{
"name": "f-dev",
"kind": "draftfm",
"run": "20260704_135822_f_dev",
"path": "runs/20260704_135822_f_dev/best.pt",
"sha256": "c10e93e6e5022c217d03f41ba3f0e5975a2d460b96b59337d33ac422a314741d",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "universe minus dev trio (BRO/TMT/SOS); doubles as scaling rung S27 (F-dev universe)."
},
{
"name": "s1",
"kind": "draftfm",
"run": "20260704_154741_s1",
"path": "runs/20260704_154741_s1/best.pt",
"sha256": "8b2b8cdf737fc5863e951563ddc11630b4c2b9ad78d8180fc225ee15cc261448",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S1={NEO} (Bertram anchor)."
},
{
"name": "s2",
"kind": "draftfm",
"run": "20260704_161450_s2",
"path": "runs/20260704_161450_s2/best.pt",
"sha256": "0be4890dc410025f1873aadaf9b9ffc5d353c20f356c2819a7abbd1e280797e4",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S2=S1+DSK."
},
{
"name": "s2b",
"kind": "draftfm",
"run": "20260704_171827_s2b",
"path": "runs/20260704_171827_s2b/best.pt",
"sha256": "79e2f953788f97449e00a322f11e76c11f914a908ece60e7562a801d47231481",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling composition probe S2b={MOM,TDM}."
},
{
"name": "s4",
"kind": "draftfm",
"run": "20260704_175554_s4",
"path": "runs/20260704_175554_s4/best.pt",
"sha256": "5b35bb209b3daaa03e4723e7d56e4ed2a222b89a130acba0da3622cf8f100068",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S4=S2+DMU,FIN."
},
{
"name": "s4b",
"kind": "draftfm",
"run": "20260704_191659_s4b",
"path": "runs/20260704_191659_s4b/best.pt",
"sha256": "2382932cb470ff98d575e37a6ebd9963bf83355261ebdd56520ade4ad1aeb176",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling composition probe S4b={STX,SNC,OTJ,TLA}."
},
{
"name": "s8",
"kind": "draftfm",
"run": "20260704_202620_s8",
"path": "runs/20260704_202620_s8/best.pt",
"sha256": "ef23bff70621a8278d1dbeb9387fa90ec6656bcc3dcf22c4e969b1f3df416d1a",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S8=S4+STX,MOM,BLB,TLA."
},
{
"name": "s16",
"kind": "draftfm",
"run": "20260704_213310_s16",
"path": "runs/20260704_213310_s16/best.pt",
"sha256": "87c9e3e9ea36281c5c8e5cf582ac513fc24f7a942740442b45bffa67a89ff2e4",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "scaling rung S16=S8+AFR,SNC,ONE,LTR,WOE,MKM,OTJ,EOE."
},
{
"name": "a-notext",
"kind": "draftfm",
"run": "20260704_135710_a_notext",
"path": "runs/20260704_135710_a_notext/best.pt",
"sha256": "824623d0aed62f8f1d039d8d195cfcd3db7a86f4da7ca06bc2d123dd32e50861",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: no oracle-text embedding (structured 391-d only). Pre-registered for the UB-shift analysis."
},
{
"name": "a-noctx",
"kind": "draftfm",
"run": "20260704_160422_a_noctx",
"path": "runs/20260704_160422_a_noctx/best.pt",
"sha256": "d3eb5e65b7a6b2446877c18c741138e0122c6cd1fef1543c4d557a8b419ea804",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: set_ctx=False (no set-context cross-attention). Won the dev-trio ablation sweep; F-full uses this recipe (F-dev keeps the set-context tower -- see method.tex sec:arch's disclosure paragraph)."
},
{
"name": "a-proportional",
"kind": "draftfm",
"run": "20260704_201946_a_proportional",
"path": "runs/20260704_201946_a_proportional/best.pt",
"sha256": "5463d3c9262148cc2127ae1d1e60716bc7a36f251c71dd2a4115dd873a038658",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: proportional (not sqrt) per-set sampling."
},
{
"name": "a-topfilter",
"kind": "draftfm",
"run": "20260705_030215_a_topfilter",
"path": "runs/20260705_030215_a_topfilter/best.pt",
"sha256": "f0240ba209eda57094b8c3afafd982fd1280afac3542b9bcc350c77b34449fd8",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: train on expert (top-skill) picks only, no skill conditioning."
},
{
"name": "a-noub",
"kind": "draftfm",
"run": "20260705_164029_a_noUB",
"path": "runs/20260705_164029_a_noUB/best.pt",
"sha256": "9d9107563b234d3715567f844be2309e13435c368a69dc2a19eb0d2b2ff8d790",
"condition": {"wr_id": 33, "games_id": 6},
"_note": "ablation: three licensed-IP training sets (LTR/FIN/TLA) removed from A-noctx (the winning architecture). Pre-registered for the UB-shift analysis; clean comparator is the A-noctx row, not Full."
}
]
}
|