File size: 25,399 Bytes
1c6dfb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ad9e5a2
 
 
 
 
 
1c6dfb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
972b64f
 
 
 
 
1c6dfb9
972b64f
1c6dfb9
 
972b64f
 
 
 
 
 
 
 
 
1c6dfb9
972b64f
 
1c6dfb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
c66bf84
1c6dfb9
 
 
 
 
ad9e5a2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1c6dfb9
 
ad9e5a2
1c6dfb9
 
 
 
 
 
 
 
fbe96e3
 
 
 
 
 
 
 
 
 
 
1c6dfb9
 
 
 
 
 
 
 
 
 
 
 
 
f6d90c6
 
 
 
 
 
 
 
 
1c6dfb9
 
 
 
 
 
 
ad9e5a2
 
 
 
 
 
1c6dfb9
 
 
 
 
 
 
 
 
972b64f
 
 
 
 
 
1c6dfb9
 
 
 
 
 
 
 
 
 
972b64f
 
 
1c6dfb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
"""Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them.

Written before any score exists, and it refuses to deviate from the pin:

* `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat
  template -- a base model with a chat template would be an unearned capability (§3.9).
* PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**.
  The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot
  be tuned per task later either.
* SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead.
* One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the
  moment it finishes so an interruption never loses a completed task (§5 Phase 6).
* A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus
  a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time
  is not acceptable. Nothing is published if the smoke pass fails.

No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and
the token never enters a log line or a published artifact.
"""

import argparse
import json
import os
import re
import shutil
import signal
import subprocess
import sys
import time

PINNED_LM_EVAL = "0.4.13"
# Every row here is the harness's own task id, not a name invented for this script.
TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1",
         "truthfulqa_mc2", "winogrande", "gsm8k"]
# Chance level of the *metric being reported*, from the number of answer options in the task, not from a
# remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at.
CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"),
          "arc_easy": ("0.25-0.33", "items mix 3 and 4 options"),
          "hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"),
          "piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"),
          "truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"),
          "truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"),
          "gsm8k": ("0.00", "free-form exact match")}


# The metric column for each row, copied from the table docs/05-eval-plan.md section 2 froze. Naming it
# here is what stops "which of GSM8K's two numbers did we publish?" from being decided after the scores
# exist, which is the whole reason the plan was written down first (review E-045/10).
PRIMARY_METRIC = {"arc_challenge": "acc,none", "arc_easy": "acc,none", "hellaswag": "acc,none",
                 "mmlu": "acc,none", "piqa": "acc,none", "truthfulqa_mc1": "acc,none",
                 "truthfulqa_mc2": "acc,none", "winogrande": "acc,none",
                 "gsm8k": "exact_match,flexible-extract"}
# The stack Gate 3 measured and Gate 5 loaded against. A silently shadowed transformers in ~/.local changes
# tokenizer and generate behaviour, and every number in this table would then describe a different
# environment than the one the card claims (E-045/12).
EXPECTED_TRANSFORMERS = os.environ.get("EXPECTED_TRANSFORMERS", "5.0.0")


def sh(argv, label, timeout, env=None):
    """Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would
    report nothing at all if the platform took the instance."""
    print("=== " + label, flush=True)
    t0 = time.time()
    e = dict(os.environ)
    e["PYTHONUNBUFFERED"] = "1"
    e.update(env or {})
    import threading
    # start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the
    # notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1).
    p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
                         env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)),
                         start_new_session=True)
    killed = []

    def _kill():
        killed.append(True)
        try:
            os.killpg(os.getpgid(p.pid), signal.SIGTERM)
        except Exception:
            p.kill()

    timer = threading.Timer(timeout, _kill)
    timer.daemon = True
    timer.start()
    keep, lines = [], []
    for line in p.stdout:
        line = line.rstrip("\n")
        lines.append(line)
        keep.append(line)
        del keep[:-60]            # only the printed tail is bounded; the returned text is the whole run
        print("   |", line[:260], flush=True)
    timer.cancel()
    rc = p.wait()
    print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0),
          flush=True)
    return rc, "\n".join(lines)


def tf_version():
    """Which transformers the harness will actually import, and from where -- see the --no-deps note."""
    rc, out = sh([sys.executable, "-c", "import transformers, os; "
                  "print(transformers.__version__, os.path.dirname(transformers.__file__))"],
                 "probe_transformers", 180)
    for line in out.splitlines():
        m = re.match(r"^\s*\|?\s*(\d[\w.]*)\s+(/\S+)", line)
        if m:
            return m.group(1), m.group(2)
    return "?", "?"


def check_transformers(when):
    """Assert the import the harness will get -- on every path, installing or not installing."""
    v, where = tf_version()
    print("transformers", v, "from", where, "at", when, flush=True)
    if v != EXPECTED_TRANSFORMERS:
        raise SystemExit("transformers is %s (%s), not the %s Gate 3 measured and Gate 5 loaded: the eval "
                         "would describe an environment the model was not verified in"
                         % (v, where, EXPECTED_TRANSFORMERS))
    if "/.local/" in where:
        raise SystemExit("transformers is being imported from a user dir (%s), which shadows the image copy"
                         % where)
    return v


def install_harness():
    """Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages:
    transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would
    resolve a different one."""
    rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120)
    have = ""
    for line in out.splitlines():
        if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line):
            have = line.strip().strip("|").strip()
    if have == PINNED_LM_EVAL:
        print("lm_eval", have, "already present", flush=True)
        return have          # main() still calls check_transformers(): an early return used to skip it
    print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL),
          flush=True)
    before = check_transformers("before install")
    # Deps ARE installed, and that is measured, not assumed: --no-deps left the harness unimportable
    # (`lm_eval/api/metrics.py:11` does `import sacrebleu` at module scope; probe v1, E-047), while a
    # resolving install moved exactly two packages -- evaluate 0.4.6 and sacrebleu 2.6.0, both new and
    # both into ~/.local -- and left transformers at 5.0.0 in the image (probe v2). The assertion below
    # is what actually protects the pin; the flag only protected a fear.
    rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet",
                "lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800)
    if rc != 0:
        raise SystemExit("could not install the pinned harness; refusing to eval on a different one")
    after = tf_version()[0]
    if before != after:
        raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run "
                         "on the stack Gate 3 measured" % (before, after))
    rc, out = sh([sys.executable, "-c", "import lm_eval, os; "
                                        "print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"],
                 "verify_lm_eval", 300)
    got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)]
    if not got:
        raise SystemExit("lm_eval is not importable at the pinned version after install: "
                         + out[-400:])
    return PINNED_LM_EVAL


def run_task(model, task, shots, outdir, limit, gpu_h):
    """One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column."""
    argv = [sys.executable, "-m", "lm_eval", "--model", "hf",
            "--model_args", ",".join(["pretrained=" + model, "dtype=float16",
                                     "trust_remote_code=False"]),
            "--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir,
            "--log_samples"]
    if shots is not None:
        argv += ["--num_fewshot", str(shots)]
    if limit:
        argv += ["--limit", str(limit)]
    grc, gout = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60)
    if grc != 0:
        # The return code used to be discarded. With no card visible the harness falls back to CPU, and
        # "still running" then reads like "still slow" for weeks (E-045, minor).
        raise SystemExit("nvidia-smi failed (rc %s) before %s: refusing to evaluate on CPU -- %s"
                         % (grc, task, gout[-200:]))
    rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots),
                 timeout=gpu_h * 3600)
    return rc, " ".join(argv)


def harvest(outdir, task):
    """Find the rows the harness just wrote. It nests results under <timestamp>/<task>/results.json, so
    glob rather than assume a layout that changes between releases."""
    hits = []
    for root, _d, files in os.walk(outdir):
        for fn in files:
            if fn.startswith("results") and fn.endswith(".json"):
                try:
                    j = json.load(open(os.path.join(root, fn)))
                except Exception:
                    continue
                if task in (j.get("results") or {}):
                    hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j))
    if not hits:
        return {}
    hits.sort()
    path, j = hits[-1][1], hits[-1][2]
    r = j["results"][task]
    keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"}
    cfg = j.get("config") or {}
    # `config["num_fewshot"]` is null unless the flag was passed, so reading only it left the PRIMARY
    # column -- the one the report quotes -- with no shot count at all. What was actually used is in
    # `configs[task]`, and `n_input` is the denominator beside the score (E-045/8).
    per = (j.get("configs") or {}).get(task) or {}
    shots = per.get("num_fewshot")
    if shots is None:
        shots = cfg.get("num_fewshot")

    def one(x):
        return x[0] if isinstance(x, list) and x else x

    # `--log_samples` writes samples_<task>_<hash>.json beside results.json, and its own row count is the
    # denominator that can be checked after the fact -- 0.4.13's results rows do not reliably carry
    # n_input, and a task that silently resolved to a subset otherwise looks exactly like a score.
    n_logged = None
    base = os.path.dirname(path)
    for root, _d, files in os.walk(base):        # subtask groups nest; a flat listdir missed them
        for fn in sorted(files):
            if not (fn.startswith("samples") and fn.endswith(".json")):
                continue
            try:
                sj = json.load(open(os.path.join(root, fn)))
            except Exception:
                continue
            sm = sj.get("samples")
            # 0.4.x writes {"samples": {group: [rows]}} for grouped tasks (mmlu's 57 subjects) and a bare
            # list for the rest. len() on the dict is the number of *groups*, which for mmlu would have
            # been 1 and made every denominator check below pass while counting nothing (E-046/4).
            if isinstance(sm, dict):
                n_logged = sum(len(v) for v in sm.values() if isinstance(v, list))
            elif isinstance(sm, list):
                n_logged = len(sm)
            if n_logged:
                break
        if n_logged:
            break

    return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}),
            "shots": shots, "n_input": r.get("n_input"), "n_effective": r.get("n_effective"),
            "n_logged": n_logged,
            "limit": cfg.get("limit"), "model": one(cfg.get("model")),
            "model_args": cfg.get("model_args") if isinstance(cfg.get("model_args"), (dict, str)) else None,
            "dataset_path": per.get("dataset_path"), "dataset_name": per.get("dataset_name"),
            "split": per.get("split") or (per.get("test_args") or {}).get("split"),
            "seed": cfg.get("seed"), "dtype": cfg.get("dtype"), "repeats": cfg.get("repeats"),
            "config_fewshot": cfg.get("num_fewshot"), "date": j.get("date"),
            "git_hash": j.get("git_hash"),
            "lm_eval_version": (j.get("lm_eval") or {}).get("version") if isinstance(j.get("lm_eval"), dict) else None}


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1")
    ap.add_argument("--repo", default="", help="where to push results (defaults to --model)")
    ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else")
    ap.add_argument("--full-gpu-hours", type=float, default=4.0,
                    help="per-task ceiling for the full table; MMLU is by far the longest")
    a = ap.parse_args()
    if not a.model:
        raise SystemExit("--model is required (the public Hub repo, loaded clean-room)")

    import ounce100m_credentials
    print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
    ver = install_harness()
    check_transformers("after harness resolved")
    os.environ["CUDA_VISIBLE_DEVICES"] = "0"      # one card, no gather path that could reorder exemplars
    root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results")
    shutil.rmtree(root, ignore_errors=True)
    os.makedirs(root, exist_ok=True)

    # Confirm the task ids resolve before any GPU time goes into them (C5). `lm_eval --tasks list` is NOT
    # how: 0.4.13's CLI sends the word to task resolution and raises `Tasks not found: list` (E-048), so
    # the registry is asked directly. Either route is fine for a human; only the API is a check.
    rc, out = sh([sys.executable, "-c",
                  "from lm_eval.tasks import TaskManager;"
                  "import json,sys;"
                  "tm=TaskManager();"
                  "print(json.dumps({'n': len(tm.task_index),"
                  " 'present': {t: (t in tm.task_index) for t in %r}}))" % (TASKS,)],
                 "list_tasks", 900)
    listed = {}
    for line in (out or "").splitlines():
        if line.startswith("{"):
            try:
                listed = json.loads(line)
            except Exception:
                listed = {}
            break
    if not listed or listed.get("n", 0) < 50:
        raise SystemExit("could not read the harness task registry; refusing to run %s unverified -- %r"
                         % (PINNED_LM_EVAL, (out or "")[-500:]))
    unknown = [t for t in TASKS if not listed["present"].get(t)]
    if unknown:
        raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown))
    print("task ids checked against the harness:", len(TASKS), "requested,", listed["n"], "registered",
          flush=True)

    smoke = {}
    for t in TASKS:
        rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5)
        smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}),
                    "cmd": cmd}
        print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True)
    # The smoke pass is also where the pre-registered metric name gets checked. PRIMARY_METRIC could not
    # be verified from the pinned YAMLs -- reading a task config gives the metric *name* (`acc`) but the
    # results key is `<name>,<aggregation>` and the aggregation half of that chain is not settled by the
    # config files (E-049) -- so the alternative to checking here is discovering it after the six-hour
    # columns have run. The metric must not be chosen then; the fix is the constant, not the row.
    bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"]
           or PRIMARY_METRIC[t] not in smoke[t]["rows"]]
    for t in TASKS:
        if PRIMARY_METRIC[t] not in smoke[t]["rows"] and smoke[t]["rows"]:
            print("SMOKE %s metric %r absent; keys are %s" % (
                t, PRIMARY_METRIC[t], sorted(smoke[t]["rows"])[:10]), flush=True)
    print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True)
    if bad or a.smoke_only:
        json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str)
        raise SystemExit(3 if bad else 0)

    table, push_fail = {}, []
    for t in TASKS:
        rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver}
        for shots, key in ((None, "primary"), (5, "secondary_5shot")):
            d = os.path.join(root, key, t)
            rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours)
            h = harvest(d, t)
            rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd,
                        "shots": (h.get("shots") if h.get("shots") is not None else 0),
                        # Six of the nine YAMLs declare no num_fewshot (docs/05 §7, re-confirmed by E-049),
                        # and the PRIMARY column deliberately passes no flag, so those six rows are
                        # 0-shot. Recording `shots: 0` with its reason is what keeps a 0-shot number from
                        # being read as the few-shot figure D-005 warned about -- the design of the columns
                        # is frozen, the description of what happened is not allowed to be vague.
                        "shots_source": ("harness config" if h.get("shots") is not None
                                         else "harness default, none declared, no flag passed"),
                        "harness_fewshot": h.get("config_fewshot"),
                        "n_input": h.get("n_input"), "n_effective": h.get("n_effective"),
                        "n_logged": h.get("n_logged"),
                        "limit": h.get("limit"), "model": h.get("model"), "dtype": h.get("dtype"),
                        "seed": h.get("seed"), "git_hash": h.get("git_hash"),
                        "dataset_path": h.get("dataset_path"), "split": h.get("split"),
                        "results_file": h.get("file"), "primary_metric": PRIMARY_METRIC[t],
                        "primary": (h.get("rows") or {}).get(PRIMARY_METRIC[t])}
            if PRIMARY_METRIC[t] not in (h.get("rows") or {}):
                # The metric name is the one docs/05 §2 froze; if the harness writes a different key, the
                # row must say so rather than publish a score of None (E-048 -- reading it out of the
                # pinned YAMLs proved awkward, so it is asserted against the artifact that matters).
                rec[key]["status"] = "PRIMARY METRIC %r ABSENT, have %s" % (
                    PRIMARY_METRIC[t], sorted(h.get("rows") or {})[:8])
            if rc != 0 or not h.get("rows"):
                # A cell pushed as `rows: {}` reads as "scored zero" to anyone holding only the JSON.
                rec[key]["status"] = "FAILED rc=%s rows=%s" % (rc, len(h.get("rows") or {}))
            elif h.get("limit") not in (None, 0):
                raise SystemExit("%s/%s ran with --limit %s: a subset must not publish as a score"
                                 % (t, key, h.get("limit")))
            elif not (h.get("n_input") or h.get("n_logged")):
                # No denominator beside a score is no denominator at all: say so in the record.
                rec[key]["status"] = "NO ROW COUNT RECORDED"
            elif key == "secondary_5shot" and h.get("shots") != 5:
                # The PRIMARY column legitimately records null for six of the nine tasks -- those YAMLs
                # declare no num_fewshot and we pass no flag, which is the frozen definition of the
                # column (docs/05 section 2 and 7). The 5-shot column is the one where we asked, so a
                # null or a different number there means the flag did not reach the harness.
                rec[key]["status"] = "SHOT COUNT NOT 5: %r" % (h.get("shots"),)
            print("DONE %s %s rc %s rows %s" % (t, key, rc,
                  json.dumps(rec[key]["rows"], default=str)[:200]), flush=True)
            table[t] = rec
            json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str)
            if not push(a, root):              # after every task: an interruption keeps what finished
                push_fail.append("after " + t)
    if not push(a, root):
        push_fail.append("final")
    def ok(t, key):
        r = table.get(t, {}).get(key, {})
        # `status` is set for a failed cell and for a cell with no denominator; ignoring it here is how
        # the verdict printed 18/18 over a table of zeros (E-046/5).
        return bool(r.get("rows")) and r.get("rc") == 0 and not r.get("status")

    # Both columns and both exit codes. `primary` rows alone would report a full house with an entirely
    # empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2).
    missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)]
    # The two columns run the same task on the same split with no limit, so a differing denominator means
    # one of them resolved to a smaller subset -- a real-looking number for less data (E-045/9).
    for t in TASKS:
        p1 = table.get(t, {}).get("primary", {})
        p2 = table.get(t, {}).get("secondary_5shot", {})
        n1 = p1.get("n_input") or p1.get("n_logged")
        n2 = p2.get("n_input") or p2.get("n_logged")
        if n1 and n2 and n1 != n2:
            missing.append("%s rows %s != %s between columns" % (t, n1, n2))
    missing += push_fail
    # Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2
    # are separate task ids in the harness (docs/05-eval-plan.md §2).
    print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % (
        2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True)
    raise SystemExit(4 if missing else 0)


def push(a, root):
    """Publish into the model repo, which is where a reader expects the evaluation to live (§8)."""
    repo = a.repo or a.model
    if not repo:
        print("(no --repo and no --model-derived target; results stay local)", flush=True)
        return False
    import ounce100m_credentials
    ounce100m_credentials.install()
    tok = os.environ.get("HF_TOKEN")
    if tok is None:
        print("FATAL: no token, refusing to claim results were published", flush=True)
        return False
    try:
        from huggingface_hub import HfApi
        rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root,
                                            path_in_repo="eval", commit_message="Phase 6 results",
                                            # The --limit 5 smoke rows must not sit in the public repo beside
                                            # the real table looking like results (C3).
                                            ignore_patterns=["smoke*", "smoke/*", "smoke.json"],
                                            commit_description="lm-eval 0.4.13, task-default shots plus a "
                                                               "uniform 5-shot column, per docs/05-eval-plan.md")
        print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True)
        # E-031's rule applied here: `upload_folder` returning is not evidence. `results.json` is the
        # artifact the report cites, so re-fetch it over `resolve/` with no token and compare bytes.
        try:
            import hashlib as _h
            import urllib.request as _u
            want = _h.sha256(open(os.path.join(root, "results.json"), "rb").read()).hexdigest()
            with _u.urlopen("https://huggingface.co/%s/resolve/main/eval/results.json?cb=%d"
                            % (repo, int(time.time())), timeout=300) as fh:
                got = _h.sha256(fh.read()).hexdigest()
            print("EVAL_READBACK sha_ok", got == want, flush=True)
            return got == want
        except Exception as e2:
            print("EVAL_READBACK_FAILED", type(e2).__name__, str(e2)[:160], flush=True)
            return False
    except Exception as e:
        print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True)
        return False


if __name__ == "__main__":
    main()