"""Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them. Written before any score exists, and it refuses to deviate from the pin: * `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat template -- a base model with a chat template would be an unearned capability (§3.9). * PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**. The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot be tuned per task later either. * SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead. * One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the moment it finishes so an interruption never loses a completed task (§5 Phase 6). * A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time is not acceptable. Nothing is published if the smoke pass fails. No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and the token never enters a log line or a published artifact. """ import argparse import json import os import re import shutil import signal import subprocess import sys import time PINNED_LM_EVAL = "0.4.13" # Every row here is the harness's own task id, not a name invented for this script. TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1", "truthfulqa_mc2", "winogrande", "gsm8k"] # Chance level of the *metric being reported*, from the number of answer options in the task, not from a # remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at. CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"), "arc_easy": ("0.25-0.33", "items mix 3 and 4 options"), "hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"), "piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"), "truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"), "truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"), "gsm8k": ("0.00", "free-form exact match")} # The metric column for each row, copied from the table docs/05-eval-plan.md section 2 froze. Naming it # here is what stops "which of GSM8K's two numbers did we publish?" from being decided after the scores # exist, which is the whole reason the plan was written down first (review E-045/10). PRIMARY_METRIC = {"arc_challenge": "acc,none", "arc_easy": "acc,none", "hellaswag": "acc,none", "mmlu": "acc,none", "piqa": "acc,none", "truthfulqa_mc1": "acc,none", "truthfulqa_mc2": "acc,none", "winogrande": "acc,none", "gsm8k": "exact_match,flexible-extract"} # The stack Gate 3 measured and Gate 5 loaded against. A silently shadowed transformers in ~/.local changes # tokenizer and generate behaviour, and every number in this table would then describe a different # environment than the one the card claims (E-045/12). EXPECTED_TRANSFORMERS = os.environ.get("EXPECTED_TRANSFORMERS", "5.0.0") def sh(argv, label, timeout, env=None): """Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would report nothing at all if the platform took the instance.""" print("=== " + label, flush=True) t0 = time.time() e = dict(os.environ) e["PYTHONUNBUFFERED"] = "1" e.update(env or {}) import threading # start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the # notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1). p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)), start_new_session=True) killed = [] def _kill(): killed.append(True) try: os.killpg(os.getpgid(p.pid), signal.SIGTERM) except Exception: p.kill() timer = threading.Timer(timeout, _kill) timer.daemon = True timer.start() keep, lines = [], [] for line in p.stdout: line = line.rstrip("\n") lines.append(line) keep.append(line) del keep[:-60] # only the printed tail is bounded; the returned text is the whole run print(" |", line[:260], flush=True) timer.cancel() rc = p.wait() print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0), flush=True) return rc, "\n".join(lines) def tf_version(): """Which transformers the harness will actually import, and from where -- see the --no-deps note.""" rc, out = sh([sys.executable, "-c", "import transformers, os; " "print(transformers.__version__, os.path.dirname(transformers.__file__))"], "probe_transformers", 180) for line in out.splitlines(): m = re.match(r"^\s*\|?\s*(\d[\w.]*)\s+(/\S+)", line) if m: return m.group(1), m.group(2) return "?", "?" def check_transformers(when): """Assert the import the harness will get -- on every path, installing or not installing.""" v, where = tf_version() print("transformers", v, "from", where, "at", when, flush=True) if v != EXPECTED_TRANSFORMERS: raise SystemExit("transformers is %s (%s), not the %s Gate 3 measured and Gate 5 loaded: the eval " "would describe an environment the model was not verified in" % (v, where, EXPECTED_TRANSFORMERS)) if "/.local/" in where: raise SystemExit("transformers is being imported from a user dir (%s), which shadows the image copy" % where) return v def install_harness(): """Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages: transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would resolve a different one.""" rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120) have = "" for line in out.splitlines(): if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line): have = line.strip().strip("|").strip() if have == PINNED_LM_EVAL: print("lm_eval", have, "already present", flush=True) return have # main() still calls check_transformers(): an early return used to skip it print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL), flush=True) before = check_transformers("before install") # Deps ARE installed, and that is measured, not assumed: --no-deps left the harness unimportable # (`lm_eval/api/metrics.py:11` does `import sacrebleu` at module scope; probe v1, E-047), while a # resolving install moved exactly two packages -- evaluate 0.4.6 and sacrebleu 2.6.0, both new and # both into ~/.local -- and left transformers at 5.0.0 in the image (probe v2). The assertion below # is what actually protects the pin; the flag only protected a fear. rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet", "lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800) if rc != 0: raise SystemExit("could not install the pinned harness; refusing to eval on a different one") after = tf_version()[0] if before != after: raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run " "on the stack Gate 3 measured" % (before, after)) rc, out = sh([sys.executable, "-c", "import lm_eval, os; " "print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"], "verify_lm_eval", 300) got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)] if not got: raise SystemExit("lm_eval is not importable at the pinned version after install: " + out[-400:]) return PINNED_LM_EVAL def run_task(model, task, shots, outdir, limit, gpu_h): """One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column.""" argv = [sys.executable, "-m", "lm_eval", "--model", "hf", "--model_args", ",".join(["pretrained=" + model, "dtype=float16", "trust_remote_code=False"]), "--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir, "--log_samples"] if shots is not None: argv += ["--num_fewshot", str(shots)] if limit: argv += ["--limit", str(limit)] grc, gout = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60) if grc != 0: # The return code used to be discarded. With no card visible the harness falls back to CPU, and # "still running" then reads like "still slow" for weeks (E-045, minor). raise SystemExit("nvidia-smi failed (rc %s) before %s: refusing to evaluate on CPU -- %s" % (grc, task, gout[-200:])) rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots), timeout=gpu_h * 3600) return rc, " ".join(argv) def harvest(outdir, task): """Find the rows the harness just wrote. It nests results under //results.json, so glob rather than assume a layout that changes between releases.""" hits = [] for root, _d, files in os.walk(outdir): for fn in files: if fn.startswith("results") and fn.endswith(".json"): try: j = json.load(open(os.path.join(root, fn))) except Exception: continue if task in (j.get("results") or {}): hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j)) if not hits: return {} hits.sort() path, j = hits[-1][1], hits[-1][2] r = j["results"][task] keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"} cfg = j.get("config") or {} # `config["num_fewshot"]` is null unless the flag was passed, so reading only it left the PRIMARY # column -- the one the report quotes -- with no shot count at all. What was actually used is in # `configs[task]`, and `n_input` is the denominator beside the score (E-045/8). per = (j.get("configs") or {}).get(task) or {} shots = per.get("num_fewshot") if shots is None: shots = cfg.get("num_fewshot") def one(x): return x[0] if isinstance(x, list) and x else x # `--log_samples` writes samples__.json beside results.json, and its own row count is the # denominator that can be checked after the fact -- 0.4.13's results rows do not reliably carry # n_input, and a task that silently resolved to a subset otherwise looks exactly like a score. n_logged = None base = os.path.dirname(path) for root, _d, files in os.walk(base): # subtask groups nest; a flat listdir missed them for fn in sorted(files): if not (fn.startswith("samples") and fn.endswith(".json")): continue try: sj = json.load(open(os.path.join(root, fn))) except Exception: continue sm = sj.get("samples") # 0.4.x writes {"samples": {group: [rows]}} for grouped tasks (mmlu's 57 subjects) and a bare # list for the rest. len() on the dict is the number of *groups*, which for mmlu would have # been 1 and made every denominator check below pass while counting nothing (E-046/4). if isinstance(sm, dict): n_logged = sum(len(v) for v in sm.values() if isinstance(v, list)) elif isinstance(sm, list): n_logged = len(sm) if n_logged: break if n_logged: break return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}), "shots": shots, "n_input": r.get("n_input"), "n_effective": r.get("n_effective"), "n_logged": n_logged, "limit": cfg.get("limit"), "model": one(cfg.get("model")), "model_args": cfg.get("model_args") if isinstance(cfg.get("model_args"), (dict, str)) else None, "dataset_path": per.get("dataset_path"), "dataset_name": per.get("dataset_name"), "split": per.get("split") or (per.get("test_args") or {}).get("split"), "seed": cfg.get("seed"), "dtype": cfg.get("dtype"), "repeats": cfg.get("repeats"), "config_fewshot": cfg.get("num_fewshot"), "date": j.get("date"), "git_hash": j.get("git_hash"), "lm_eval_version": (j.get("lm_eval") or {}).get("version") if isinstance(j.get("lm_eval"), dict) else None} def main(): ap = argparse.ArgumentParser() ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1") ap.add_argument("--repo", default="", help="where to push results (defaults to --model)") ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else") ap.add_argument("--full-gpu-hours", type=float, default=4.0, help="per-task ceiling for the full table; MMLU is by far the longest") a = ap.parse_args() if not a.model: raise SystemExit("--model is required (the public Hub repo, loaded clean-room)") import ounce100m_credentials print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True) ver = install_harness() check_transformers("after harness resolved") os.environ["CUDA_VISIBLE_DEVICES"] = "0" # one card, no gather path that could reorder exemplars root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results") shutil.rmtree(root, ignore_errors=True) os.makedirs(root, exist_ok=True) # Confirm the task ids resolve before any GPU time goes into them (C5). `lm_eval --tasks list` is NOT # how: 0.4.13's CLI sends the word to task resolution and raises `Tasks not found: list` (E-048), so # the registry is asked directly. Either route is fine for a human; only the API is a check. rc, out = sh([sys.executable, "-c", "from lm_eval.tasks import TaskManager;" "import json,sys;" "tm=TaskManager();" "print(json.dumps({'n': len(tm.task_index)," " 'present': {t: (t in tm.task_index) for t in %r}}))" % (TASKS,)], "list_tasks", 900) listed = {} for line in (out or "").splitlines(): if line.startswith("{"): try: listed = json.loads(line) except Exception: listed = {} break if not listed or listed.get("n", 0) < 50: raise SystemExit("could not read the harness task registry; refusing to run %s unverified -- %r" % (PINNED_LM_EVAL, (out or "")[-500:])) unknown = [t for t in TASKS if not listed["present"].get(t)] if unknown: raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown)) print("task ids checked against the harness:", len(TASKS), "requested,", listed["n"], "registered", flush=True) smoke = {} for t in TASKS: rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5) smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}), "cmd": cmd} print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True) # The smoke pass is also where the pre-registered metric name gets checked. PRIMARY_METRIC could not # be verified from the pinned YAMLs -- reading a task config gives the metric *name* (`acc`) but the # results key is `,` and the aggregation half of that chain is not settled by the # config files (E-049) -- so the alternative to checking here is discovering it after the six-hour # columns have run. The metric must not be chosen then; the fix is the constant, not the row. bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"] or PRIMARY_METRIC[t] not in smoke[t]["rows"]] for t in TASKS: if PRIMARY_METRIC[t] not in smoke[t]["rows"] and smoke[t]["rows"]: print("SMOKE %s metric %r absent; keys are %s" % ( t, PRIMARY_METRIC[t], sorted(smoke[t]["rows"])[:10]), flush=True) print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True) if bad or a.smoke_only: json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str) raise SystemExit(3 if bad else 0) table, push_fail = {}, [] for t in TASKS: rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver} for shots, key in ((None, "primary"), (5, "secondary_5shot")): d = os.path.join(root, key, t) rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours) h = harvest(d, t) rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd, "shots": (h.get("shots") if h.get("shots") is not None else 0), # Six of the nine YAMLs declare no num_fewshot (docs/05 §7, re-confirmed by E-049), # and the PRIMARY column deliberately passes no flag, so those six rows are # 0-shot. Recording `shots: 0` with its reason is what keeps a 0-shot number from # being read as the few-shot figure D-005 warned about -- the design of the columns # is frozen, the description of what happened is not allowed to be vague. "shots_source": ("harness config" if h.get("shots") is not None else "harness default, none declared, no flag passed"), "harness_fewshot": h.get("config_fewshot"), "n_input": h.get("n_input"), "n_effective": h.get("n_effective"), "n_logged": h.get("n_logged"), "limit": h.get("limit"), "model": h.get("model"), "dtype": h.get("dtype"), "seed": h.get("seed"), "git_hash": h.get("git_hash"), "dataset_path": h.get("dataset_path"), "split": h.get("split"), "results_file": h.get("file"), "primary_metric": PRIMARY_METRIC[t], "primary": (h.get("rows") or {}).get(PRIMARY_METRIC[t])} if PRIMARY_METRIC[t] not in (h.get("rows") or {}): # The metric name is the one docs/05 §2 froze; if the harness writes a different key, the # row must say so rather than publish a score of None (E-048 -- reading it out of the # pinned YAMLs proved awkward, so it is asserted against the artifact that matters). rec[key]["status"] = "PRIMARY METRIC %r ABSENT, have %s" % ( PRIMARY_METRIC[t], sorted(h.get("rows") or {})[:8]) if rc != 0 or not h.get("rows"): # A cell pushed as `rows: {}` reads as "scored zero" to anyone holding only the JSON. rec[key]["status"] = "FAILED rc=%s rows=%s" % (rc, len(h.get("rows") or {})) elif h.get("limit") not in (None, 0): raise SystemExit("%s/%s ran with --limit %s: a subset must not publish as a score" % (t, key, h.get("limit"))) elif not (h.get("n_input") or h.get("n_logged")): # No denominator beside a score is no denominator at all: say so in the record. rec[key]["status"] = "NO ROW COUNT RECORDED" elif key == "secondary_5shot" and h.get("shots") != 5: # The PRIMARY column legitimately records null for six of the nine tasks -- those YAMLs # declare no num_fewshot and we pass no flag, which is the frozen definition of the # column (docs/05 section 2 and 7). The 5-shot column is the one where we asked, so a # null or a different number there means the flag did not reach the harness. rec[key]["status"] = "SHOT COUNT NOT 5: %r" % (h.get("shots"),) print("DONE %s %s rc %s rows %s" % (t, key, rc, json.dumps(rec[key]["rows"], default=str)[:200]), flush=True) table[t] = rec json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str) if not push(a, root): # after every task: an interruption keeps what finished push_fail.append("after " + t) if not push(a, root): push_fail.append("final") def ok(t, key): r = table.get(t, {}).get(key, {}) # `status` is set for a failed cell and for a cell with no denominator; ignoring it here is how # the verdict printed 18/18 over a table of zeros (E-046/5). return bool(r.get("rows")) and r.get("rc") == 0 and not r.get("status") # Both columns and both exit codes. `primary` rows alone would report a full house with an entirely # empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2). missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)] # The two columns run the same task on the same split with no limit, so a differing denominator means # one of them resolved to a smaller subset -- a real-looking number for less data (E-045/9). for t in TASKS: p1 = table.get(t, {}).get("primary", {}) p2 = table.get(t, {}).get("secondary_5shot", {}) n1 = p1.get("n_input") or p1.get("n_logged") n2 = p2.get("n_input") or p2.get("n_logged") if n1 and n2 and n1 != n2: missing.append("%s rows %s != %s between columns" % (t, n1, n2)) missing += push_fail # Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2 # are separate task ids in the harness (docs/05-eval-plan.md §2). print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % ( 2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True) raise SystemExit(4 if missing else 0) def push(a, root): """Publish into the model repo, which is where a reader expects the evaluation to live (§8).""" repo = a.repo or a.model if not repo: print("(no --repo and no --model-derived target; results stay local)", flush=True) return False import ounce100m_credentials ounce100m_credentials.install() tok = os.environ.get("HF_TOKEN") if tok is None: print("FATAL: no token, refusing to claim results were published", flush=True) return False try: from huggingface_hub import HfApi rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root, path_in_repo="eval", commit_message="Phase 6 results", # The --limit 5 smoke rows must not sit in the public repo beside # the real table looking like results (C3). ignore_patterns=["smoke*", "smoke/*", "smoke.json"], commit_description="lm-eval 0.4.13, task-default shots plus a " "uniform 5-shot column, per docs/05-eval-plan.md") print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True) # E-031's rule applied here: `upload_folder` returning is not evidence. `results.json` is the # artifact the report cites, so re-fetch it over `resolve/` with no token and compare bytes. try: import hashlib as _h import urllib.request as _u want = _h.sha256(open(os.path.join(root, "results.json"), "rb").read()).hexdigest() with _u.urlopen("https://huggingface.co/%s/resolve/main/eval/results.json?cb=%d" % (repo, int(time.time())), timeout=300) as fh: got = _h.sha256(fh.read()).hexdigest() print("EVAL_READBACK sha_ok", got == want, flush=True) return got == want except Exception as e2: print("EVAL_READBACK_FAILED", type(e2).__name__, str(e2)[:160], flush=True) return False except Exception as e: print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True) return False if __name__ == "__main__": main()