ounce100m-code / eval /run_benchmarks.py
Cion-lab's picture
run_benchmarks: publish the effective shot count with its provenance, so a 0-shot primary row can never be read as few-shot
f6d90c6 verified
Raw History Blame Contribute Delete
25.4 kB
"""Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them.
Written before any score exists, and it refuses to deviate from the pin:
* `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat
template -- a base model with a chat template would be an unearned capability (§3.9).
* PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**.
The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot
be tuned per task later either.
* SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead.
* One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the
moment it finishes so an interruption never loses a completed task (§5 Phase 6).
* A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus
a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time
is not acceptable. Nothing is published if the smoke pass fails.
No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and
the token never enters a log line or a published artifact.
"""
import argparse
import json
import os
import re
import shutil
import signal
import subprocess
import sys
import time
PINNED_LM_EVAL = "0.4.13"
# Every row here is the harness's own task id, not a name invented for this script.
TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1",
"truthfulqa_mc2", "winogrande", "gsm8k"]
# Chance level of the *metric being reported*, from the number of answer options in the task, not from a
# remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at.
CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"),
"arc_easy": ("0.25-0.33", "items mix 3 and 4 options"),
"hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"),
"piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"),
"truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"),
"truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"),
"gsm8k": ("0.00", "free-form exact match")}
# The metric column for each row, copied from the table docs/05-eval-plan.md section 2 froze. Naming it
# here is what stops "which of GSM8K's two numbers did we publish?" from being decided after the scores
# exist, which is the whole reason the plan was written down first (review E-045/10).
PRIMARY_METRIC = {"arc_challenge": "acc,none", "arc_easy": "acc,none", "hellaswag": "acc,none",
"mmlu": "acc,none", "piqa": "acc,none", "truthfulqa_mc1": "acc,none",
"truthfulqa_mc2": "acc,none", "winogrande": "acc,none",
"gsm8k": "exact_match,flexible-extract"}
# The stack Gate 3 measured and Gate 5 loaded against. A silently shadowed transformers in ~/.local changes
# tokenizer and generate behaviour, and every number in this table would then describe a different
# environment than the one the card claims (E-045/12).
EXPECTED_TRANSFORMERS = os.environ.get("EXPECTED_TRANSFORMERS", "5.0.0")
def sh(argv, label, timeout, env=None):
"""Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would
report nothing at all if the platform took the instance."""
print("=== " + label, flush=True)
t0 = time.time()
e = dict(os.environ)
e["PYTHONUNBUFFERED"] = "1"
e.update(env or {})
import threading
# start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the
# notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1).
p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)),
start_new_session=True)
killed = []
def _kill():
killed.append(True)
try:
os.killpg(os.getpgid(p.pid), signal.SIGTERM)
except Exception:
p.kill()
timer = threading.Timer(timeout, _kill)
timer.daemon = True
timer.start()
keep, lines = [], []
for line in p.stdout:
line = line.rstrip("\n")
lines.append(line)
keep.append(line)
del keep[:-60] # only the printed tail is bounded; the returned text is the whole run
print(" |", line[:260], flush=True)
timer.cancel()
rc = p.wait()
print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0),
flush=True)
return rc, "\n".join(lines)
def tf_version():
"""Which transformers the harness will actually import, and from where -- see the --no-deps note."""
rc, out = sh([sys.executable, "-c", "import transformers, os; "
"print(transformers.__version__, os.path.dirname(transformers.__file__))"],
"probe_transformers", 180)
for line in out.splitlines():
m = re.match(r"^\s*\|?\s*(\d[\w.]*)\s+(/\S+)", line)
if m:
return m.group(1), m.group(2)
return "?", "?"
def check_transformers(when):
"""Assert the import the harness will get -- on every path, installing or not installing."""
v, where = tf_version()
print("transformers", v, "from", where, "at", when, flush=True)
if v != EXPECTED_TRANSFORMERS:
raise SystemExit("transformers is %s (%s), not the %s Gate 3 measured and Gate 5 loaded: the eval "
"would describe an environment the model was not verified in"
% (v, where, EXPECTED_TRANSFORMERS))
if "/.local/" in where:
raise SystemExit("transformers is being imported from a user dir (%s), which shadows the image copy"
% where)
return v
def install_harness():
"""Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages:
transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would
resolve a different one."""
rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120)
have = ""
for line in out.splitlines():
if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line):
have = line.strip().strip("|").strip()
if have == PINNED_LM_EVAL:
print("lm_eval", have, "already present", flush=True)
return have # main() still calls check_transformers(): an early return used to skip it
print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL),
flush=True)
before = check_transformers("before install")
# Deps ARE installed, and that is measured, not assumed: --no-deps left the harness unimportable
# (`lm_eval/api/metrics.py:11` does `import sacrebleu` at module scope; probe v1, E-047), while a
# resolving install moved exactly two packages -- evaluate 0.4.6 and sacrebleu 2.6.0, both new and
# both into ~/.local -- and left transformers at 5.0.0 in the image (probe v2). The assertion below
# is what actually protects the pin; the flag only protected a fear.
rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet",
"lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800)
if rc != 0:
raise SystemExit("could not install the pinned harness; refusing to eval on a different one")
after = tf_version()[0]
if before != after:
raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run "
"on the stack Gate 3 measured" % (before, after))
rc, out = sh([sys.executable, "-c", "import lm_eval, os; "
"print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"],
"verify_lm_eval", 300)
got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)]
if not got:
raise SystemExit("lm_eval is not importable at the pinned version after install: "
+ out[-400:])
return PINNED_LM_EVAL
def run_task(model, task, shots, outdir, limit, gpu_h):
"""One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column."""
argv = [sys.executable, "-m", "lm_eval", "--model", "hf",
"--model_args", ",".join(["pretrained=" + model, "dtype=float16",
"trust_remote_code=False"]),
"--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir,
"--log_samples"]
if shots is not None:
argv += ["--num_fewshot", str(shots)]
if limit:
argv += ["--limit", str(limit)]
grc, gout = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60)
if grc != 0:
# The return code used to be discarded. With no card visible the harness falls back to CPU, and
# "still running" then reads like "still slow" for weeks (E-045, minor).
raise SystemExit("nvidia-smi failed (rc %s) before %s: refusing to evaluate on CPU -- %s"
% (grc, task, gout[-200:]))
rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots),
timeout=gpu_h * 3600)
return rc, " ".join(argv)
def harvest(outdir, task):
"""Find the rows the harness just wrote. It nests results under <timestamp>/<task>/results.json, so
glob rather than assume a layout that changes between releases."""
hits = []
for root, _d, files in os.walk(outdir):
for fn in files:
if fn.startswith("results") and fn.endswith(".json"):
try:
j = json.load(open(os.path.join(root, fn)))
except Exception:
continue
if task in (j.get("results") or {}):
hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j))
if not hits:
return {}
hits.sort()
path, j = hits[-1][1], hits[-1][2]
r = j["results"][task]
keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"}
cfg = j.get("config") or {}
# `config["num_fewshot"]` is null unless the flag was passed, so reading only it left the PRIMARY
# column -- the one the report quotes -- with no shot count at all. What was actually used is in
# `configs[task]`, and `n_input` is the denominator beside the score (E-045/8).
per = (j.get("configs") or {}).get(task) or {}
shots = per.get("num_fewshot")
if shots is None:
shots = cfg.get("num_fewshot")
def one(x):
return x[0] if isinstance(x, list) and x else x
# `--log_samples` writes samples_<task>_<hash>.json beside results.json, and its own row count is the
# denominator that can be checked after the fact -- 0.4.13's results rows do not reliably carry
# n_input, and a task that silently resolved to a subset otherwise looks exactly like a score.
n_logged = None
base = os.path.dirname(path)
for root, _d, files in os.walk(base): # subtask groups nest; a flat listdir missed them
for fn in sorted(files):
if not (fn.startswith("samples") and fn.endswith(".json")):
continue
try:
sj = json.load(open(os.path.join(root, fn)))
except Exception:
continue
sm = sj.get("samples")
# 0.4.x writes {"samples": {group: [rows]}} for grouped tasks (mmlu's 57 subjects) and a bare
# list for the rest. len() on the dict is the number of *groups*, which for mmlu would have
# been 1 and made every denominator check below pass while counting nothing (E-046/4).
if isinstance(sm, dict):
n_logged = sum(len(v) for v in sm.values() if isinstance(v, list))
elif isinstance(sm, list):
n_logged = len(sm)
if n_logged:
break
if n_logged:
break
return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}),
"shots": shots, "n_input": r.get("n_input"), "n_effective": r.get("n_effective"),
"n_logged": n_logged,
"limit": cfg.get("limit"), "model": one(cfg.get("model")),
"model_args": cfg.get("model_args") if isinstance(cfg.get("model_args"), (dict, str)) else None,
"dataset_path": per.get("dataset_path"), "dataset_name": per.get("dataset_name"),
"split": per.get("split") or (per.get("test_args") or {}).get("split"),
"seed": cfg.get("seed"), "dtype": cfg.get("dtype"), "repeats": cfg.get("repeats"),
"config_fewshot": cfg.get("num_fewshot"), "date": j.get("date"),
"git_hash": j.get("git_hash"),
"lm_eval_version": (j.get("lm_eval") or {}).get("version") if isinstance(j.get("lm_eval"), dict) else None}
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1")
ap.add_argument("--repo", default="", help="where to push results (defaults to --model)")
ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else")
ap.add_argument("--full-gpu-hours", type=float, default=4.0,
help="per-task ceiling for the full table; MMLU is by far the longest")
a = ap.parse_args()
if not a.model:
raise SystemExit("--model is required (the public Hub repo, loaded clean-room)")
import ounce100m_credentials
print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
ver = install_harness()
check_transformers("after harness resolved")
os.environ["CUDA_VISIBLE_DEVICES"] = "0" # one card, no gather path that could reorder exemplars
root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results")
shutil.rmtree(root, ignore_errors=True)
os.makedirs(root, exist_ok=True)
# Confirm the task ids resolve before any GPU time goes into them (C5). `lm_eval --tasks list` is NOT
# how: 0.4.13's CLI sends the word to task resolution and raises `Tasks not found: list` (E-048), so
# the registry is asked directly. Either route is fine for a human; only the API is a check.
rc, out = sh([sys.executable, "-c",
"from lm_eval.tasks import TaskManager;"
"import json,sys;"
"tm=TaskManager();"
"print(json.dumps({'n': len(tm.task_index),"
" 'present': {t: (t in tm.task_index) for t in %r}}))" % (TASKS,)],
"list_tasks", 900)
listed = {}
for line in (out or "").splitlines():
if line.startswith("{"):
try:
listed = json.loads(line)
except Exception:
listed = {}
break
if not listed or listed.get("n", 0) < 50:
raise SystemExit("could not read the harness task registry; refusing to run %s unverified -- %r"
% (PINNED_LM_EVAL, (out or "")[-500:]))
unknown = [t for t in TASKS if not listed["present"].get(t)]
if unknown:
raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown))
print("task ids checked against the harness:", len(TASKS), "requested,", listed["n"], "registered",
flush=True)
smoke = {}
for t in TASKS:
rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5)
smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}),
"cmd": cmd}
print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True)
# The smoke pass is also where the pre-registered metric name gets checked. PRIMARY_METRIC could not
# be verified from the pinned YAMLs -- reading a task config gives the metric *name* (`acc`) but the
# results key is `<name>,<aggregation>` and the aggregation half of that chain is not settled by the
# config files (E-049) -- so the alternative to checking here is discovering it after the six-hour
# columns have run. The metric must not be chosen then; the fix is the constant, not the row.
bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"]
or PRIMARY_METRIC[t] not in smoke[t]["rows"]]
for t in TASKS:
if PRIMARY_METRIC[t] not in smoke[t]["rows"] and smoke[t]["rows"]:
print("SMOKE %s metric %r absent; keys are %s" % (
t, PRIMARY_METRIC[t], sorted(smoke[t]["rows"])[:10]), flush=True)
print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True)
if bad or a.smoke_only:
json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str)
raise SystemExit(3 if bad else 0)
table, push_fail = {}, []
for t in TASKS:
rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver}
for shots, key in ((None, "primary"), (5, "secondary_5shot")):
d = os.path.join(root, key, t)
rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours)
h = harvest(d, t)
rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd,
"shots": (h.get("shots") if h.get("shots") is not None else 0),
# Six of the nine YAMLs declare no num_fewshot (docs/05 §7, re-confirmed by E-049),
# and the PRIMARY column deliberately passes no flag, so those six rows are
# 0-shot. Recording `shots: 0` with its reason is what keeps a 0-shot number from
# being read as the few-shot figure D-005 warned about -- the design of the columns
# is frozen, the description of what happened is not allowed to be vague.
"shots_source": ("harness config" if h.get("shots") is not None
else "harness default, none declared, no flag passed"),
"harness_fewshot": h.get("config_fewshot"),
"n_input": h.get("n_input"), "n_effective": h.get("n_effective"),
"n_logged": h.get("n_logged"),
"limit": h.get("limit"), "model": h.get("model"), "dtype": h.get("dtype"),
"seed": h.get("seed"), "git_hash": h.get("git_hash"),
"dataset_path": h.get("dataset_path"), "split": h.get("split"),
"results_file": h.get("file"), "primary_metric": PRIMARY_METRIC[t],
"primary": (h.get("rows") or {}).get(PRIMARY_METRIC[t])}
if PRIMARY_METRIC[t] not in (h.get("rows") or {}):
# The metric name is the one docs/05 §2 froze; if the harness writes a different key, the
# row must say so rather than publish a score of None (E-048 -- reading it out of the
# pinned YAMLs proved awkward, so it is asserted against the artifact that matters).
rec[key]["status"] = "PRIMARY METRIC %r ABSENT, have %s" % (
PRIMARY_METRIC[t], sorted(h.get("rows") or {})[:8])
if rc != 0 or not h.get("rows"):
# A cell pushed as `rows: {}` reads as "scored zero" to anyone holding only the JSON.
rec[key]["status"] = "FAILED rc=%s rows=%s" % (rc, len(h.get("rows") or {}))
elif h.get("limit") not in (None, 0):
raise SystemExit("%s/%s ran with --limit %s: a subset must not publish as a score"
% (t, key, h.get("limit")))
elif not (h.get("n_input") or h.get("n_logged")):
# No denominator beside a score is no denominator at all: say so in the record.
rec[key]["status"] = "NO ROW COUNT RECORDED"
elif key == "secondary_5shot" and h.get("shots") != 5:
# The PRIMARY column legitimately records null for six of the nine tasks -- those YAMLs
# declare no num_fewshot and we pass no flag, which is the frozen definition of the
# column (docs/05 section 2 and 7). The 5-shot column is the one where we asked, so a
# null or a different number there means the flag did not reach the harness.
rec[key]["status"] = "SHOT COUNT NOT 5: %r" % (h.get("shots"),)
print("DONE %s %s rc %s rows %s" % (t, key, rc,
json.dumps(rec[key]["rows"], default=str)[:200]), flush=True)
table[t] = rec
json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str)
if not push(a, root): # after every task: an interruption keeps what finished
push_fail.append("after " + t)
if not push(a, root):
push_fail.append("final")
def ok(t, key):
r = table.get(t, {}).get(key, {})
# `status` is set for a failed cell and for a cell with no denominator; ignoring it here is how
# the verdict printed 18/18 over a table of zeros (E-046/5).
return bool(r.get("rows")) and r.get("rc") == 0 and not r.get("status")
# Both columns and both exit codes. `primary` rows alone would report a full house with an entirely
# empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2).
missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)]
# The two columns run the same task on the same split with no limit, so a differing denominator means
# one of them resolved to a smaller subset -- a real-looking number for less data (E-045/9).
for t in TASKS:
p1 = table.get(t, {}).get("primary", {})
p2 = table.get(t, {}).get("secondary_5shot", {})
n1 = p1.get("n_input") or p1.get("n_logged")
n2 = p2.get("n_input") or p2.get("n_logged")
if n1 and n2 and n1 != n2:
missing.append("%s rows %s != %s between columns" % (t, n1, n2))
missing += push_fail
# Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2
# are separate task ids in the harness (docs/05-eval-plan.md §2).
print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % (
2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True)
raise SystemExit(4 if missing else 0)
def push(a, root):
"""Publish into the model repo, which is where a reader expects the evaluation to live (§8)."""
repo = a.repo or a.model
if not repo:
print("(no --repo and no --model-derived target; results stay local)", flush=True)
return False
import ounce100m_credentials
ounce100m_credentials.install()
tok = os.environ.get("HF_TOKEN")
if tok is None:
print("FATAL: no token, refusing to claim results were published", flush=True)
return False
try:
from huggingface_hub import HfApi
rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root,
path_in_repo="eval", commit_message="Phase 6 results",
# The --limit 5 smoke rows must not sit in the public repo beside
# the real table looking like results (C3).
ignore_patterns=["smoke*", "smoke/*", "smoke.json"],
commit_description="lm-eval 0.4.13, task-default shots plus a "
"uniform 5-shot column, per docs/05-eval-plan.md")
print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True)
# E-031's rule applied here: `upload_folder` returning is not evidence. `results.json` is the
# artifact the report cites, so re-fetch it over `resolve/` with no token and compare bytes.
try:
import hashlib as _h
import urllib.request as _u
want = _h.sha256(open(os.path.join(root, "results.json"), "rb").read()).hexdigest()
with _u.urlopen("https://huggingface.co/%s/resolve/main/eval/results.json?cb=%d"
% (repo, int(time.time())), timeout=300) as fh:
got = _h.sha256(fh.read()).hexdigest()
print("EVAL_READBACK sha_ok", got == want, flush=True)
return got == want
except Exception as e2:
print("EVAL_READBACK_FAILED", type(e2).__name__, str(e2)[:160], flush=True)
return False
except Exception as e:
print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True)
return False
if __name__ == "__main__":
main()