Download eval/run_benchmarks.py from Cion-lab/ounce100m-code: direct link, hf CLI and curl.
- Browser
- Download file 25.4 kB
-
https://huggingface.co/Cion-lab/ounce100m-code/resolve/main/eval/run_benchmarks.py
- Command line
-
hf download hf://Cion-lab/ounce100m-code/eval/run_benchmarks.py
-
curl -L -o run_benchmarks.py https://huggingface.co/Cion-lab/ounce100m-code/resolve/main/eval/run_benchmarks.py
25.4 kB
| """Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them. | |
| Written before any score exists, and it refuses to deviate from the pin: | |
| * `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat | |
| template -- a base model with a chat template would be an unearned capability (§3.9). | |
| * PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**. | |
| The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot | |
| be tuned per task later either. | |
| * SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead. | |
| * One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the | |
| moment it finishes so an interruption never loses a completed task (§5 Phase 6). | |
| * A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus | |
| a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time | |
| is not acceptable. Nothing is published if the smoke pass fails. | |
| No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and | |
| the token never enters a log line or a published artifact. | |
| """ | |
| import argparse | |
| import json | |
| import os | |
| import re | |
| import shutil | |
| import signal | |
| import subprocess | |
| import sys | |
| import time | |
| PINNED_LM_EVAL = "0.4.13" | |
| # Every row here is the harness's own task id, not a name invented for this script. | |
| TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1", | |
| "truthfulqa_mc2", "winogrande", "gsm8k"] | |
| # Chance level of the *metric being reported*, from the number of answer options in the task, not from a | |
| # remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at. | |
| CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"), | |
| "arc_easy": ("0.25-0.33", "items mix 3 and 4 options"), | |
| "hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"), | |
| "piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"), | |
| "truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"), | |
| "truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"), | |
| "gsm8k": ("0.00", "free-form exact match")} | |
| # The metric column for each row, copied from the table docs/05-eval-plan.md section 2 froze. Naming it | |
| # here is what stops "which of GSM8K's two numbers did we publish?" from being decided after the scores | |
| # exist, which is the whole reason the plan was written down first (review E-045/10). | |
| PRIMARY_METRIC = {"arc_challenge": "acc,none", "arc_easy": "acc,none", "hellaswag": "acc,none", | |
| "mmlu": "acc,none", "piqa": "acc,none", "truthfulqa_mc1": "acc,none", | |
| "truthfulqa_mc2": "acc,none", "winogrande": "acc,none", | |
| "gsm8k": "exact_match,flexible-extract"} | |
| # The stack Gate 3 measured and Gate 5 loaded against. A silently shadowed transformers in ~/.local changes | |
| # tokenizer and generate behaviour, and every number in this table would then describe a different | |
| # environment than the one the card claims (E-045/12). | |
| EXPECTED_TRANSFORMERS = os.environ.get("EXPECTED_TRANSFORMERS", "5.0.0") | |
| def sh(argv, label, timeout, env=None): | |
| """Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would | |
| report nothing at all if the platform took the instance.""" | |
| print("=== " + label, flush=True) | |
| t0 = time.time() | |
| e = dict(os.environ) | |
| e["PYTHONUNBUFFERED"] = "1" | |
| e.update(env or {}) | |
| import threading | |
| # start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the | |
| # notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1). | |
| p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, | |
| env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)), | |
| start_new_session=True) | |
| killed = [] | |
| def _kill(): | |
| killed.append(True) | |
| try: | |
| os.killpg(os.getpgid(p.pid), signal.SIGTERM) | |
| except Exception: | |
| p.kill() | |
| timer = threading.Timer(timeout, _kill) | |
| timer.daemon = True | |
| timer.start() | |
| keep, lines = [], [] | |
| for line in p.stdout: | |
| line = line.rstrip("\n") | |
| lines.append(line) | |
| keep.append(line) | |
| del keep[:-60] # only the printed tail is bounded; the returned text is the whole run | |
| print(" |", line[:260], flush=True) | |
| timer.cancel() | |
| rc = p.wait() | |
| print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0), | |
| flush=True) | |
| return rc, "\n".join(lines) | |
| def tf_version(): | |
| """Which transformers the harness will actually import, and from where -- see the --no-deps note.""" | |
| rc, out = sh([sys.executable, "-c", "import transformers, os; " | |
| "print(transformers.__version__, os.path.dirname(transformers.__file__))"], | |
| "probe_transformers", 180) | |
| for line in out.splitlines(): | |
| m = re.match(r"^\s*\|?\s*(\d[\w.]*)\s+(/\S+)", line) | |
| if m: | |
| return m.group(1), m.group(2) | |
| return "?", "?" | |
| def check_transformers(when): | |
| """Assert the import the harness will get -- on every path, installing or not installing.""" | |
| v, where = tf_version() | |
| print("transformers", v, "from", where, "at", when, flush=True) | |
| if v != EXPECTED_TRANSFORMERS: | |
| raise SystemExit("transformers is %s (%s), not the %s Gate 3 measured and Gate 5 loaded: the eval " | |
| "would describe an environment the model was not verified in" | |
| % (v, where, EXPECTED_TRANSFORMERS)) | |
| if "/.local/" in where: | |
| raise SystemExit("transformers is being imported from a user dir (%s), which shadows the image copy" | |
| % where) | |
| return v | |
| def install_harness(): | |
| """Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages: | |
| transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would | |
| resolve a different one.""" | |
| rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120) | |
| have = "" | |
| for line in out.splitlines(): | |
| if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line): | |
| have = line.strip().strip("|").strip() | |
| if have == PINNED_LM_EVAL: | |
| print("lm_eval", have, "already present", flush=True) | |
| return have # main() still calls check_transformers(): an early return used to skip it | |
| print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL), | |
| flush=True) | |
| before = check_transformers("before install") | |
| # Deps ARE installed, and that is measured, not assumed: --no-deps left the harness unimportable | |
| # (`lm_eval/api/metrics.py:11` does `import sacrebleu` at module scope; probe v1, E-047), while a | |
| # resolving install moved exactly two packages -- evaluate 0.4.6 and sacrebleu 2.6.0, both new and | |
| # both into ~/.local -- and left transformers at 5.0.0 in the image (probe v2). The assertion below | |
| # is what actually protects the pin; the flag only protected a fear. | |
| rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet", | |
| "lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800) | |
| if rc != 0: | |
| raise SystemExit("could not install the pinned harness; refusing to eval on a different one") | |
| after = tf_version()[0] | |
| if before != after: | |
| raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run " | |
| "on the stack Gate 3 measured" % (before, after)) | |
| rc, out = sh([sys.executable, "-c", "import lm_eval, os; " | |
| "print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"], | |
| "verify_lm_eval", 300) | |
| got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)] | |
| if not got: | |
| raise SystemExit("lm_eval is not importable at the pinned version after install: " | |
| + out[-400:]) | |
| return PINNED_LM_EVAL | |
| def run_task(model, task, shots, outdir, limit, gpu_h): | |
| """One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column.""" | |
| argv = [sys.executable, "-m", "lm_eval", "--model", "hf", | |
| "--model_args", ",".join(["pretrained=" + model, "dtype=float16", | |
| "trust_remote_code=False"]), | |
| "--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir, | |
| "--log_samples"] | |
| if shots is not None: | |
| argv += ["--num_fewshot", str(shots)] | |
| if limit: | |
| argv += ["--limit", str(limit)] | |
| grc, gout = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60) | |
| if grc != 0: | |
| # The return code used to be discarded. With no card visible the harness falls back to CPU, and | |
| # "still running" then reads like "still slow" for weeks (E-045, minor). | |
| raise SystemExit("nvidia-smi failed (rc %s) before %s: refusing to evaluate on CPU -- %s" | |
| % (grc, task, gout[-200:])) | |
| rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots), | |
| timeout=gpu_h * 3600) | |
| return rc, " ".join(argv) | |
| def harvest(outdir, task): | |
| """Find the rows the harness just wrote. It nests results under <timestamp>/<task>/results.json, so | |
| glob rather than assume a layout that changes between releases.""" | |
| hits = [] | |
| for root, _d, files in os.walk(outdir): | |
| for fn in files: | |
| if fn.startswith("results") and fn.endswith(".json"): | |
| try: | |
| j = json.load(open(os.path.join(root, fn))) | |
| except Exception: | |
| continue | |
| if task in (j.get("results") or {}): | |
| hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j)) | |
| if not hits: | |
| return {} | |
| hits.sort() | |
| path, j = hits[-1][1], hits[-1][2] | |
| r = j["results"][task] | |
| keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"} | |
| cfg = j.get("config") or {} | |
| # `config["num_fewshot"]` is null unless the flag was passed, so reading only it left the PRIMARY | |
| # column -- the one the report quotes -- with no shot count at all. What was actually used is in | |
| # `configs[task]`, and `n_input` is the denominator beside the score (E-045/8). | |
| per = (j.get("configs") or {}).get(task) or {} | |
| shots = per.get("num_fewshot") | |
| if shots is None: | |
| shots = cfg.get("num_fewshot") | |
| def one(x): | |
| return x[0] if isinstance(x, list) and x else x | |
| # `--log_samples` writes samples_<task>_<hash>.json beside results.json, and its own row count is the | |
| # denominator that can be checked after the fact -- 0.4.13's results rows do not reliably carry | |
| # n_input, and a task that silently resolved to a subset otherwise looks exactly like a score. | |
| n_logged = None | |
| base = os.path.dirname(path) | |
| for root, _d, files in os.walk(base): # subtask groups nest; a flat listdir missed them | |
| for fn in sorted(files): | |
| if not (fn.startswith("samples") and fn.endswith(".json")): | |
| continue | |
| try: | |
| sj = json.load(open(os.path.join(root, fn))) | |
| except Exception: | |
| continue | |
| sm = sj.get("samples") | |
| # 0.4.x writes {"samples": {group: [rows]}} for grouped tasks (mmlu's 57 subjects) and a bare | |
| # list for the rest. len() on the dict is the number of *groups*, which for mmlu would have | |
| # been 1 and made every denominator check below pass while counting nothing (E-046/4). | |
| if isinstance(sm, dict): | |
| n_logged = sum(len(v) for v in sm.values() if isinstance(v, list)) | |
| elif isinstance(sm, list): | |
| n_logged = len(sm) | |
| if n_logged: | |
| break | |
| if n_logged: | |
| break | |
| return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}), | |
| "shots": shots, "n_input": r.get("n_input"), "n_effective": r.get("n_effective"), | |
| "n_logged": n_logged, | |
| "limit": cfg.get("limit"), "model": one(cfg.get("model")), | |
| "model_args": cfg.get("model_args") if isinstance(cfg.get("model_args"), (dict, str)) else None, | |
| "dataset_path": per.get("dataset_path"), "dataset_name": per.get("dataset_name"), | |
| "split": per.get("split") or (per.get("test_args") or {}).get("split"), | |
| "seed": cfg.get("seed"), "dtype": cfg.get("dtype"), "repeats": cfg.get("repeats"), | |
| "config_fewshot": cfg.get("num_fewshot"), "date": j.get("date"), | |
| "git_hash": j.get("git_hash"), | |
| "lm_eval_version": (j.get("lm_eval") or {}).get("version") if isinstance(j.get("lm_eval"), dict) else None} | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1") | |
| ap.add_argument("--repo", default="", help="where to push results (defaults to --model)") | |
| ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else") | |
| ap.add_argument("--full-gpu-hours", type=float, default=4.0, | |
| help="per-task ceiling for the full table; MMLU is by far the longest") | |
| a = ap.parse_args() | |
| if not a.model: | |
| raise SystemExit("--model is required (the public Hub repo, loaded clean-room)") | |
| import ounce100m_credentials | |
| print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True) | |
| ver = install_harness() | |
| check_transformers("after harness resolved") | |
| os.environ["CUDA_VISIBLE_DEVICES"] = "0" # one card, no gather path that could reorder exemplars | |
| root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results") | |
| shutil.rmtree(root, ignore_errors=True) | |
| os.makedirs(root, exist_ok=True) | |
| # Confirm the task ids resolve before any GPU time goes into them (C5). `lm_eval --tasks list` is NOT | |
| # how: 0.4.13's CLI sends the word to task resolution and raises `Tasks not found: list` (E-048), so | |
| # the registry is asked directly. Either route is fine for a human; only the API is a check. | |
| rc, out = sh([sys.executable, "-c", | |
| "from lm_eval.tasks import TaskManager;" | |
| "import json,sys;" | |
| "tm=TaskManager();" | |
| "print(json.dumps({'n': len(tm.task_index)," | |
| " 'present': {t: (t in tm.task_index) for t in %r}}))" % (TASKS,)], | |
| "list_tasks", 900) | |
| listed = {} | |
| for line in (out or "").splitlines(): | |
| if line.startswith("{"): | |
| try: | |
| listed = json.loads(line) | |
| except Exception: | |
| listed = {} | |
| break | |
| if not listed or listed.get("n", 0) < 50: | |
| raise SystemExit("could not read the harness task registry; refusing to run %s unverified -- %r" | |
| % (PINNED_LM_EVAL, (out or "")[-500:])) | |
| unknown = [t for t in TASKS if not listed["present"].get(t)] | |
| if unknown: | |
| raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown)) | |
| print("task ids checked against the harness:", len(TASKS), "requested,", listed["n"], "registered", | |
| flush=True) | |
| smoke = {} | |
| for t in TASKS: | |
| rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5) | |
| smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}), | |
| "cmd": cmd} | |
| print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True) | |
| # The smoke pass is also where the pre-registered metric name gets checked. PRIMARY_METRIC could not | |
| # be verified from the pinned YAMLs -- reading a task config gives the metric *name* (`acc`) but the | |
| # results key is `<name>,<aggregation>` and the aggregation half of that chain is not settled by the | |
| # config files (E-049) -- so the alternative to checking here is discovering it after the six-hour | |
| # columns have run. The metric must not be chosen then; the fix is the constant, not the row. | |
| bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"] | |
| or PRIMARY_METRIC[t] not in smoke[t]["rows"]] | |
| for t in TASKS: | |
| if PRIMARY_METRIC[t] not in smoke[t]["rows"] and smoke[t]["rows"]: | |
| print("SMOKE %s metric %r absent; keys are %s" % ( | |
| t, PRIMARY_METRIC[t], sorted(smoke[t]["rows"])[:10]), flush=True) | |
| print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True) | |
| if bad or a.smoke_only: | |
| json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str) | |
| raise SystemExit(3 if bad else 0) | |
| table, push_fail = {}, [] | |
| for t in TASKS: | |
| rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver} | |
| for shots, key in ((None, "primary"), (5, "secondary_5shot")): | |
| d = os.path.join(root, key, t) | |
| rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours) | |
| h = harvest(d, t) | |
| rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd, | |
| "shots": (h.get("shots") if h.get("shots") is not None else 0), | |
| # Six of the nine YAMLs declare no num_fewshot (docs/05 §7, re-confirmed by E-049), | |
| # and the PRIMARY column deliberately passes no flag, so those six rows are | |
| # 0-shot. Recording `shots: 0` with its reason is what keeps a 0-shot number from | |
| # being read as the few-shot figure D-005 warned about -- the design of the columns | |
| # is frozen, the description of what happened is not allowed to be vague. | |
| "shots_source": ("harness config" if h.get("shots") is not None | |
| else "harness default, none declared, no flag passed"), | |
| "harness_fewshot": h.get("config_fewshot"), | |
| "n_input": h.get("n_input"), "n_effective": h.get("n_effective"), | |
| "n_logged": h.get("n_logged"), | |
| "limit": h.get("limit"), "model": h.get("model"), "dtype": h.get("dtype"), | |
| "seed": h.get("seed"), "git_hash": h.get("git_hash"), | |
| "dataset_path": h.get("dataset_path"), "split": h.get("split"), | |
| "results_file": h.get("file"), "primary_metric": PRIMARY_METRIC[t], | |
| "primary": (h.get("rows") or {}).get(PRIMARY_METRIC[t])} | |
| if PRIMARY_METRIC[t] not in (h.get("rows") or {}): | |
| # The metric name is the one docs/05 §2 froze; if the harness writes a different key, the | |
| # row must say so rather than publish a score of None (E-048 -- reading it out of the | |
| # pinned YAMLs proved awkward, so it is asserted against the artifact that matters). | |
| rec[key]["status"] = "PRIMARY METRIC %r ABSENT, have %s" % ( | |
| PRIMARY_METRIC[t], sorted(h.get("rows") or {})[:8]) | |
| if rc != 0 or not h.get("rows"): | |
| # A cell pushed as `rows: {}` reads as "scored zero" to anyone holding only the JSON. | |
| rec[key]["status"] = "FAILED rc=%s rows=%s" % (rc, len(h.get("rows") or {})) | |
| elif h.get("limit") not in (None, 0): | |
| raise SystemExit("%s/%s ran with --limit %s: a subset must not publish as a score" | |
| % (t, key, h.get("limit"))) | |
| elif not (h.get("n_input") or h.get("n_logged")): | |
| # No denominator beside a score is no denominator at all: say so in the record. | |
| rec[key]["status"] = "NO ROW COUNT RECORDED" | |
| elif key == "secondary_5shot" and h.get("shots") != 5: | |
| # The PRIMARY column legitimately records null for six of the nine tasks -- those YAMLs | |
| # declare no num_fewshot and we pass no flag, which is the frozen definition of the | |
| # column (docs/05 section 2 and 7). The 5-shot column is the one where we asked, so a | |
| # null or a different number there means the flag did not reach the harness. | |
| rec[key]["status"] = "SHOT COUNT NOT 5: %r" % (h.get("shots"),) | |
| print("DONE %s %s rc %s rows %s" % (t, key, rc, | |
| json.dumps(rec[key]["rows"], default=str)[:200]), flush=True) | |
| table[t] = rec | |
| json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str) | |
| if not push(a, root): # after every task: an interruption keeps what finished | |
| push_fail.append("after " + t) | |
| if not push(a, root): | |
| push_fail.append("final") | |
| def ok(t, key): | |
| r = table.get(t, {}).get(key, {}) | |
| # `status` is set for a failed cell and for a cell with no denominator; ignoring it here is how | |
| # the verdict printed 18/18 over a table of zeros (E-046/5). | |
| return bool(r.get("rows")) and r.get("rc") == 0 and not r.get("status") | |
| # Both columns and both exit codes. `primary` rows alone would report a full house with an entirely | |
| # empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2). | |
| missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)] | |
| # The two columns run the same task on the same split with no limit, so a differing denominator means | |
| # one of them resolved to a smaller subset -- a real-looking number for less data (E-045/9). | |
| for t in TASKS: | |
| p1 = table.get(t, {}).get("primary", {}) | |
| p2 = table.get(t, {}).get("secondary_5shot", {}) | |
| n1 = p1.get("n_input") or p1.get("n_logged") | |
| n2 = p2.get("n_input") or p2.get("n_logged") | |
| if n1 and n2 and n1 != n2: | |
| missing.append("%s rows %s != %s between columns" % (t, n1, n2)) | |
| missing += push_fail | |
| # Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2 | |
| # are separate task ids in the harness (docs/05-eval-plan.md §2). | |
| print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % ( | |
| 2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True) | |
| raise SystemExit(4 if missing else 0) | |
| def push(a, root): | |
| """Publish into the model repo, which is where a reader expects the evaluation to live (§8).""" | |
| repo = a.repo or a.model | |
| if not repo: | |
| print("(no --repo and no --model-derived target; results stay local)", flush=True) | |
| return False | |
| import ounce100m_credentials | |
| ounce100m_credentials.install() | |
| tok = os.environ.get("HF_TOKEN") | |
| if tok is None: | |
| print("FATAL: no token, refusing to claim results were published", flush=True) | |
| return False | |
| try: | |
| from huggingface_hub import HfApi | |
| rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root, | |
| path_in_repo="eval", commit_message="Phase 6 results", | |
| # The --limit 5 smoke rows must not sit in the public repo beside | |
| # the real table looking like results (C3). | |
| ignore_patterns=["smoke*", "smoke/*", "smoke.json"], | |
| commit_description="lm-eval 0.4.13, task-default shots plus a " | |
| "uniform 5-shot column, per docs/05-eval-plan.md") | |
| print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True) | |
| # E-031's rule applied here: `upload_folder` returning is not evidence. `results.json` is the | |
| # artifact the report cites, so re-fetch it over `resolve/` with no token and compare bytes. | |
| try: | |
| import hashlib as _h | |
| import urllib.request as _u | |
| want = _h.sha256(open(os.path.join(root, "results.json"), "rb").read()).hexdigest() | |
| with _u.urlopen("https://huggingface.co/%s/resolve/main/eval/results.json?cb=%d" | |
| % (repo, int(time.time())), timeout=300) as fh: | |
| got = _h.sha256(fh.read()).hexdigest() | |
| print("EVAL_READBACK sha_ok", got == want, flush=True) | |
| return got == want | |
| except Exception as e2: | |
| print("EVAL_READBACK_FAILED", type(e2).__name__, str(e2)[:160], flush=True) | |
| return False | |
| except Exception as e: | |
| print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True) | |
| return False | |
| if __name__ == "__main__": | |
| main() | |