File size: 25,399 Bytes
1c6dfb9 ad9e5a2 1c6dfb9 972b64f 1c6dfb9 972b64f 1c6dfb9 972b64f 1c6dfb9 972b64f 1c6dfb9 c66bf84 1c6dfb9 ad9e5a2 1c6dfb9 ad9e5a2 1c6dfb9 fbe96e3 1c6dfb9 f6d90c6 1c6dfb9 ad9e5a2 1c6dfb9 972b64f 1c6dfb9 972b64f 1c6dfb9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 | """Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them.
Written before any score exists, and it refuses to deviate from the pin:
* `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat
template -- a base model with a chat template would be an unearned capability (§3.9).
* PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**.
The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot
be tuned per task later either.
* SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead.
* One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the
moment it finishes so an interruption never loses a completed task (§5 Phase 6).
* A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus
a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time
is not acceptable. Nothing is published if the smoke pass fails.
No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and
the token never enters a log line or a published artifact.
"""
import argparse
import json
import os
import re
import shutil
import signal
import subprocess
import sys
import time
PINNED_LM_EVAL = "0.4.13"
# Every row here is the harness's own task id, not a name invented for this script.
TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1",
"truthfulqa_mc2", "winogrande", "gsm8k"]
# Chance level of the *metric being reported*, from the number of answer options in the task, not from a
# remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at.
CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"),
"arc_easy": ("0.25-0.33", "items mix 3 and 4 options"),
"hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"),
"piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"),
"truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"),
"truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"),
"gsm8k": ("0.00", "free-form exact match")}
# The metric column for each row, copied from the table docs/05-eval-plan.md section 2 froze. Naming it
# here is what stops "which of GSM8K's two numbers did we publish?" from being decided after the scores
# exist, which is the whole reason the plan was written down first (review E-045/10).
PRIMARY_METRIC = {"arc_challenge": "acc,none", "arc_easy": "acc,none", "hellaswag": "acc,none",
"mmlu": "acc,none", "piqa": "acc,none", "truthfulqa_mc1": "acc,none",
"truthfulqa_mc2": "acc,none", "winogrande": "acc,none",
"gsm8k": "exact_match,flexible-extract"}
# The stack Gate 3 measured and Gate 5 loaded against. A silently shadowed transformers in ~/.local changes
# tokenizer and generate behaviour, and every number in this table would then describe a different
# environment than the one the card claims (E-045/12).
EXPECTED_TRANSFORMERS = os.environ.get("EXPECTED_TRANSFORMERS", "5.0.0")
def sh(argv, label, timeout, env=None):
"""Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would
report nothing at all if the platform took the instance."""
print("=== " + label, flush=True)
t0 = time.time()
e = dict(os.environ)
e["PYTHONUNBUFFERED"] = "1"
e.update(env or {})
import threading
# start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the
# notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1).
p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)),
start_new_session=True)
killed = []
def _kill():
killed.append(True)
try:
os.killpg(os.getpgid(p.pid), signal.SIGTERM)
except Exception:
p.kill()
timer = threading.Timer(timeout, _kill)
timer.daemon = True
timer.start()
keep, lines = [], []
for line in p.stdout:
line = line.rstrip("\n")
lines.append(line)
keep.append(line)
del keep[:-60] # only the printed tail is bounded; the returned text is the whole run
print(" |", line[:260], flush=True)
timer.cancel()
rc = p.wait()
print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0),
flush=True)
return rc, "\n".join(lines)
def tf_version():
"""Which transformers the harness will actually import, and from where -- see the --no-deps note."""
rc, out = sh([sys.executable, "-c", "import transformers, os; "
"print(transformers.__version__, os.path.dirname(transformers.__file__))"],
"probe_transformers", 180)
for line in out.splitlines():
m = re.match(r"^\s*\|?\s*(\d[\w.]*)\s+(/\S+)", line)
if m:
return m.group(1), m.group(2)
return "?", "?"
def check_transformers(when):
"""Assert the import the harness will get -- on every path, installing or not installing."""
v, where = tf_version()
print("transformers", v, "from", where, "at", when, flush=True)
if v != EXPECTED_TRANSFORMERS:
raise SystemExit("transformers is %s (%s), not the %s Gate 3 measured and Gate 5 loaded: the eval "
"would describe an environment the model was not verified in"
% (v, where, EXPECTED_TRANSFORMERS))
if "/.local/" in where:
raise SystemExit("transformers is being imported from a user dir (%s), which shadows the image copy"
% where)
return v
def install_harness():
"""Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages:
transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would
resolve a different one."""
rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120)
have = ""
for line in out.splitlines():
if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line):
have = line.strip().strip("|").strip()
if have == PINNED_LM_EVAL:
print("lm_eval", have, "already present", flush=True)
return have # main() still calls check_transformers(): an early return used to skip it
print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL),
flush=True)
before = check_transformers("before install")
# Deps ARE installed, and that is measured, not assumed: --no-deps left the harness unimportable
# (`lm_eval/api/metrics.py:11` does `import sacrebleu` at module scope; probe v1, E-047), while a
# resolving install moved exactly two packages -- evaluate 0.4.6 and sacrebleu 2.6.0, both new and
# both into ~/.local -- and left transformers at 5.0.0 in the image (probe v2). The assertion below
# is what actually protects the pin; the flag only protected a fear.
rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet",
"lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800)
if rc != 0:
raise SystemExit("could not install the pinned harness; refusing to eval on a different one")
after = tf_version()[0]
if before != after:
raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run "
"on the stack Gate 3 measured" % (before, after))
rc, out = sh([sys.executable, "-c", "import lm_eval, os; "
"print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"],
"verify_lm_eval", 300)
got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)]
if not got:
raise SystemExit("lm_eval is not importable at the pinned version after install: "
+ out[-400:])
return PINNED_LM_EVAL
def run_task(model, task, shots, outdir, limit, gpu_h):
"""One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column."""
argv = [sys.executable, "-m", "lm_eval", "--model", "hf",
"--model_args", ",".join(["pretrained=" + model, "dtype=float16",
"trust_remote_code=False"]),
"--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir,
"--log_samples"]
if shots is not None:
argv += ["--num_fewshot", str(shots)]
if limit:
argv += ["--limit", str(limit)]
grc, gout = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60)
if grc != 0:
# The return code used to be discarded. With no card visible the harness falls back to CPU, and
# "still running" then reads like "still slow" for weeks (E-045, minor).
raise SystemExit("nvidia-smi failed (rc %s) before %s: refusing to evaluate on CPU -- %s"
% (grc, task, gout[-200:]))
rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots),
timeout=gpu_h * 3600)
return rc, " ".join(argv)
def harvest(outdir, task):
"""Find the rows the harness just wrote. It nests results under <timestamp>/<task>/results.json, so
glob rather than assume a layout that changes between releases."""
hits = []
for root, _d, files in os.walk(outdir):
for fn in files:
if fn.startswith("results") and fn.endswith(".json"):
try:
j = json.load(open(os.path.join(root, fn)))
except Exception:
continue
if task in (j.get("results") or {}):
hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j))
if not hits:
return {}
hits.sort()
path, j = hits[-1][1], hits[-1][2]
r = j["results"][task]
keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"}
cfg = j.get("config") or {}
# `config["num_fewshot"]` is null unless the flag was passed, so reading only it left the PRIMARY
# column -- the one the report quotes -- with no shot count at all. What was actually used is in
# `configs[task]`, and `n_input` is the denominator beside the score (E-045/8).
per = (j.get("configs") or {}).get(task) or {}
shots = per.get("num_fewshot")
if shots is None:
shots = cfg.get("num_fewshot")
def one(x):
return x[0] if isinstance(x, list) and x else x
# `--log_samples` writes samples_<task>_<hash>.json beside results.json, and its own row count is the
# denominator that can be checked after the fact -- 0.4.13's results rows do not reliably carry
# n_input, and a task that silently resolved to a subset otherwise looks exactly like a score.
n_logged = None
base = os.path.dirname(path)
for root, _d, files in os.walk(base): # subtask groups nest; a flat listdir missed them
for fn in sorted(files):
if not (fn.startswith("samples") and fn.endswith(".json")):
continue
try:
sj = json.load(open(os.path.join(root, fn)))
except Exception:
continue
sm = sj.get("samples")
# 0.4.x writes {"samples": {group: [rows]}} for grouped tasks (mmlu's 57 subjects) and a bare
# list for the rest. len() on the dict is the number of *groups*, which for mmlu would have
# been 1 and made every denominator check below pass while counting nothing (E-046/4).
if isinstance(sm, dict):
n_logged = sum(len(v) for v in sm.values() if isinstance(v, list))
elif isinstance(sm, list):
n_logged = len(sm)
if n_logged:
break
if n_logged:
break
return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}),
"shots": shots, "n_input": r.get("n_input"), "n_effective": r.get("n_effective"),
"n_logged": n_logged,
"limit": cfg.get("limit"), "model": one(cfg.get("model")),
"model_args": cfg.get("model_args") if isinstance(cfg.get("model_args"), (dict, str)) else None,
"dataset_path": per.get("dataset_path"), "dataset_name": per.get("dataset_name"),
"split": per.get("split") or (per.get("test_args") or {}).get("split"),
"seed": cfg.get("seed"), "dtype": cfg.get("dtype"), "repeats": cfg.get("repeats"),
"config_fewshot": cfg.get("num_fewshot"), "date": j.get("date"),
"git_hash": j.get("git_hash"),
"lm_eval_version": (j.get("lm_eval") or {}).get("version") if isinstance(j.get("lm_eval"), dict) else None}
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1")
ap.add_argument("--repo", default="", help="where to push results (defaults to --model)")
ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else")
ap.add_argument("--full-gpu-hours", type=float, default=4.0,
help="per-task ceiling for the full table; MMLU is by far the longest")
a = ap.parse_args()
if not a.model:
raise SystemExit("--model is required (the public Hub repo, loaded clean-room)")
import ounce100m_credentials
print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
ver = install_harness()
check_transformers("after harness resolved")
os.environ["CUDA_VISIBLE_DEVICES"] = "0" # one card, no gather path that could reorder exemplars
root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results")
shutil.rmtree(root, ignore_errors=True)
os.makedirs(root, exist_ok=True)
# Confirm the task ids resolve before any GPU time goes into them (C5). `lm_eval --tasks list` is NOT
# how: 0.4.13's CLI sends the word to task resolution and raises `Tasks not found: list` (E-048), so
# the registry is asked directly. Either route is fine for a human; only the API is a check.
rc, out = sh([sys.executable, "-c",
"from lm_eval.tasks import TaskManager;"
"import json,sys;"
"tm=TaskManager();"
"print(json.dumps({'n': len(tm.task_index),"
" 'present': {t: (t in tm.task_index) for t in %r}}))" % (TASKS,)],
"list_tasks", 900)
listed = {}
for line in (out or "").splitlines():
if line.startswith("{"):
try:
listed = json.loads(line)
except Exception:
listed = {}
break
if not listed or listed.get("n", 0) < 50:
raise SystemExit("could not read the harness task registry; refusing to run %s unverified -- %r"
% (PINNED_LM_EVAL, (out or "")[-500:]))
unknown = [t for t in TASKS if not listed["present"].get(t)]
if unknown:
raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown))
print("task ids checked against the harness:", len(TASKS), "requested,", listed["n"], "registered",
flush=True)
smoke = {}
for t in TASKS:
rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5)
smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}),
"cmd": cmd}
print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True)
# The smoke pass is also where the pre-registered metric name gets checked. PRIMARY_METRIC could not
# be verified from the pinned YAMLs -- reading a task config gives the metric *name* (`acc`) but the
# results key is `<name>,<aggregation>` and the aggregation half of that chain is not settled by the
# config files (E-049) -- so the alternative to checking here is discovering it after the six-hour
# columns have run. The metric must not be chosen then; the fix is the constant, not the row.
bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"]
or PRIMARY_METRIC[t] not in smoke[t]["rows"]]
for t in TASKS:
if PRIMARY_METRIC[t] not in smoke[t]["rows"] and smoke[t]["rows"]:
print("SMOKE %s metric %r absent; keys are %s" % (
t, PRIMARY_METRIC[t], sorted(smoke[t]["rows"])[:10]), flush=True)
print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True)
if bad or a.smoke_only:
json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str)
raise SystemExit(3 if bad else 0)
table, push_fail = {}, []
for t in TASKS:
rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver}
for shots, key in ((None, "primary"), (5, "secondary_5shot")):
d = os.path.join(root, key, t)
rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours)
h = harvest(d, t)
rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd,
"shots": (h.get("shots") if h.get("shots") is not None else 0),
# Six of the nine YAMLs declare no num_fewshot (docs/05 §7, re-confirmed by E-049),
# and the PRIMARY column deliberately passes no flag, so those six rows are
# 0-shot. Recording `shots: 0` with its reason is what keeps a 0-shot number from
# being read as the few-shot figure D-005 warned about -- the design of the columns
# is frozen, the description of what happened is not allowed to be vague.
"shots_source": ("harness config" if h.get("shots") is not None
else "harness default, none declared, no flag passed"),
"harness_fewshot": h.get("config_fewshot"),
"n_input": h.get("n_input"), "n_effective": h.get("n_effective"),
"n_logged": h.get("n_logged"),
"limit": h.get("limit"), "model": h.get("model"), "dtype": h.get("dtype"),
"seed": h.get("seed"), "git_hash": h.get("git_hash"),
"dataset_path": h.get("dataset_path"), "split": h.get("split"),
"results_file": h.get("file"), "primary_metric": PRIMARY_METRIC[t],
"primary": (h.get("rows") or {}).get(PRIMARY_METRIC[t])}
if PRIMARY_METRIC[t] not in (h.get("rows") or {}):
# The metric name is the one docs/05 §2 froze; if the harness writes a different key, the
# row must say so rather than publish a score of None (E-048 -- reading it out of the
# pinned YAMLs proved awkward, so it is asserted against the artifact that matters).
rec[key]["status"] = "PRIMARY METRIC %r ABSENT, have %s" % (
PRIMARY_METRIC[t], sorted(h.get("rows") or {})[:8])
if rc != 0 or not h.get("rows"):
# A cell pushed as `rows: {}` reads as "scored zero" to anyone holding only the JSON.
rec[key]["status"] = "FAILED rc=%s rows=%s" % (rc, len(h.get("rows") or {}))
elif h.get("limit") not in (None, 0):
raise SystemExit("%s/%s ran with --limit %s: a subset must not publish as a score"
% (t, key, h.get("limit")))
elif not (h.get("n_input") or h.get("n_logged")):
# No denominator beside a score is no denominator at all: say so in the record.
rec[key]["status"] = "NO ROW COUNT RECORDED"
elif key == "secondary_5shot" and h.get("shots") != 5:
# The PRIMARY column legitimately records null for six of the nine tasks -- those YAMLs
# declare no num_fewshot and we pass no flag, which is the frozen definition of the
# column (docs/05 section 2 and 7). The 5-shot column is the one where we asked, so a
# null or a different number there means the flag did not reach the harness.
rec[key]["status"] = "SHOT COUNT NOT 5: %r" % (h.get("shots"),)
print("DONE %s %s rc %s rows %s" % (t, key, rc,
json.dumps(rec[key]["rows"], default=str)[:200]), flush=True)
table[t] = rec
json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str)
if not push(a, root): # after every task: an interruption keeps what finished
push_fail.append("after " + t)
if not push(a, root):
push_fail.append("final")
def ok(t, key):
r = table.get(t, {}).get(key, {})
# `status` is set for a failed cell and for a cell with no denominator; ignoring it here is how
# the verdict printed 18/18 over a table of zeros (E-046/5).
return bool(r.get("rows")) and r.get("rc") == 0 and not r.get("status")
# Both columns and both exit codes. `primary` rows alone would report a full house with an entirely
# empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2).
missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)]
# The two columns run the same task on the same split with no limit, so a differing denominator means
# one of them resolved to a smaller subset -- a real-looking number for less data (E-045/9).
for t in TASKS:
p1 = table.get(t, {}).get("primary", {})
p2 = table.get(t, {}).get("secondary_5shot", {})
n1 = p1.get("n_input") or p1.get("n_logged")
n2 = p2.get("n_input") or p2.get("n_logged")
if n1 and n2 and n1 != n2:
missing.append("%s rows %s != %s between columns" % (t, n1, n2))
missing += push_fail
# Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2
# are separate task ids in the harness (docs/05-eval-plan.md §2).
print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % (
2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True)
raise SystemExit(4 if missing else 0)
def push(a, root):
"""Publish into the model repo, which is where a reader expects the evaluation to live (§8)."""
repo = a.repo or a.model
if not repo:
print("(no --repo and no --model-derived target; results stay local)", flush=True)
return False
import ounce100m_credentials
ounce100m_credentials.install()
tok = os.environ.get("HF_TOKEN")
if tok is None:
print("FATAL: no token, refusing to claim results were published", flush=True)
return False
try:
from huggingface_hub import HfApi
rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root,
path_in_repo="eval", commit_message="Phase 6 results",
# The --limit 5 smoke rows must not sit in the public repo beside
# the real table looking like results (C3).
ignore_patterns=["smoke*", "smoke/*", "smoke.json"],
commit_description="lm-eval 0.4.13, task-default shots plus a "
"uniform 5-shot column, per docs/05-eval-plan.md")
print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True)
# E-031's rule applied here: `upload_folder` returning is not evidence. `results.json` is the
# artifact the report cites, so re-fetch it over `resolve/` with no token and compare bytes.
try:
import hashlib as _h
import urllib.request as _u
want = _h.sha256(open(os.path.join(root, "results.json"), "rb").read()).hexdigest()
with _u.urlopen("https://huggingface.co/%s/resolve/main/eval/results.json?cb=%d"
% (repo, int(time.time())), timeout=300) as fh:
got = _h.sha256(fh.read()).hexdigest()
print("EVAL_READBACK sha_ok", got == want, flush=True)
return got == want
except Exception as e2:
print("EVAL_READBACK_FAILED", type(e2).__name__, str(e2)[:160], flush=True)
return False
except Exception as e:
print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True)
return False
if __name__ == "__main__":
main()
|