AdithyaSK's picture
AdithyaSK HF Staff
Harbor run viewer: Phase 0 eval, 15 harnesses x 50 tasks, pass@4
3d20eb8 verified
Raw
History Blame Contribute Delete
7.08 kB
"""Derive the stable / experimental tiers from a sweep, instead of asserting them.
The tier is a claim about evidence, so it should be computed from the evidence and carry it. A badge
that someone typed by hand goes stale the moment a harness improves or regresses, and a tier with no
stated reason is just a colour.
RULES, in the order they are applied:
stable the harness produced graded rollouts for essentially every task AND has a verified
training run. Both halves matter: capture working proves the tokens are right, and a
completed training step proves the trainer can consume them — they are separate failure
modes, and this stack has hit each independently.
experimental anything else, with the specific gap named. Never a bare tier.
WHAT IS NOT A REASON TO DOWNGRADE. A low pass rate. A harness scoring 0.0 on hard tasks is working
correctly and reporting a real result; treating that as a defect would rank harnesses by how easy their
tasks were. Only unmeasured rollouts, pauses, and known skew count against a harness here.
Prompt re-render skew IS recorded as a caveat rather than a downgrade on its own: it is harmless for
eval (nothing is trained) and disqualifying for training, so the caveat says which.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
HERE = Path(__file__).resolve().parents[1]
# Measured over 670 turns against the engine's own prompt_token_ids. Eval-safe, training-unsafe.
KNOWN_SKEW = {
"claude-code": "+2 tokens per prompt re-render — harmless for eval, forks every turn when training",
"gemini-cli": "+2 tokens per prompt re-render — harmless for eval, forks every turn when training",
"kimi-cli": "-10 tokens per tool call — the largest measured skew; unsafe to train on",
}
# Reads os.environ inside run(), so concurrency depends on the context-local overlay.
NEEDS_OVERLAY = {"goose", "claude-code", "gemini-cli"}
NO_STEP_LIMIT_NOTE = ("no step-limit expression in its seam, so rollouts run to the timeout; "
"the kill surfaces as exit 137 and the rollout is retried")
def classify(sweep: dict, trained: set[str], measured_floor: float) -> dict:
summary = sweep.get("summary", {})
per = summary.get("harnesses", {})
paused = summary.get("paused_harnesses", {})
k = summary.get("k", 4)
out = {}
for harness, m in sorted(per.items()):
caveats, tier = [], "experimental"
n_tasks = m.get("n_tasks") or 0
measured = m.get("n_measured") or 0
coverage = (measured / n_tasks) if n_tasks else 0.0
if harness in paused:
caveats.append(f"PAUSED mid-sweep: {paused[harness]}")
elif coverage < measured_floor:
caveats.append(
f"only {measured}/{n_tasks} tasks produced a graded rollout "
f"({coverage:.0%} < {measured_floor:.0%} required)"
)
elif harness not in trained:
caveats.append(
"eval measured but no verified training run — capture working does not prove the "
"trainer can consume it, which is a separate failure mode"
)
else:
tier = "stable"
if harness in KNOWN_SKEW:
caveats.append(KNOWN_SKEW[harness])
if harness in NEEDS_OVERLAY:
caveats.append("reads os.environ inside run(); concurrent only via the context-local overlay")
entry = {
"tier": tier,
"evidence": (
f"pass@{k} {m.get(f'pass@{k}')}, pass@1 {m.get('pass@1')}, "
f"{measured}/{n_tasks} tasks measured, mean {m.get('mean_turns')} turns"
),
}
if caveats:
entry["caveats"] = caveats
out[harness] = entry
# A harness that never appeared in the sweep at all is not 'experimental', it is untested — saying
# otherwise would imply it was tried.
for harness in paused:
out.setdefault(harness, {"tier": "experimental", "caveats": [f"PAUSED: {paused[harness]}"]})
return out
def main() -> int:
ap = argparse.ArgumentParser()
# Several files, because a sweep can be split across jobs — and it was: one 15-harness job projected
# past its time limit, so it became three. Merging here rather than requiring one file means the
# split is an operational detail instead of something the classification has to know about.
ap.add_argument("--sweep", required=True, nargs="+", help="one or more eval sweep JSONs")
ap.add_argument("--project", default="data-agent")
ap.add_argument("--trained", default="mini-swe-agent,opencode",
help="harnesses with a verified training run")
ap.add_argument("--measured-floor", type=float, default=0.9,
help="fraction of tasks that must produce a graded rollout to be stable")
ap.add_argument("--dry-run", action="store_true")
args = ap.parse_args()
merged = {"summary": {"harnesses": {}, "paused_harnesses": {}, "k": None}}
for f in args.sweep:
one = json.loads(Path(f).read_text())
sm = one.get("summary", {})
merged["summary"]["k"] = merged["summary"]["k"] or sm.get("k")
merged["summary"]["paused_harnesses"].update(sm.get("paused_harnesses") or {})
for h, m in (sm.get("harnesses") or {}).items():
prev = merged["summary"]["harnesses"].get(h)
# A harness can appear in more than one file — opencode ran in the cancelled job AND in the
# relaunch. Keep whichever measured more tasks: that is the more complete evidence, and
# averaging two partial runs of different sizes would invent a number neither produced.
if prev is None or (m.get("n_measured") or 0) > (prev.get("n_measured") or 0):
merged["summary"]["harnesses"][h] = m
sweep = merged
trained = {h.strip() for h in args.trained.split(",") if h.strip()}
support = classify(sweep, trained, args.measured_floor)
for h, e in sorted(support.items(), key=lambda kv: (kv[1]["tier"] != "stable", kv[0])):
print(f" {h:18s} {e['tier']:13s} {e.get('evidence','')}")
for c in e.get("caveats", []):
print(f" · {c}")
if args.dry_run:
return 0
p = HERE / "data" / "projects" / args.project / "project.json"
d = json.loads(p.read_text()) if p.exists() else {"project_id": args.project}
d["support"] = support
d["tier_rule"] = (
"stable = graded rollouts on >=90% of tasks AND a verified training run. experimental = "
"anything else, with the gap named. A low pass rate is never a downgrade: a harness scoring 0.0 "
"is reporting a real result, and penalising that would rank harnesses by task difficulty."
)
d["support_source"] = [Path(f).name for f in args.sweep]
p.write_text(json.dumps(d, indent=2))
print(f"\nwrote {p.relative_to(HERE)}")
return 0
if __name__ == "__main__":
raise SystemExit(main())