pybytecode-v3-1.5b / harness /size_curve.py
coolblaze03's picture
Add files using upload-large-folder tool
0b19a1b verified
Raw
History Blame Contribute Delete
5.85 kB
#!/usr/bin/env python3
"""Stratify certification by UNIT SIZE. CPU only; no model, no network.
The model's binding limit is the size of the unit you hand it, not the Python version. This
script is the instrument behind the size guidance on the model card, and it runs on the PUBLISHED
benchmark from the PUBLISHED generations, so every bucket on the card is recomputable from files
you downloaded.
Size axis is REPRESENTATION LINES: the number of lines in the disassembly text handed to the
model. That is literally the model's input length, it is already stored in each benchmark row's
`input` field, and you can measure your own input the same way before you run anything:
rep_lines = disassemble_v2(code_object).count("\\n")
Buckets match the ones used in the pooled cross-benchmark analysis so the two are comparable.
./size_curve.py --bench ../benchmarks/csn-3.12-licensed/bench.jsonl \\
--greedy ../generations/gen_v3_csn600.jsonl \\
--samples ../generations/boN_v3_csn600.jsonl \\
--base ../generations/gen_base_csn600.jsonl \\
--out ../results/size_curve_csn600.json
"""
from __future__ import annotations
import argparse
import json
import sys
from collections import defaultdict
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from analyze_scores import cluster_bootstrap # noqa: E402
from common import load_bench, load_jsonl, ours_ok, require_sound, self_test, strip_fences # noqa: E402
BUCKETS = [(0, 50), (50, 100), (100, 200), (200, 300), (300, 400), (400, 600), (600, 10**9)]
KNEE = 200 # where the greedy curve turns, per the pooled analysis
def label(lo: int, hi: int) -> str:
return f"{lo}-{hi - 1}" if hi < 10**9 else f"{lo}+"
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--bench", required=True)
ap.add_argument("--greedy", required=True)
ap.add_argument("--samples")
ap.add_argument("--base")
ap.add_argument("--out", required=True)
a = ap.parse_args()
bench, _ = load_bench(a.bench)
st = self_test(bench, "ours")
print(f"self-test: preflight {st['preflight_pct']}% mutation kill "
f"{st['mutation_kill_rate_pct']}%", file=sys.stderr, flush=True)
require_sound(st)
greedy = {r["i"]: r for r in load_jsonl(a.greedy)}
base = {r["i"]: r for r in load_jsonl(a.base)} if a.base else {}
samples: dict[int, dict[int, str]] = defaultdict(dict)
have_samples = bool(a.samples and Path(a.samples).exists())
if have_samples:
for r in load_jsonl(a.samples):
samples[r["i"]][r["s"]] = r["got"]
rows = []
for i in sorted(bench):
exp = bench[i]["expected"]
g_ok = i in greedy and ours_ok(strip_fences(greedy[i]["got"]), exp)
first = None
if not g_ok:
for s in sorted(samples.get(i, {})):
if ours_ok(strip_fences(samples[i][s]), exp):
first = s
break
rows.append({
"i": i,
"repo": bench[i]["provenance"]["repo"],
# the representation the model is actually given, one line per disassembly line
"rep_lines": bench[i]["input"].count("\n"),
"n_instr": bench[i]["n_instr"],
"v3_greedy": bool(g_ok),
"v3_boN32": bool(g_ok or (first is not None and first <= 30)),
"base_greedy": bool(i in base and ours_ok(strip_fences(base[i]["got"]), exp)),
})
systems = ["v3_greedy", "base_greedy"] + (["v3_boN32"] if have_samples else [])
out = {
"bench": str(a.bench),
"n": len(rows),
"size_axis": "rep_lines = lines of the disassembly handed to the model (bench row `input`)",
"best_of_n_included": have_samples,
"self_test": {k: st[k] for k in ("preflight_pct", "mutation_kill_rate_pct", "SOUND")},
"rep_lines_distribution": {},
"by_rep_lines": [],
"share_of_certifications_below_knee": {},
}
vals = sorted(r["rep_lines"] for r in rows)
def pct(p): return vals[min(len(vals) - 1, int(p * len(vals)))]
out["rep_lines_distribution"] = {
"min": vals[0], "p25": pct(.25), "median": pct(.5), "p75": pct(.75),
"p90": pct(.9), "p99": pct(.99), "max": vals[-1],
}
for lo, hi in BUCKETS:
sel = [r for r in rows if lo <= r["rep_lines"] < hi]
e = {"bucket": label(lo, hi), "n": len(sel)}
for s in systems:
c = sum(r[s] for r in sel)
e[s] = {"certified": c, "n": len(sel),
"pct": round(100 * c / len(sel), 2) if sel else None}
# A clustered CI needs enough repos to resample; below that it is noise dressed as
# precision, so it is omitted rather than printed.
if len(sel) >= 30 and len({r["repo"] for r in sel}) >= 10:
d = defaultdict(list)
for r in sel:
d[r["repo"]].append(1 if r[s] else 0)
ci = cluster_bootstrap(d)
e[s]["ci95"] = [ci["ci95_lo"], ci["ci95_hi"]]
e[s]["ci_method"] = "repo-clustered bootstrap"
else:
e[s]["ci95"] = None
e[s]["ci_method"] = "omitted: too few rows/repos to estimate"
out["by_rep_lines"].append(e)
for s in systems:
tot = sum(r[s] for r in rows)
small = sum(r[s] for r in rows if r["rep_lines"] < KNEE)
out["share_of_certifications_below_knee"][s] = {
"knee_rep_lines": KNEE, "certified_total": tot, "certified_below": small,
"pct": round(100 * small / tot, 2) if tot else None,
}
Path(a.out).parent.mkdir(parents=True, exist_ok=True)
Path(a.out).write_text(json.dumps(out, indent=2))
print(json.dumps(out, indent=2))
if __name__ == "__main__":
main()