#!/usr/bin/env python3 """Stratify certification by UNIT SIZE. CPU only; no model, no network. The model's binding limit is the size of the unit you hand it, not the Python version. This script is the instrument behind the size guidance on the model card, and it runs on the PUBLISHED benchmark from the PUBLISHED generations, so every bucket on the card is recomputable from files you downloaded. Size axis is REPRESENTATION LINES: the number of lines in the disassembly text handed to the model. That is literally the model's input length, it is already stored in each benchmark row's `input` field, and you can measure your own input the same way before you run anything: rep_lines = disassemble_v2(code_object).count("\\n") Buckets match the ones used in the pooled cross-benchmark analysis so the two are comparable. ./size_curve.py --bench ../benchmarks/csn-3.12-licensed/bench.jsonl \\ --greedy ../generations/gen_v3_csn600.jsonl \\ --samples ../generations/boN_v3_csn600.jsonl \\ --base ../generations/gen_base_csn600.jsonl \\ --out ../results/size_curve_csn600.json """ from __future__ import annotations import argparse import json import sys from collections import defaultdict from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from analyze_scores import cluster_bootstrap # noqa: E402 from common import load_bench, load_jsonl, ours_ok, require_sound, self_test, strip_fences # noqa: E402 BUCKETS = [(0, 50), (50, 100), (100, 200), (200, 300), (300, 400), (400, 600), (600, 10**9)] KNEE = 200 # where the greedy curve turns, per the pooled analysis def label(lo: int, hi: int) -> str: return f"{lo}-{hi - 1}" if hi < 10**9 else f"{lo}+" def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--bench", required=True) ap.add_argument("--greedy", required=True) ap.add_argument("--samples") ap.add_argument("--base") ap.add_argument("--out", required=True) a = ap.parse_args() bench, _ = load_bench(a.bench) st = self_test(bench, "ours") print(f"self-test: preflight {st['preflight_pct']}% mutation kill " f"{st['mutation_kill_rate_pct']}%", file=sys.stderr, flush=True) require_sound(st) greedy = {r["i"]: r for r in load_jsonl(a.greedy)} base = {r["i"]: r for r in load_jsonl(a.base)} if a.base else {} samples: dict[int, dict[int, str]] = defaultdict(dict) have_samples = bool(a.samples and Path(a.samples).exists()) if have_samples: for r in load_jsonl(a.samples): samples[r["i"]][r["s"]] = r["got"] rows = [] for i in sorted(bench): exp = bench[i]["expected"] g_ok = i in greedy and ours_ok(strip_fences(greedy[i]["got"]), exp) first = None if not g_ok: for s in sorted(samples.get(i, {})): if ours_ok(strip_fences(samples[i][s]), exp): first = s break rows.append({ "i": i, "repo": bench[i]["provenance"]["repo"], # the representation the model is actually given, one line per disassembly line "rep_lines": bench[i]["input"].count("\n"), "n_instr": bench[i]["n_instr"], "v3_greedy": bool(g_ok), "v3_boN32": bool(g_ok or (first is not None and first <= 30)), "base_greedy": bool(i in base and ours_ok(strip_fences(base[i]["got"]), exp)), }) systems = ["v3_greedy", "base_greedy"] + (["v3_boN32"] if have_samples else []) out = { "bench": str(a.bench), "n": len(rows), "size_axis": "rep_lines = lines of the disassembly handed to the model (bench row `input`)", "best_of_n_included": have_samples, "self_test": {k: st[k] for k in ("preflight_pct", "mutation_kill_rate_pct", "SOUND")}, "rep_lines_distribution": {}, "by_rep_lines": [], "share_of_certifications_below_knee": {}, } vals = sorted(r["rep_lines"] for r in rows) def pct(p): return vals[min(len(vals) - 1, int(p * len(vals)))] out["rep_lines_distribution"] = { "min": vals[0], "p25": pct(.25), "median": pct(.5), "p75": pct(.75), "p90": pct(.9), "p99": pct(.99), "max": vals[-1], } for lo, hi in BUCKETS: sel = [r for r in rows if lo <= r["rep_lines"] < hi] e = {"bucket": label(lo, hi), "n": len(sel)} for s in systems: c = sum(r[s] for r in sel) e[s] = {"certified": c, "n": len(sel), "pct": round(100 * c / len(sel), 2) if sel else None} # A clustered CI needs enough repos to resample; below that it is noise dressed as # precision, so it is omitted rather than printed. if len(sel) >= 30 and len({r["repo"] for r in sel}) >= 10: d = defaultdict(list) for r in sel: d[r["repo"]].append(1 if r[s] else 0) ci = cluster_bootstrap(d) e[s]["ci95"] = [ci["ci95_lo"], ci["ci95_hi"]] e[s]["ci_method"] = "repo-clustered bootstrap" else: e[s]["ci95"] = None e[s]["ci_method"] = "omitted: too few rows/repos to estimate" out["by_rep_lines"].append(e) for s in systems: tot = sum(r[s] for r in rows) small = sum(r[s] for r in rows if r["rep_lines"] < KNEE) out["share_of_certifications_below_knee"][s] = { "knee_rep_lines": KNEE, "certified_total": tot, "certified_below": small, "pct": round(100 * small / tot, 2) if tot else None, } Path(a.out).parent.mkdir(parents=True, exist_ok=True) Path(a.out).write_text(json.dumps(out, indent=2)) print(json.dumps(out, indent=2)) if __name__ == "__main__": main()