opus-max-record / harness /scripts /final_2x2.py
simonycl's picture
Upload folder using huggingface_hub
b2ebc95 verified
Raw
History Blame Contribute Delete
5.52 kB
#!/usr/bin/env python3
"""The final 2x2: {base, submitted} x {stock, pi-ws} on one shared task set.
Prints the cell table (solve rate + Wilson 95% CI) and every comparison that matters, each as a
*paired* McNemar test on the tasks the two runs actually share. Paired is the right test here:
`verifiers` pins SEED=0, so `shuffle=true` draws the identical sample every run and the arms line
up task for task. An unpaired test on the same data throws that away and needs a far larger effect
to reach the same confidence.
Usage:
final_2x2.py base_stock=runs/base250_stock base_ws=runs/base250_ws \
sub_stock=runs/big_swe_stock sub_ws=runs/big_swe_ws
Reports `ungraded` per cell — episodes with no reward record at all, i.e. lost to infrastructure
rather than failed. They are excluded from the rate, and a cell where they differ a lot between
arms is not a fair comparison however good the p-value looks.
"""
import json
import math
import os
import sys
def wilson(k, n, z=1.96):
if n == 0:
return 0.0, 0.0, 0.0
p = k / n
d = 1 + z * z / n
c = (p + z * z / (2 * n)) / d
h = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / d
return p, max(0.0, c - h), min(1.0, c + h)
def mcnemar_exact(b, c):
"""Two-sided exact McNemar on discordant pairs: b wins for A, c wins for B."""
n = b + c
if n == 0:
return 1.0
tail = sum(math.comb(n, i) for i in range(0, min(b, c) + 1)) * 0.5**n
return min(1.0, 2 * tail)
def load(path):
"""-> {task_key: solved_bool}, ungraded_count. Task key prefers the dataset's own name."""
f = os.path.join(path, "traces.jsonl")
per, ungraded = {}, 0
if not os.path.exists(f):
return per, ungraded
for line in open(f):
rec = json.loads(line)
for t in rec.get("traces", [rec]):
rw = {k: v for k, v in (t.get("rewards") or {}).items()
if isinstance(v, dict) and v.get("score") is not None}
if not rw:
ungraded += 1
continue
data = (t.get("task") or {}).get("data") or {}
key = data.get("name") or data.get("instance_id") or data.get("idx")
if key is None:
continue
total = sum(v["score"] * v.get("weight", 1.0) for v in rw.values())
weight = sum(v.get("weight", 1.0) for v in rw.values()) or 1.0
per[key] = (total / weight) > 0
return per, ungraded
def main():
cells = {}
for arg in sys.argv[1:]:
name, _, path = arg.partition("=")
per, ungraded = load(path)
cells[name] = (per, ungraded, path)
if not cells:
print(__doc__)
return
print(f"{'cell':<12} {'n':>4} {'solved':>7} {'score':>7} {'ci95':>16} {'ungraded':>8}")
for name, (per, ungraded, path) in cells.items():
n, k = len(per), sum(per.values())
p, lo, hi = wilson(k, n)
print(f"{name:<12} {n:>4} {k:>7} {p:>7.3f} [{lo:.3f}, {hi:.3f}] {ungraded:>8}")
print()
print("paired comparisons (only tasks present in both runs):")
print(f"{'A -> B':<28} {'shared':>7} {'only A':>7} {'only B':>7} {'delta':>8} {'McNemar':>9}")
names = list(cells)
for i, a in enumerate(names):
for b in names[i + 1:]:
pa, pb = cells[a][0], cells[b][0]
shared = set(pa) & set(pb)
if not shared:
continue
only_a = sum(1 for t in shared if pa[t] and not pb[t])
only_b = sum(1 for t in shared if pb[t] and not pa[t])
ka = sum(1 for t in shared if pa[t])
kb = sum(1 for t in shared if pb[t])
delta = (kb - ka) / len(shared)
p = mcnemar_exact(only_a, only_b)
print(f"{a+' -> '+b:<28} {len(shared):>7} {only_a:>7} {only_b:>7} {delta:>+8.3f} {p:>9.4f}")
# Matched subset. Cells finished at different n (the pool starved the base arms), and a
# partially-completed run over-represents easy tasks — episodes that finish first are the
# short ones. Restricting every cell to the tasks graded in ALL of them removes that
# composition difference: the subset may be easier than the full suite, but it is the same
# subset for all four, so cell-to-cell comparisons are clean.
if len(cells) > 2:
common = set.intersection(*[set(c[0]) for c in cells.values()])
if common:
print()
print(f"matched subset — the {len(common)} tasks graded in ALL cells:")
for name, (per, _u, _p) in cells.items():
k = sum(per[t] for t in common)
p_, lo, hi = wilson(k, len(common))
print(f" {name:<12} {k:>3}/{len(common)} = {p_:.3f} [{lo:.3f}, {hi:.3f}]")
print()
print(f" {'A -> B':<26} {'only A':>7} {'only B':>7} {'delta':>8} {'McNemar':>9}")
for i, a in enumerate(names):
for b in names[i + 1:]:
pa, pb = cells[a][0], cells[b][0]
only_a = sum(1 for t in common if pa[t] and not pb[t])
only_b = sum(1 for t in common if pb[t] and not pa[t])
ka = sum(1 for t in common if pa[t])
kb = sum(1 for t in common if pb[t])
delta = (kb - ka) / len(common)
p = mcnemar_exact(only_a, only_b)
print(f" {a+' -> '+b:<26} {only_a:>7} {only_b:>7} {delta:>+8.3f} {p:>9.4f}")
if __name__ == "__main__":
main()