kwvtSA9ed3 / code /crossconfig_compare.py
DineshAI's picture
All six claims decided with reproducible evidence (5 VERIFIED, 1 FALSIFIED as stated in the main text)
e2d54c9 verified
Raw
History Blame Contribute Delete
8.82 kB
"""Decide Claims 5 and 6 from the comparison their statements actually make.
Both claims are COMPARATIVE: pluralistic (multi-reward) curation is said to sustain more
diversity across recursive retraining than single-reward curation. A single configuration
cannot test that, yet the per-configuration stages hardcoded status=VERIFIED and checked
only that the experiment ran and that metrics were recorded. Those gates passed
vacuously. This stage repairs that: it takes the measured series from every sibling
configuration and DERIVES the status from the comparison, so a reversed or absent effect
produces FALSIFIED rather than a green check.
Inputs are the per-configuration summaries emitted by the sibling runs, carried here as
committed data with their originating run id and SHA-256 (see repro/data/README.md).
Nothing is recomputed: the numbers are exactly what those runs printed.
"""
from __future__ import annotations
import hashlib
import json
from pathlib import Path
import numpy as np
from repro.lib import report
from repro.lib.verdict import BLOCKED, FALSIFIED, VERIFIED, Verdict
DATA = Path("repro/data/sibling_summaries.json")
# A margin below this is treated as "no effect": the tail means are noisy at these
# sample sizes, so a hair's-breadth ordering must not be reported as support.
MIN_MARGIN = 0.05
# The single-reward arm must end below this fraction of its OWN first round for the
# comparison to have a live collapsing baseline.
COLLAPSE_FRACTION = 0.6
def _load() -> dict:
raw = DATA.read_bytes()
d = json.loads(raw)
report.kv("sibling summary file", f"{DATA} ({len(raw)} bytes, "
f"sha256 {hashlib.sha256(raw).hexdigest()[:16]}...)")
for e in d["configs"]:
report.kv(f" {e['label']}", f"run {e['run_id']} ({e['claim']})")
return d
def _compare(entries: list[dict], metric: str, plural_key, label: str) -> tuple[bool, dict]:
"""Is the pluralistic arm's tail metric higher than the single-reward arm's?"""
plural = [e for e in entries if plural_key(e)]
single = [e for e in entries if not plural_key(e)]
if not plural or not single:
return False, {"error": f"{label}: need both arms, got "
f"{len(plural)} pluralistic and {len(single)} single"}
p_best = max(plural, key=lambda e: e[metric])
s_best = max(single, key=lambda e: e[metric])
margin = p_best[metric] - s_best[metric]
return margin > MIN_MARGIN, {
"metric": metric,
"pluralistic": {e["label"]: round(e[metric], 4) for e in plural},
"single_reward": {e["label"]: round(e[metric], 4) for e in single},
"best_pluralistic": p_best["label"], "best_single": s_best["label"],
"margin": round(margin, 4), "min_margin_required": MIN_MARGIN,
}
def run(params: dict) -> list[Verdict]:
out = report.artifact_dir("crossconfig")
if not DATA.exists():
v = Verdict(claim_id="claim5+6/cross-configuration", title="Cross-configuration comparison",
status=BLOCKED, statement="Pluralistic curation sustains more diversity "
"than single-reward curation.")
v.add("sibling summaries are available", False, f"{DATA} not present")
v.limitations = ["No sibling summaries were committed, so the comparative claims "
"cannot be evaluated."]
return [v]
data = _load()
verdicts = []
for claim, metric, statement, pk in (
("claim5", "entropy_tail_mean",
"CIFAR-10 flow retraining: curation under balanced multi-reward preferences "
"sustains higher class entropy across recursive generations than single-reward "
"curation.",
lambda e: e["M"] > 1),
("claim6", "H_tail_mean",
"Text retraining: curation under two length preferences sustains higher length "
"entropy H(L) across recursive generations than single-preference curation.",
lambda e: e["M"] > 1),
):
entries = [e for e in data["configs"] if e["claim"] == claim]
report.banner(f"{claim}: {metric} across configurations")
for e in sorted(entries, key=lambda e: e["M"]):
report.kv(f" {e['label']} (M={e['M']})",
f"first {e.get('first', float('nan')):.4f} -> tail {e[metric]:.4f}")
holds, detail = _compare(entries, metric, pk, claim)
v = Verdict(claim_id=f"{claim}/pluralistic-vs-single",
title=f"{claim.upper()}: pluralistic vs single-reward curation",
status=VERIFIED if holds else FALSIFIED,
statement=statement)
if "error" in detail:
v.status = BLOCKED
v.add("both arms were measured", False, detail["error"])
v.limitations = [detail["error"]]
verdicts.append(v)
continue
v.add(
"the comparison the claim states was computed from measured series, "
"not assumed",
True,
f"best pluralistic arm {detail['best_pluralistic']} reached {metric} "
f"{max(detail['pluralistic'].values()):.4f}; best single-reward arm "
f"{detail['best_single']} reached {max(detail['single_reward'].values()):.4f}; "
f"margin {detail['margin']:+.4f} against a required {MIN_MARGIN}",
**detail,
)
v.add(
"the ordering predicted by the paper holds with a margin above noise",
holds,
f"margin {detail['margin']:+.4f} "
+ ("exceeds" if holds else "does NOT exceed")
+ f" the {MIN_MARGIN} threshold, so the claim is "
+ ("supported" if holds else "not supported")
+ " by this reproduction",
)
# The collapse test is RELATIVE to the arm's own first round, not an absolute
# cutoff: the two domains have different entropy scales (10 CIFAR classes vs a
# length distribution over ~60 values), so any fixed threshold would be a number
# chosen to fit one of them.
s_arm = min((e for e in entries if not pk(e)), key=lambda e: e[metric])
decline_ok = s_arm[metric] < COLLAPSE_FRACTION * s_arm["first"]
v.add_control(
"the single-reward arm actually collapsed, so the comparison has a live "
"baseline",
decline_ok,
f"single-reward arm {s_arm['label']} fell from {s_arm['first']:.4f} to "
f"{s_arm[metric]:.4f}, i.e. to {s_arm[metric] / s_arm['first']:.1%} of its "
f"own first round (required: below {COLLAPSE_FRACTION:.0%}). If the "
"single-reward arm had also stayed diverse there would be no collapse to "
"avoid and the comparison would be vacuous.",
)
# Non-circularity: show that this comparison is capable of returning FALSIFIED.
# Exchange the two arms' values and re-run the identical decision procedure; if
# the swapped input still "passed", the test would be a rubber stamp.
swapped = []
for e in entries:
e2 = dict(e)
e2[metric] = (min if pk(e) else max)(x[metric] for x in entries)
swapped.append(e2)
swap_holds, swap_detail = _compare(swapped, metric, pk, claim)
v.add_control(
"the decision procedure can return FALSIFIED (it is not a rubber stamp)",
not swap_holds,
f"re-running the identical comparison with the two arms' values exchanged "
f"gives margin {swap_detail.get('margin')} and would report "
f"{'VERIFIED -- BROKEN' if swap_holds else 'FALSIFIED'}, so the outcome is "
f"driven by the measurements and not by the code path.",
)
v.numbers = detail
v.limitations = [
"Downscaled from the paper's Appendix C.5/C.6 settings; the selection "
"pressure (5% keep ratio) and the loop structure are preserved, the absolute "
"sample counts and model sizes are not.",
"One seed per configuration: the margin is a point estimate, not a "
f"confidence interval, which is why a {MIN_MARGIN} threshold is required "
"rather than any positive difference.",
"Summaries are carried as committed data from the sibling runs listed above; "
"each is traceable to its run id and the run log it was recovered from.",
]
verdicts.append(v)
report.write_json(out / "crossconfig_comparison.json",
{"verdicts": [v.claim_id for v in verdicts],
"detail": [v.numbers for v in verdicts],
"source_runs": [e["run_id"] for e in data["configs"]]})
return verdicts