File size: 10,852 Bytes
5a46e5d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 | """Verify the curated BarunAction-35M public release without network access."""
from __future__ import annotations
import argparse
import hashlib
import json
from collections import Counter
from pathlib import Path
from typing import Any
EXPECTED_FILES = {
"barunaction-aggregate.json": (
5841,
"5d7244d2fa449a7ce4b2aa3a20fc095fb118a38920b1dd9181f56669a7ddfea1",
),
"barunaction-predictions.jsonl": (
282046,
"e5aea59e5090e3aa7c9dbd6b1a810c753410a159ce18dc8bb889fde3a58b8665",
),
"int8-paired-outcomes.jsonl": (
169082,
"e63e48dc8149b615ab03a14fb9fff9b1435ab59c6623e72c16b44d34d7c651ad",
),
"manifest.json": (
1908,
"d7d4223a58ac0277653c44209d9cb35edc0d98754b0c543019c0be67545bd318",
),
"qwen-aggregate.json": (
5822,
"2184b93d383f1136c050d95e71fb129bbe083b3fea4de86da6bdf5ec7f3e1756",
),
"qwen-paired-outcomes.jsonl": (
95113,
"ad5d44e741f9b98a0a0ffaadfa147c142cdaf8fea8801a11b62b7d323e93730a",
),
"qwen-predictions.jsonl": (
282394,
"836788091d08b931bf37565d197c4053e6e6f4664d379c38e7668ae9b8739b08",
),
"qwen-provenance.json": (
3434,
"1a1c19cdfacaa25e296ab1c3f92a42ebc8c08802c0cec8047ab581d50669769e",
),
}
EXPECTED_MODEL_FILES = {
"barun_config.json": "9b3a1d71baa95a198744d250f9629231738d942570b8685c44307fd83dd33565",
"checkpoint_manifest.json": (
"c743ab7c4d33ae75c6b0aa4547458a961b92766da8fcf85fd148fda2ebb5530a"
),
"model.safetensors": "fdb95ccf58a095e0d321be998924318b35ee59a334f6dd97d8726d2cf80021d3",
"tokenizer.json": "70ded9605fccd09c2340ca7e225361eab0ae8b4dbbb0d6e26343ab5183979db6",
}
EXPECTED_QWEN_OUTCOMES = {
"both_correct": 583,
"both_wrong": 74,
"candidate_only_win": 19,
"baseline_only_win": 80,
}
EXPECTED_MANIFEST_FILES = frozenset(EXPECTED_FILES) - {"manifest.json"}
ALLOWED_PREDICTION_KEYS = {
"generated_tokens",
"generation_failure",
"id",
"prediction_raw",
"prompt_tokens",
"truncated",
}
class ReleaseVerificationError(ValueError):
"""Raised when a public release commitment does not reproduce."""
def _sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for block in iter(lambda: source.read(1024 * 1024), b""):
digest.update(block)
return digest.hexdigest()
def _strict_json(path: Path) -> Any:
def reject_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
value: dict[str, Any] = {}
for key, item in pairs:
if key in value:
raise ReleaseVerificationError(f"duplicate JSON key {key!r} in {path}")
value[key] = item
return value
try:
return json.loads(path.read_text(encoding="utf-8"), object_pairs_hook=reject_pairs)
except (OSError, UnicodeError, json.JSONDecodeError) as error:
raise ReleaseVerificationError(f"invalid JSON: {path}") from error
def _jsonl(path: Path) -> list[dict[str, Any]]:
rows: list[dict[str, Any]] = []
for line_number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1):
try:
row = json.loads(line)
except json.JSONDecodeError as error:
raise ReleaseVerificationError(f"invalid JSONL at {path}:{line_number}") from error
if not isinstance(row, dict):
raise ReleaseVerificationError(f"non-object JSONL row at {path}:{line_number}")
rows.append(row)
return rows
def verify_release(root: Path, *, checkpoint: Path | None = None) -> dict[str, Any]:
evidence = root / "benchmarks" / "evidence"
checked: dict[str, dict[str, int | str]] = {}
for name, (expected_bytes, expected_sha256) in EXPECTED_FILES.items():
path = evidence / name
if not path.is_file():
raise ReleaseVerificationError(f"missing curated evidence file: {path}")
actual_bytes = path.stat().st_size
actual_sha256 = _sha256(path)
if actual_bytes != expected_bytes or actual_sha256 != expected_sha256:
raise ReleaseVerificationError(f"curated evidence hash mismatch: {path}")
checked[name] = {"bytes": actual_bytes, "sha256": actual_sha256}
predictions: dict[str, list[dict[str, Any]]] = {}
for name in ("barunaction-predictions.jsonl", "qwen-predictions.jsonl"):
rows = _jsonl(evidence / name)
if len(rows) != 756:
raise ReleaseVerificationError(f"{name} must contain exactly 756 rows")
if any(set(row) != ALLOWED_PREDICTION_KEYS for row in rows):
raise ReleaseVerificationError(f"{name} contains a non-public prediction field")
ids = [row["id"] for row in rows]
if len(ids) != len(set(ids)):
raise ReleaseVerificationError(f"{name} contains duplicate sample IDs")
predictions[name] = rows
barun_ids = [row["id"] for row in predictions["barunaction-predictions.jsonl"]]
qwen_ids = [row["id"] for row in predictions["qwen-predictions.jsonl"]]
if barun_ids != qwen_ids:
raise ReleaseVerificationError("BarunAction and Qwen prediction IDs are not aligned")
paired = _jsonl(evidence / "qwen-paired-outcomes.jsonl")
if len(paired) != 756 or [row.get("id") for row in paired] != barun_ids:
raise ReleaseVerificationError("paired comparison IDs do not align with predictions")
outcome_counts = Counter(str(row.get("outcome")) for row in paired)
if dict(outcome_counts) != EXPECTED_QWEN_OUTCOMES:
raise ReleaseVerificationError("paired comparison outcomes do not reproduce")
if any(
set(row) != {"baseline_ast_exact", "candidate_ast_exact", "id", "outcome"} for row in paired
):
raise ReleaseVerificationError("paired outcomes contain non-public fields")
int8_rows = _jsonl(evidence / "int8-paired-outcomes.jsonl")
if len(int8_rows) != 756:
raise ReleaseVerificationError("int8 paired evidence must contain exactly 756 rows")
aggregates = {
"barunaction": _strict_json(evidence / "barunaction-aggregate.json"),
"qwen": _strict_json(evidence / "qwen-aggregate.json"),
}
expected_exact = {"barunaction": (602, 756), "qwen": (663, 756)}
for name, (numerator, denominator) in expected_exact.items():
metric = aggregates[name].get("ast_exact_match")
if not isinstance(metric, dict):
raise ReleaseVerificationError(f"{name} aggregate lacks ast_exact_match")
if metric.get("numerator") != numerator or metric.get("denominator") != denominator:
raise ReleaseVerificationError(f"{name} aggregate exact count changed")
checkpoint_result: dict[str, str] | None = None
if checkpoint is not None:
checkpoint_result = {}
for name, expected_sha256 in EXPECTED_MODEL_FILES.items():
path = checkpoint / name
if not path.is_file():
raise ReleaseVerificationError(f"missing model file: {path}")
actual_sha256 = _sha256(path)
if actual_sha256 != expected_sha256:
raise ReleaseVerificationError(f"model file hash mismatch: {path}")
checkpoint_result[name] = actual_sha256
manifest = _strict_json(evidence / "manifest.json")
if manifest.get("schema_version") != "barunaction-curated-benchmark-evidence-v1":
raise ReleaseVerificationError("unexpected curated evidence manifest version")
benchmark = manifest.get("benchmark")
if not isinstance(benchmark, dict):
raise ReleaseVerificationError("curated evidence manifest lacks benchmark boundary")
expected_benchmark = {
"dataset": "google/mobile-actions",
"dataset_revision": "e920309bc2acbc2e99a5e3201cf37df2b9fd9151",
"dev_manifest_sha256": ("988bdce5874d1f1a775feeb5ba2b58cd2bdc128f57e73cb9a63d535fae7c1d55"),
"dev_rows": 756,
"evaluator_version": "action-ir-v1.0.0",
"official_evaluation_rows_accessed": 0,
"scorer_version": "barun-mobile-actions-score-v1",
}
if benchmark != expected_benchmark:
raise ReleaseVerificationError("curated evidence benchmark boundary changed")
redistribution = manifest.get("redistribution")
expected_redistribution = {
"gold_labels_included": False,
"official_evaluation_included": False,
"prompts_included": False,
"raw_or_processed_dataset_rows_included": False,
}
if redistribution != expected_redistribution:
raise ReleaseVerificationError("curated evidence redistribution boundary changed")
manifest_files = manifest.get("files")
if not isinstance(manifest_files, dict) or set(manifest_files) != EXPECTED_MANIFEST_FILES:
raise ReleaseVerificationError("curated evidence manifest inventory changed")
for name in sorted(EXPECTED_MANIFEST_FILES):
declaration = manifest_files.get(name)
if not isinstance(declaration, dict):
raise ReleaseVerificationError(f"curated evidence declaration missing: {name}")
expected_bytes, expected_sha256 = EXPECTED_FILES[name]
if (
declaration.get("bytes") != expected_bytes
or declaration.get("sha256") != expected_sha256
):
raise ReleaseVerificationError(f"curated evidence declaration changed: {name}")
if name.endswith(("predictions.jsonl", "paired-outcomes.jsonl")):
if (
declaration.get("rows") != 756
or declaration.get("contains_gold_labels") is not False
):
raise ReleaseVerificationError(f"curated evidence row boundary changed: {name}")
elif set(declaration) != {"bytes", "sha256"}:
raise ReleaseVerificationError(
f"unexpected curated evidence declaration fields: {name}"
)
return {
"barunaction_ast_exact": {"denominator": 756, "numerator": 602},
"checkpoint": checkpoint_result,
"evidence_files": checked,
"official_evaluation_rows_accessed": benchmark["official_evaluation_rows_accessed"],
"ok": True,
"qwen_ast_exact": {"denominator": 756, "numerator": 663},
"qwen_paired_outcomes": dict(sorted(outcome_counts.items())),
"schema_version": "barunaction-public-release-verification-v1",
}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--root", type=Path, default=Path(__file__).resolve().parents[1])
parser.add_argument("--checkpoint", type=Path)
args = parser.parse_args()
result = verify_release(args.root.resolve(), checkpoint=args.checkpoint)
print(json.dumps(result, indent=2, sort_keys=True))
return 0
if __name__ == "__main__":
raise SystemExit(main())
|