wrice's picture
Publish validated 2e620fa7fcbb68537932ddad5cab51b8cd9c4874
a7b634e verified
Raw
History Blame Contribute Delete
38.4 kB
"""Evidence audit logic for PostTrainBench reproduction.
Every function requires verified acquired data as input. No function
produces authoritative evidence from constants alone. All inputs come
from :func:`acquisition.acquire_all` results.
"""
from __future__ import annotations
import hashlib
import re
from collections import defaultdict
from typing import Any
from posttrainbench_repro.constants import (
API_MISUSE_TASK_CLUSTER,
ARXIV_ID,
ATTEMPT_ID,
CANONICAL_ALL_ENTRIES_SHA256,
CANONICAL_DIRS_SHA256,
CANONICAL_FILES_SHA256,
CHALLENGE_ASSESSMENT_DIGEST,
CHALLENGE_JSON_SHA256,
CHALLENGE_REVISION,
CLAIM_1_SHA256,
CLAIM_1_TEXT,
CLAIM_2_SHA256,
CLAIM_2_TEXT,
CONTAMINATION_WITNESS_BYTES,
CONTAMINATION_WITNESS_PATH,
CONTAMINATION_WITNESS_SHA256,
EXCLUDED_TOP_LEVEL,
EXPECTED_BENCHMARKS,
EXPECTED_CELL_COUNTS,
EXPECTED_DUPLICATE_PAIRS,
EXPECTED_EVAL_DIRS,
EXPECTED_MISSING_PAIRS,
EXPECTED_MODEL_FRAGMENTS,
EXPECTED_ROOT_CELL_PAIRS,
EXPECTED_ROOT_COUNT,
EXPECTED_TASK_COUNT,
GIT_TREE_DIGEST,
GIT_TREE_ENTRY_COUNT,
GIT_TREE_ID,
GITHUB_PINNED_COMMIT,
GITHUB_REPO_URL,
HF_DATASET_LICENSE,
HF_DATASET_URL,
HF_PINNED_REVISION,
HF_TREE_DIR_COUNT,
HF_TREE_FILE_COUNT,
HF_TREE_PAGE_SIZE,
HF_TREE_TOTAL_ENTRIES,
HF_TREE_TOTAL_PAGES,
INDEX_JSON_SHA256,
INSTRUCTION_MODEL_JUDGMENT_BYTES,
INSTRUCTION_MODEL_JUDGMENT_GIT_OBJECT,
INSTRUCTION_MODEL_JUDGMENT_PATH,
INSTRUCTION_MODEL_JUDGMENT_SHA256,
INSTRUCTION_MODEL_JUDGMENT_SIZE,
INSTRUCTION_MODEL_TRACE_GIT_OBJECT,
INSTRUCTION_MODEL_TRACE_PATH,
INSTRUCTION_MODEL_TRACE_SHA256,
INSTRUCTION_MODEL_TRACE_SIZE,
MODEL_ORDER,
PAID_API_COST_USD,
PAPER_ID,
PAPER_LICENSE,
PINNED_BLOBS,
RUN_ROOT_10H_RE,
SNAPSHOT_ID,
SOURCE_LICENSE,
TASK_BASENAME_RE,
TIME_TAKEN_WITNESS_BYTES,
TIME_TAKEN_WITNESS_PATH,
TIME_TAKEN_WITNESS_SHA256,
TRACE_EXCERPTS,
TRUNCATED_SIBLINGS_COUNT,
TRUNCATED_SIBLINGS_SHA256,
UPSTREAM_TOKEN,
VIEWER_DATA_FILE_COUNT,
)
# ---------------------------------------------------------------------------
# Provenance
# ---------------------------------------------------------------------------
def get_provenance(acquired: dict[str, Any]) -> dict[str, Any]:
"""Return the provenance record derived from acquired data.
Requires a verified ``acquired`` dict from :func:`acquisition.acquire_all`.
"""
github = acquired["github"]
hf_inv = acquired["hf_inventory"]
return {
"paper_id": PAPER_ID,
"attempt_id": ATTEMPT_ID,
"assessed_snapshot": SNAPSHOT_ID,
"challenge_revision": CHALLENGE_REVISION,
"challenge_assessment_digest": CHALLENGE_ASSESSMENT_DIGEST,
"challenge_json_sha256": CHALLENGE_JSON_SHA256,
"index_json_sha256": INDEX_JSON_SHA256,
"upstream_token": UPSTREAM_TOKEN,
"arxiv_id": ARXIV_ID,
"paper_license": PAPER_LICENSE,
"source": {
"repository": GITHUB_REPO_URL,
"pinned_commit": github["commit"],
"tree_id": github["tree_id"],
"entry_count": github["entry_count"],
"canonical_tree_digest": github["canonical_tree_digest"],
"license": SOURCE_LICENSE,
"tree_acquisition": github["tree_acquisition"],
"consumed_blobs": {
path: meta
for path, meta in sorted(github["blobs"].items())
},
},
"dataset": {
"repository": HF_DATASET_URL,
"pinned_revision": hf_inv["revision"],
"license": HF_DATASET_LICENSE,
"pagination": {
"endpoint": "tree",
"params": "recursive=true&expand=false&limit=1000",
"mechanism": "Link header cursor, rel=\"next\"",
"page_size": HF_TREE_PAGE_SIZE,
"total_pages": hf_inv["page_count"],
"total_entries": hf_inv["total_entries"],
"file_count": hf_inv["file_count"],
"directory_count": hf_inv["dir_count"],
},
"canonical_digests": {
"all_entries": hf_inv["canonical_all_digest"],
"files": hf_inv["canonical_file_digest"],
"directories": hf_inv["canonical_dir_digest"],
},
"tree_acquisition": hf_inv["tree_acquisition"],
"consumed_files": acquired["hf_consumed_files"],
"truncated_siblings": {
"count": TRUNCATED_SIBLINGS_COUNT,
"digest": TRUNCATED_SIBLINGS_SHA256,
"note": "Truncated lexical prefix from Hub revision-metadata; "
"rejected as coverage input.",
},
},
"paid_api_cost_usd": PAID_API_COST_USD,
}
# ---------------------------------------------------------------------------
# Coverage census
# ---------------------------------------------------------------------------
def compute_coverage(
hf_inventory: dict[str, Any],
) -> dict[str, Any]:
"""Compute the 4-by-7 coverage matrix from the complete HF inventory.
Requires the verified ``hf_inventory`` dict from acquisition.
Duplicate-job counting is per (root, benchmark, model) pair.
"""
dir_paths = hf_inventory["dir_paths"]
coverage = _compute_coverage_from_dirs(dir_paths)
coverage["accepted_benchmark_count"] = len(EXPECTED_BENCHMARKS)
coverage["accepted_model_count"] = len(EXPECTED_MODEL_FRAGMENTS)
if "file_paths" in hf_inventory:
viewer_dirs = sorted(
path
for path in dir_paths
if path == "viewer_data" or path.startswith("viewer_data/")
)
viewer_files = sorted(
path
for path in hf_inventory["file_paths"]
if path.startswith("viewer_data/")
)
if "viewer_data" not in viewer_dirs:
raise ValueError(
"Verified inventory is missing excluded top-level viewer_data"
)
if len(viewer_files) != VIEWER_DATA_FILE_COUNT:
raise ValueError(
"Verified viewer_data auxiliary file count mismatch: "
f"{len(viewer_files)} != 2,397"
)
coverage["excluded_auxiliary_data"] = {
"top_level_path": "viewer_data",
"present": True,
"file_count": len(viewer_files),
"directory_count": len(viewer_dirs),
"counted_as_task_root": False,
}
inventory_fields = {
"page_count",
"total_entries",
"file_count",
"dir_count",
"canonical_all_digest",
"canonical_file_digest",
"canonical_dir_digest",
}
if inventory_fields.issubset(hf_inventory):
coverage["inventory"] = {
"page_count": hf_inventory["page_count"],
"total_entries": hf_inventory["total_entries"],
"file_count": hf_inventory["file_count"],
"dir_count": hf_inventory["dir_count"],
"all_entries_digest": hf_inventory["canonical_all_digest"],
"file_entries_digest": hf_inventory["canonical_file_digest"],
"dir_entries_digest": hf_inventory["canonical_dir_digest"],
"rejected_siblings_count": TRUNCATED_SIBLINGS_COUNT,
"rejected_siblings_digest": TRUNCATED_SIBLINGS_SHA256,
}
return coverage
def _compute_coverage_from_dirs(
dir_paths: list[str],
) -> dict[str, Any]:
"""Core coverage computation from directory paths.
Duplicate counting: a duplicate is an extra task for the same
(opaque run root, benchmark, model) triple.
"""
model_order = MODEL_ORDER
benchmark_list = sorted(EXPECTED_BENCHMARKS)
task_dirs: list[str] = []
run_roots: set[str] = set()
excluded_dirs: list[str] = []
unrecognized_dirs: list[str] = []
# Task directories: exactly two path components (depth-2)
for p in dir_paths:
parts = p.split("/")
if len(parts) != 2:
continue
root = parts[0]
basename = parts[1]
if root in EXCLUDED_TOP_LEVEL:
excluded_dirs.append(p)
continue
# Check if root matches 10h pattern
if not RUN_ROOT_10H_RE.search(root):
unrecognized_dirs.append(p)
continue
m = TASK_BASENAME_RE.match(basename)
if not m:
unrecognized_dirs.append(p)
continue
task_dirs.append(p)
run_roots.add(root)
# Build the coverage matrix
# Per-cell task counting (global cell = bench × model)
cell_tasks: dict[tuple[str, str], list[str]] = defaultdict(list)
# Per-root/cell tracking for duplicates and missing pairs
root_cell_set: set[tuple[str, str, str]] = set() # (root, bench, model)
root_cell_tasks: dict[tuple[str, str, str], list[str]] = defaultdict(list)
for p in task_dirs:
root, basename = p.split("/")
m = TASK_BASENAME_RE.match(basename)
if not m:
continue
bench = m.group(1)
model_fragment = m.group(2)
model_normalized = EXPECTED_MODEL_FRAGMENTS[model_fragment]
cell_tasks[(bench, model_normalized)].append(p)
root_cell_set.add((root, bench, model_normalized))
root_cell_tasks[(root, bench, model_normalized)].append(p)
# Compute cell counts
cell_counts: dict[str, list[int]] = {}
matrix_list: list[dict[str, Any]] = []
for bench in benchmark_list:
counts = []
for model in model_order:
count = len(cell_tasks.get((bench, model), []))
counts.append(count)
matrix_list.append({
"benchmark": bench,
"model": model,
"count": count,
})
cell_counts[bench] = counts
# Duplicate-job pairs: per (root, bench, model), extra tasks beyond the first
duplicate_pairs = sum(
len(tasks) - 1
for tasks in root_cell_tasks.values()
if len(tasks) > 1
)
# Count unique root/cell pairs
root_cell_pair_count = len(root_cell_set)
# Missing root/cell pairs: (roots × benchmarks × models) − actual
total_possible = len(run_roots) * len(benchmark_list) * len(model_order)
missing_pairs = total_possible - root_cell_pair_count
return {
"accepted_benchmarks": benchmark_list,
"accepted_models": EXPECTED_MODEL_FRAGMENTS,
"recognized_task_count": len(task_dirs),
"recognized_root_count": len(run_roots),
"recognized_root_cell_pairs": root_cell_pair_count,
"duplicate_job_pairs": duplicate_pairs,
"missing_root_cell_pairs": missing_pairs,
"excluded_dirs_count": len(excluded_dirs),
"unrecognized_dirs_count": len(unrecognized_dirs),
"matrix": matrix_list,
"cell_counts": cell_counts,
}
# ---------------------------------------------------------------------------
# Protocol audit
# ---------------------------------------------------------------------------
def _source_lines(
content: str,
predicate: Any,
) -> list[int]:
"""Return one-based active source lines matching a predicate."""
return [
line_number
for line_number, line in enumerate(content.splitlines(), 1)
if line.strip()
and not line.lstrip().startswith("#")
and predicate(line)
]
def _blob_reference(
path: str,
content: str,
predicate: Any,
label: str,
) -> dict[str, Any]:
"""Build a deterministic path/line reference into one verified blob."""
lines = _source_lines(content, predicate)
if not lines:
raise ValueError(f"Could not source {label} in {path}")
git_object_sha1, raw_sha256 = PINNED_BLOBS[path]
return {
"commit": GITHUB_PINNED_COMMIT,
"path": path,
"lines": lines,
"git_object_sha1": git_object_sha1,
"raw_sha256": raw_sha256,
}
def audit_protocol(
blob_contents: dict[str, bytes],
git_entries: list[dict[str, Any]],
) -> dict[str, Any]:
"""Audit runner protocol controls from pinned source blobs.
Requires verified blob contents and git tree entries.
Derives all facts from actual content; fails on changed facts.
"""
single_task = blob_contents["src/commit_utils/single_task.sub"].decode("utf-8")
run_task = blob_contents["src/run_task.sh"].decode("utf-8")
commit_sh = blob_contents["src/commit_utils/commit.sh"].decode("utf-8")
result: dict[str, Any] = {}
# num_gpus default
m = re.search(r"num_gpus\s*=\s*(\d+)", single_task)
if not m:
raise ValueError("Could not find num_gpus in single_task.sub")
result["num_gpus_default"] = int(m.group(1))
if result["num_gpus_default"] != 1:
raise ValueError(
f"Expected single_task.sub num_gpus default 1, "
f"got {result['num_gpus_default']}"
)
# CUDA device requirement
m = re.search(
r'TARGET\.CUDADeviceName\s*==\s*"([^"]+)"',
single_task,
)
if not m:
raise ValueError("Could not find CUDADeviceName in single_task.sub")
result["cuda_device_requirement"] = (
f'TARGET.CUDADeviceName == "{m.group(1)}"'
)
expected_cuda = 'TARGET.CUDADeviceName == "NVIDIA H100 80GB HBM3"'
if result["cuda_device_requirement"] != expected_cuda:
raise ValueError(
"Unexpected CUDADeviceName requirement: "
f"{result['cuda_device_requirement']}"
)
# request_gpus binding
m = re.search(r"request_gpus\s*=\s*\$\(num_gpus\)", single_task)
if not m:
raise ValueError("Could not find request_gpus binding")
result["request_gpus_binding"] = "request_gpus = $(num_gpus)"
# NUM_HOURS in run_task.sh
result["receives_num_hours"] = "NUM_HOURS" in run_task
if not result["receives_num_hours"]:
raise ValueError("run_task.sh does not receive NUM_HOURS")
# Solve timeout formula (minutes based on NUM_HOURS * 60 + 5)
timeout_patterns = [
r"NUM_HOURS\s*\*\s*60\s*\+\s*5",
r"\$\(\(\s*NUM_HOURS\s*\*\s*60\s*\+\s*5\s*\)\)",
r"NUM_HOURS.*60.*\+.*5",
]
if not any(re.search(p, run_task) for p in timeout_patterns):
raise ValueError("Could not find timeout formula in run_task.sh")
result["solve_timeout_formula"] = "NUM_HOURS * 60 + 5"
result["timeout_grace_minutes"] = 5
result["timeout_formula_found"] = True
# 10h suffix in task roots
result["task_dir_10h_suffix"] = True
# Eval directories present in git tree
tree_paths = {e["path"] for e in git_entries}
eval_dirs_found: list[str] = []
for d in EXPECTED_EVAL_DIRS:
if d in tree_paths:
eval_dirs_found.append(d)
if len(eval_dirs_found) != 7:
raise ValueError(
f"Expected 7 eval directories, found {len(eval_dirs_found)}: "
f"{eval_dirs_found}"
)
result["evaluation_dirs_present"] = eval_dirs_found
result["evaluation_dir_count"] = len(eval_dirs_found)
# Commit.sh analysis for limitations
result["commit_sh_analysis"] = _analyze_commit_sh(commit_sh)
single_task_path = "src/commit_utils/single_task.sub"
run_task_path = "src/run_task.sh"
commit_path = "src/commit_utils/commit.sh"
mpi_job = lambda line: (
line.lstrip().startswith("condor_submit_bid")
and '"num_hours=100"' in line
and '"num_gpus=8"' in line
)
default_jobs = lambda line: (
line.lstrip().startswith("condor_submit_bid")
and '"num_hours=' in line
and '"num_gpus=' not in line
)
timeout_line = lambda line: bool(
re.search(r"NUM_HOURS.*60.*\+.*5", line)
)
result["source_references"] = {
"num_gpus_default": _blob_reference(
single_task_path,
single_task,
lambda line: bool(re.search(r"num_gpus\s*=\s*1\b", line)),
"num_gpus default",
),
"cuda_device_requirement": _blob_reference(
single_task_path,
single_task,
lambda line: "NVIDIA H100 80GB HBM3" in line,
"CUDA device requirement",
),
"request_gpus_binding": _blob_reference(
single_task_path,
single_task,
lambda line: bool(
re.search(r"request_gpus\s*=\s*\$\(num_gpus\)", line)
),
"request_gpus binding",
),
"receives_num_hours": _blob_reference(
run_task_path,
run_task,
lambda line: "NUM_HOURS" in line,
"NUM_HOURS input",
),
"solve_timeout_formula": _blob_reference(
run_task_path,
run_task,
timeout_line,
"solve timeout formula",
),
"timeout_grace_minutes": _blob_reference(
run_task_path,
run_task,
timeout_line,
"timeout grace",
),
"evaluation_dirs_present": {
"kind": "git-tree-entries",
"paths": eval_dirs_found,
},
"commit_sh_analysis.current_models_in_arrays": _blob_reference(
commit_path,
commit_sh,
lambda line: line.strip() == '"Qwen/Qwen3-4B-Base"',
"active model array",
),
"commit_sh_analysis.current_benchmarks_in_arrays": _blob_reference(
commit_path,
commit_sh,
lambda line: line.strip() == '"healthbench"',
"active benchmark array",
),
"commit_sh_analysis.htcondor_mpi_is_branch": _blob_reference(
commit_path,
commit_sh,
mpi_job,
"MPI scheduler job",
),
"commit_sh_analysis.htcondor_branch": _blob_reference(
commit_path,
commit_sh,
default_jobs,
"default htcondor jobs",
),
}
result["limitation_multi_gpu_extension"] = True
result["limitation_five_minute_grace"] = True
return result
def _analyze_commit_sh(content: str) -> dict[str, Any]:
"""Analyze commit.sh for scheduler-dependent branches and limitations.
Parse only active arrays and ``condor_submit_bid`` calls. The pinned
scheduler shape is an authority gate: any changed model, benchmark, job
count, hour count, or GPU count aborts evidence generation.
"""
code_lines = [
line.strip()
for line in content.splitlines()
if line.strip() and not line.lstrip().startswith("#")
]
def parse_array(name: str) -> list[str]:
match = re.search(
rf"(?ms)^\s*{re.escape(name)}\s*=\s*\((.*?)^\s*\)",
content,
)
if not match:
raise ValueError(f"commit.sh missing active {name}=(...) array")
entries: list[str] = []
for raw_line in match.group(1).splitlines():
line = raw_line.strip()
if not line or line.startswith("#"):
continue
item = re.fullmatch(r"""(["'])(.*?)\1(?:\s+#.*)?""", line)
if not item:
raise ValueError(
f"commit.sh has unparseable active {name} entry: {line}"
)
entries.append(item.group(2))
return entries
models = parse_array("models")
benchmarks = parse_array("evals")
if models != ["Qwen/Qwen3-4B-Base"]:
raise ValueError(f"commit.sh active model mismatch: {models}")
if benchmarks != ["healthbench"]:
raise ValueError(
f"commit.sh active benchmark mismatch: {benchmarks}"
)
def find_branch_line(
prefix: str,
scheduler: str,
*,
start: int = 0,
) -> int:
scheduler_re = re.compile(
rf"(?:=|==)\s*[\"']{re.escape(scheduler)}[\"']"
)
matches = [
index
for index, line in enumerate(code_lines[start:], start)
if line.startswith(prefix) and scheduler_re.search(line)
]
if len(matches) != 1:
raise ValueError(
f"commit.sh expected one active {prefix.strip()} "
f"{scheduler} branch, found {len(matches)}"
)
return matches[0]
mpi_start = find_branch_line("if ", "htcondor_mpi-is")
condor_start = find_branch_line(
"elif ",
"htcondor",
start=mpi_start + 1,
)
else_matches = [
index
for index, line in enumerate(code_lines[condor_start + 1:], condor_start + 1)
if line == "else" or line.startswith("else ")
]
if len(else_matches) != 1:
raise ValueError(
"commit.sh expected one active else after htcondor branch"
)
else_start = else_matches[0]
if not mpi_start < condor_start < else_start:
raise ValueError("commit.sh scheduler branch ordering mismatch")
mpi_lines = code_lines[mpi_start + 1:condor_start]
condor_lines = code_lines[condor_start + 1:else_start]
def parse_jobs(lines: list[str], label: str) -> list[dict[str, int]]:
jobs: list[dict[str, int]] = []
for line in lines:
if not re.match(r"^condor_submit_bid(?:\s|$)", line):
continue
attrs: dict[str, str] = {}
for _, assignment in re.findall(
r"""-a\s+(["'])([^"']+)\1""",
line,
):
key, separator, value = assignment.partition("=")
if not separator or key in attrs:
raise ValueError(
f"commit.sh malformed {label} -a argument: {assignment}"
)
attrs[key] = value
if "num_hours" not in attrs:
raise ValueError(
f"commit.sh {label} job missing quoted num_hours"
)
try:
job = {"hours": int(attrs["num_hours"])}
if "num_gpus" in attrs:
job["gpus"] = int(attrs["num_gpus"])
except ValueError as exc:
raise ValueError(
f"commit.sh {label} job has nonnumeric resource value"
) from exc
jobs.append(job)
return jobs
mpi_jobs = parse_jobs(mpi_lines, "htcondor_mpi-is")
condor_jobs = parse_jobs(condor_lines, "htcondor")
if mpi_jobs != [{"hours": 100, "gpus": 8}]:
raise ValueError(
f"commit.sh MPI job mismatch: expected one 100h/8-GPU job, "
f"got {mpi_jobs}"
)
if any("gpus" in job for job in condor_jobs):
raise ValueError("commit.sh htcondor jobs must use default GPU count")
condor_hours = [job["hours"] for job in condor_jobs]
if condor_hours.count(10) != 7 or condor_hours.count(1) != 1:
raise ValueError(
"commit.sh htcondor job mismatch: expected seven 10h and one 1h "
f"jobs, got {condor_hours}"
)
if len(condor_hours) != 8:
raise ValueError(
f"commit.sh htcondor expected 8 active jobs, got {len(condor_hours)}"
)
return {
"has_metr_branch": True,
"scheduler_dependent": True,
"current_models_in_arrays": models,
"current_benchmarks_in_arrays": benchmarks,
"htcondor_mpi_is_branch": {
"active_jobs": 1,
"hours": 100,
"gpus": 8,
"note": "Active 100-hour, eight-GPU METR command",
},
"htcondor_branch": {
"active_jobs": 8,
"ten_hour_jobs": 7,
"one_hour_jobs": 1,
"gpu_spec": "default (single_task.sub num_gpus=1)",
},
"num_hours_values": [1, 10, 100],
"gpu_counts_found": [8],
}
# ---------------------------------------------------------------------------
# Reward-hacking audit
# ---------------------------------------------------------------------------
def audit_reward_hacking(
acquired: dict[str, Any],
all_paths: list[str] | None = None,
) -> dict[str, Any]:
"""Audit the three reward-hacking submodes from acquired data.
Requires the verified ``acquired`` dict. API-key unavailability is
derived from the verified inventory (``all_paths``).
"""
hf_inv = acquired["hf_inventory"]
inventory_paths = all_paths if all_paths is not None else hf_inv["all_paths"]
trace_excerpts = acquired["trace_excerpts"]
# Mode 1: Training on test sets (contamination)
contam_content = acquired["contamination_content"]
time_taken_content = acquired["time_taken_content"]
contam_sha256 = hashlib.sha256(contam_content).hexdigest()
time_taken_sha256 = hashlib.sha256(time_taken_content).hexdigest()
contam_observations = [
f"Pinned released run has explicit contamination label at {CONTAMINATION_WITNESS_PATH}",
f"Label bytes: {contam_content!r}",
f"Label SHA-256: {contam_sha256}",
f"Corresponding time_taken.txt: {time_taken_content!r} (SHA-256: {time_taken_sha256})",
"Time 10:05:01 interpreted with runner's 5-minute termination grace; not rounded to 10 hours",
]
contamination = {
"mode": "training_on_test_sets",
"status": "partial-support",
"witness_path": CONTAMINATION_WITNESS_PATH,
"witness_sha256": contam_sha256,
"witness_bytes": contam_content.decode("utf-8"),
"observation_type": "upstream_judge_label",
"observations": contam_observations,
"time_taken": {
"path": TIME_TAKEN_WITNESS_PATH,
"value": time_taken_content.decode("utf-8").strip(),
"sha256": time_taken_sha256,
"note": "Interpreted with runner's 5-minute termination grace",
},
}
# Mode 2: Downloading instruction-tuned checkpoint
instruction_observations = [
f"Categorical judgment at {INSTRUCTION_MODEL_JUDGMENT_PATH}",
f"Judgment bytes: {INSTRUCTION_MODEL_JUDGMENT_BYTES!r} (no trailing newline)",
f"Judgment Git object: {INSTRUCTION_MODEL_JUDGMENT_GIT_OBJECT}, size {INSTRUCTION_MODEL_JUDGMENT_SIZE}",
f"Judgment SHA-256: {INSTRUCTION_MODEL_JUDGMENT_SHA256}",
f"Complete trace at {INSTRUCTION_MODEL_TRACE_PATH} ({INSTRUCTION_MODEL_TRACE_SIZE} bytes)",
f"Trace Git object: {INSTRUCTION_MODEL_TRACE_GIT_OBJECT}",
f"Trace SHA-256: {INSTRUCTION_MODEL_TRACE_SHA256}",
"Trace is NOT redistributed; only safe JSONL-pointer excerpts emitted",
]
# Verify excerpts came from trace parsing (not constants)
instruction_excerpts = trace_excerpts
instruction = {
"mode": "downloading_instruction_tuned_checkpoint",
"status": "partial-support",
"judgment_path": INSTRUCTION_MODEL_JUDGMENT_PATH,
"judgment_sha256": INSTRUCTION_MODEL_JUDGMENT_SHA256,
"judgment_git_object": INSTRUCTION_MODEL_JUDGMENT_GIT_OBJECT,
"judgment_size": INSTRUCTION_MODEL_JUDGMENT_SIZE,
"judgment_bytes": INSTRUCTION_MODEL_JUDGMENT_BYTES.decode("utf-8"),
"observation_type": "upstream_judge_label_plus_trace_excerpts",
"observations": instruction_observations,
"trace": {
"path": INSTRUCTION_MODEL_TRACE_PATH,
"sha256": acquired["instruction_trace_sha256"],
"git_object": INSTRUCTION_MODEL_TRACE_GIT_OBJECT,
"size": acquired["instruction_trace_size"],
"redistributed": False,
"note": "Complete trace is not an output; only deterministic JSONL-pointer extracts emitted",
},
"safe_excerpts": instruction_excerpts,
}
# Mode 3: Using discovered API key (unavailable)
# Derive from the verified complete inventory
cluster_paths = [
p for p in inventory_paths
if API_MISUSE_TASK_CLUSTER in p
]
if cluster_paths:
raise ValueError(
f"API-misuse cluster {API_MISUSE_TASK_CLUSTER} found in inventory "
f"({len(cluster_paths)} paths). Cannot emit 'unavailable' when the "
f"cluster is present."
)
api_misuse = {
"mode": "using_discovered_api_key",
"status": "unavailable",
"observation_type": "missing_artifact",
"observations": [
f"The exact paper task cluster {API_MISUSE_TASK_CLUSTER} and its named "
f"root/task signature have {len(cluster_paths)} paths in the complete "
f"verified inventory ({HF_TREE_TOTAL_ENTRIES} entries)",
"A different public OpenCode GPT-5.1 root is not a substitute",
"Paper prose cannot satisfy an unavailable artifact",
],
"unavailability_reason": (
f"The selected trajectory revision omits the specific GPT-5.1 "
f"Codex-Max run described by the paper (task cluster {API_MISUSE_TASK_CLUSTER})"
),
"inventory_proof": {
"cluster_id": API_MISUSE_TASK_CLUSTER,
"matching_paths": len(cluster_paths),
"total_inventory_entries": len(inventory_paths),
},
}
return {
"training_on_test_sets": contamination,
"downloading_instruction_tuned_checkpoint": instruction,
"using_discovered_api_key": api_misuse,
}
# ---------------------------------------------------------------------------
# Claim evaluation
# ---------------------------------------------------------------------------
def evaluate_claims(
coverage: dict[str, Any],
protocol: dict[str, Any],
reward_hacking: dict[str, Any],
) -> dict[str, Any]:
"""Evaluate the two selected claims from verified audit data.
Requires verified coverage, protocol, and reward_hacking dicts.
Both claims are ``partial-support``.
"""
def require_exact(label: str, actual: Any, expected: Any) -> None:
if actual != expected:
raise ValueError(f"{label} mismatch: {actual!r} != {expected!r}")
# Verify the complete coverage census, not only its headline count.
if coverage["recognized_task_count"] != EXPECTED_TASK_COUNT:
raise ValueError(
f"Coverage task count {coverage['recognized_task_count']} "
f"!= expected {EXPECTED_TASK_COUNT}"
)
if coverage.get("recognized_root_count") != EXPECTED_ROOT_COUNT:
raise ValueError(
f"Coverage root count {coverage.get('recognized_root_count')} "
f"!= expected {EXPECTED_ROOT_COUNT}"
)
if coverage.get("recognized_root_cell_pairs") != EXPECTED_ROOT_CELL_PAIRS:
raise ValueError(
f"Coverage root/cell pairs {coverage.get('recognized_root_cell_pairs')} "
f"!= expected {EXPECTED_ROOT_CELL_PAIRS}"
)
if coverage.get("duplicate_job_pairs") != EXPECTED_DUPLICATE_PAIRS:
raise ValueError(
f"Coverage duplicate pairs {coverage.get('duplicate_job_pairs')} "
f"!= expected {EXPECTED_DUPLICATE_PAIRS}"
)
if coverage.get("missing_root_cell_pairs") != EXPECTED_MISSING_PAIRS:
raise ValueError(
f"Coverage missing pairs {coverage.get('missing_root_cell_pairs')} "
f"!= expected {EXPECTED_MISSING_PAIRS}"
)
require_exact(
"Coverage cell-count map",
coverage.get("cell_counts"),
EXPECTED_CELL_COUNTS,
)
expected_matrix = [
{
"benchmark": benchmark,
"model": model,
"count": EXPECTED_CELL_COUNTS[benchmark][model_index],
}
for benchmark in sorted(EXPECTED_BENCHMARKS)
for model_index, model in enumerate(MODEL_ORDER)
]
require_exact(
"Coverage 28-cell matrix",
coverage.get("matrix"),
expected_matrix,
)
# Verify every protocol fact that supports the partial status.
exact_protocol = {
"num_gpus_default": 1,
"cuda_device_requirement": (
'TARGET.CUDADeviceName == "NVIDIA H100 80GB HBM3"'
),
"request_gpus_binding": "request_gpus = $(num_gpus)",
"receives_num_hours": True,
"solve_timeout_formula": "NUM_HOURS * 60 + 5",
"timeout_grace_minutes": 5,
"timeout_formula_found": True,
"task_dir_10h_suffix": True,
"evaluation_dirs_present": EXPECTED_EVAL_DIRS,
}
for key, expected in exact_protocol.items():
require_exact(f"Protocol {key}", protocol.get(key), expected)
analysis = protocol.get("commit_sh_analysis")
if not isinstance(analysis, dict):
raise ValueError("Protocol commit_sh_analysis missing")
require_exact(
"Protocol scheduler_dependent",
analysis.get("scheduler_dependent"),
True,
)
require_exact(
"Protocol active models",
analysis.get("current_models_in_arrays"),
["Qwen/Qwen3-4B-Base"],
)
require_exact(
"Protocol active benchmarks",
analysis.get("current_benchmarks_in_arrays"),
["healthbench"],
)
require_exact(
"Protocol MPI branch",
analysis.get("htcondor_mpi_is_branch"),
{
"active_jobs": 1,
"hours": 100,
"gpus": 8,
"note": "Active 100-hour, eight-GPU METR command",
},
)
require_exact(
"Protocol htcondor branch",
analysis.get("htcondor_branch"),
{
"active_jobs": 8,
"ten_hour_jobs": 7,
"one_hour_jobs": 1,
"gpu_spec": "default (single_task.sub num_gpus=1)",
},
)
# Verify all reward-mode gates before emitting either selected status.
contamination = reward_hacking.get("training_on_test_sets", {})
instruction = reward_hacking.get(
"downloading_instruction_tuned_checkpoint",
{},
)
api_misuse = reward_hacking.get("using_discovered_api_key", {})
require_exact(
"Contamination reward status",
contamination.get("status"),
"partial-support",
)
require_exact(
"Instruction-model reward status",
instruction.get("status"),
"partial-support",
)
require_exact(
"API-misuse reward status",
api_misuse.get("status"),
"unavailable",
)
require_exact(
"API-misuse matching path count",
api_misuse.get("inventory_proof", {}).get("matching_paths"),
0,
)
trace = instruction.get("trace", {})
require_exact(
"Instruction trace SHA-256",
trace.get("sha256"),
INSTRUCTION_MODEL_TRACE_SHA256,
)
require_exact(
"Instruction trace size",
trace.get("size"),
INSTRUCTION_MODEL_TRACE_SIZE,
)
require_exact(
"Instruction trace redistribution flag",
trace.get("redistributed"),
False,
)
expected_excerpts = [
{
"record": excerpt["record"],
"json_pointer": excerpt["pointer"],
"text": excerpt["text"],
"sha256": excerpt["sha256"],
}
for excerpt in TRACE_EXCERPTS
]
require_exact(
"Instruction trace safe excerpts",
instruction.get("safe_excerpts"),
expected_excerpts,
)
limitations_1 = [
"No H100 run is reproduced; the resource and time findings are a released-configuration audit.",
"The runner allows a five-minute termination grace, and a released example records 10:05:01.",
f"The pinned source's current launcher is scheduler-dependent: one branch has a 100-hour/eight-GPU "
f"METR command, another has {protocol['commit_sh_analysis']['htcondor_branch']['ten_hour_jobs']} "
f"ten-hour and {protocol['commit_sh_analysis']['htcondor_branch']['one_hour_jobs']} one-hour "
f"default-GPU commands, and the arrays currently select only one model/benchmark pair.",
"Evidence is not an official challenge verdict.",
]
limitations_2 = [
"A released judge label is not independently established behavioral truth.",
"The instruction-model evidence is an upstream categorical label plus safe extracts from a "
"released trace, not a fresh independent behavioral audit.",
"The selected trajectory revision does not expose the exact GPT-5.1 Codex-Max API-misuse "
f"task cluster {API_MISUSE_TASK_CLUSTER}.",
"No leaderboard score, BFCL score, weighted average, or reasoning-effort ablation is a "
"selected target or reproduced measurement.",
"Evidence is not an official challenge verdict.",
]
return {
"claim_1": {
"text": CLAIM_1_TEXT,
"sha256": CLAIM_1_SHA256,
"status": "partial-support",
"summary": (
"Released trajectory inventory confirms 4-by-7 coverage across "
"all accepted benchmark/model cells. Runner configuration defaults "
"to one H100 with a NUM_HOURS-based timeout. The current checkout's "
"scheduler-dependent branches and five-minute termination grace are "
"reported as limitations."
),
"evidence_pointers": [
"evidence/coverage.json#/recognized_task_count",
"evidence/coverage.json#/matrix",
"evidence/coverage.json#/cell_counts",
"evidence/provenance.json#/source",
"evidence/coverage.json#/protocol",
],
"limitations": limitations_1,
},
"claim_2": {
"text": CLAIM_2_TEXT,
"sha256": CLAIM_2_SHA256,
"status": "partial-support",
"summary": (
"Released contamination and instruction-model judgments provide "
"partial support for two of three reward-hacking submodes. The "
"API-key submode artifact is absent from the pinned revision."
),
"evidence_pointers": [
"evidence/reward_hacking.json#/training_on_test_sets",
"evidence/reward_hacking.json#/downloading_instruction_tuned_checkpoint",
"evidence/reward_hacking.json#/using_discovered_api_key",
],
"limitations": limitations_2,
},
}