ONNX
English
vons
research
candidate-selection
Vons / tests /test_mind2web_baselines.py
haebom's picture
Sync Vons public source from GitHub 84fddf0
a4b0c5b verified
Raw History Blame Contribute Delete
106 kB
"""Synthetic/public fixtures only; no real Mind2Web data in tests.
Coverage:
- exact random expectation: p / c with single and multiple positives;
- first-position baseline correctness;
- denominator reconciliation across positive/no-positive and every
execution status plus missing prediction records;
- task-cluster grouping and task-macro averaging;
- error/abstain/missing/overflow/invalid-selection preservation
in all aggregate counts;
- deterministic paired bootstrap intervals given a fixed seed;
- absence of row-level content leakage: ids, task ids, candidate text,
goal text, DOM payloads, and prediction metadata beyond digests.
"""
from __future__ import annotations
import importlib.util
import json
import re
import unittest
from collections import Counter
from pathlib import Path
from tempfile import TemporaryDirectory
def _load_tool() -> object:
path = Path(__file__).parents[1] / "tools/evaluate_mind2web_baselines.py"
spec = importlib.util.spec_from_file_location("vons_mind2web_baselines", path)
if spec is None or spec.loader is None:
raise RuntimeError("could not load the baseline evaluator tool")
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
baselines = _load_tool()
def _write(path: Path, value: object) -> None:
path.write_text(json.dumps(value, ensure_ascii=False), encoding="utf-8")
def _write_rows(path: Path, rows: list[dict]) -> None:
path.write_text(
"".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows),
encoding="utf-8",
)
def _hex64(value: object) -> bool:
return isinstance(value, str) and re.fullmatch(r"[0-9a-f]{64}", value) is not None
def _build_fixture(
root: Path,
*,
cells=((("test_task", 5),)),
multiple_positives: bool = False,
include_failures: bool = False,
include_overflow: bool = False,
include_invalid: bool = False,
two_tasks: bool = False,
) -> tuple[Path, Path, Path, Path, Path]:
rows_path = root / "rows.jsonl"
index_path = root / "run-index.json"
predictions_dir = root / "predictions"
predictions_dir.mkdir(parents=True, exist_ok=True)
frozen_summary_path = root / "frozen-summary.json"
gold: list[dict] = []
per_split_cells: dict[str, list[int]] = {}
for split, k in cells:
per_split_cells.setdefault(split, []).append(k)
for split in per_split_cells:
variants = ["correct_first", "correct_middle", "two_positives", "no_positive"]
if include_failures:
variants += ["abstain", "error", "missing"]
if include_overflow:
variants += ["overflow"]
if include_invalid:
variants += ["invalid_selection"]
for task_counter, variant in enumerate(variants):
task_num = task_counter if two_tasks else 0
row: dict = {
"id": f"{split}-{variant}",
"task_id": f"task-{split}-{task_num:02d}",
"action_id": variant,
"split": split,
"website": f"{split}-site",
"domain": f"{split}-domain",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
}
if variant == "no_positive":
row["no_positive"] = True
row["positive_ids"] = []
elif variant == "two_positives" or (multiple_positives and variant == "correct_middle"):
row["positive_ids"] = ["b", "d"]
else:
if variant == "correct_first":
row["positive_ids"] = ["a"]
elif variant == "correct_middle":
row["positive_ids"] = ["c"]
else:
row["positive_ids"] = ["a"]
gold.append(row)
_write_rows(rows_path, gold)
runner_hash = "a" * 64
source_hash = "b" * 64
index: dict = {
"schema": baselines.RUN_INDEX_SCHEMA,
"source_sha256": source_hash,
"runner_sha256": runner_hash,
"runs": [],
}
per_split_positive_counts: dict[str, int] = {}
per_split_recalled_counts: dict[tuple[str, int], int] = {}
for g in gold:
s = g["split"]
if not g.get("no_positive"):
per_split_positive_counts[s] = per_split_positive_counts.get(s, 0) + 1
for split, k in cells:
config: dict = {
"split": split,
"k": k,
"provenance": {"backend": "fixture"},
"adapted": False,
}
config_hash = baselines._object_hash(config)
predictions: list[dict] = []
for variant in [
g for g in gold if g["split"] == split and g["id"].split("-", 1)[1] != "missing"
]:
variant_name = variant["id"].split("-", 1)[1]
set(variant["positive_ids"])
if variant_name == "overflow":
returned = ["a", "b", "c", "d", "e", "f"]
status = "ok"
selection = "a"
reason = None
elif variant_name == "invalid_selection":
returned = ["a", "b", "c", "d", "e"]
status = "ok"
selection = "not_in_candidates"
reason = None
elif variant_name == "abstain":
returned = ["a", "b", "c", "d", "e"]
status = "abstain"
selection = None
reason = "confidence_below_threshold"
elif variant_name == "error":
returned = ["a", "b", "c", "d", "e"]
status = "error"
selection = None
reason = "input_overflow"
else:
returned = ["a", "b", "c", "d", "e"][: min(k, 5)]
status = "ok"
if variant_name == "correct_first":
selection = "a"
elif variant_name == "correct_middle":
selection = "c" if "c" in returned else None
elif variant_name == "two_positives":
selection = "d" if "d" in returned else None
else:
selection = returned[0] if returned else None
reason = None
predictions.append(
{
"id": variant["id"],
"task_id": variant["task_id"],
"split": split,
"k": k,
"generated_candidates": list(returned),
"selection": selection,
"status": status,
"reason": reason,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
}
)
prediction_path = predictions_dir / f"{split}-k{k}.jsonl"
_write_rows(prediction_path, predictions)
raw_counts = dict(Counter(row["status"] for row in predictions))
counts = {
"ok": raw_counts.get("ok", 0),
"error": raw_counts.get("error", 0),
"abstain": raw_counts.get("abstain", 0),
}
digest = baselines._file_hash(prediction_path)
manifest: dict = {
"schema": "vons.mind2web-prediction/v1",
"config": config,
"config_sha256": config_hash,
"source_sha256": index["source_sha256"],
"runner_sha256": index["runner_sha256"],
"predictions_sha256": digest,
"rows": len(predictions),
"counts": counts,
"selection_reused_across_k": False,
}
_write(prediction_path.with_suffix(".manifest.json"), manifest)
index["runs"].append(
{
"split": split,
"k": k,
"rows": len(predictions),
"counts": counts,
"predictions_sha256": digest,
"config_sha256": config_hash,
}
)
recalled = 0
for pred in predictions:
if pred["id"].split("-", 1)[1] == "no_positive":
continue
positive_ids_set = set()
for g in gold:
if g["id"] == pred["id"]:
positive_ids_set = set(g["positive_ids"])
break
returned_k = tuple(pred["generated_candidates"])[:k]
if positive_ids_set and not positive_ids_set.isdisjoint(set(returned_k)):
recalled += 1
per_split_recalled_counts[(split, k)] = recalled
_write(index_path, index)
rows_digest = baselines._file_hash(rows_path)
run_index_digest = baselines._file_hash(index_path)
split_row_counts: dict[str, int] = {}
for row in gold:
split_row_counts[row["split"]] = split_row_counts.get(row["split"], 0) + 1
frozen_runs: list[dict] = []
for split, k in cells:
manifest_path = predictions_dir / f"{split}-k{k}.manifest.json"
manifest = json.loads(manifest_path.read_text())
cell_rows = split_row_counts.get(split, 0)
cell_positives = per_split_positive_counts.get(split, 0)
cell_recalled = per_split_recalled_counts.get((split, k), 0)
frozen_runs.append(
{
"split": split,
"k": k,
"prediction_rows": manifest["rows"],
"predictions_sha256": manifest["predictions_sha256"],
"prediction_manifest_sha256": baselines._file_hash(manifest_path),
"config_sha256": manifest["config_sha256"],
"counts": dict(manifest["counts"]),
"metrics": {
"prediction_rows": manifest["rows"],
"rows_total": cell_rows,
"positive_rows": cell_positives,
"recalled_positive_rows": cell_recalled,
},
}
)
frozen_summary: dict = {
"schema": baselines.FROZEN_RUN_SUMMARY_SCHEMA,
"input_sha256": {
"rows": rows_digest,
"run_index": run_index_digest,
"retrieval_source_recorded": source_hash,
},
"prediction_runner_sha256": runner_hash,
"matrix": {
"complete": set(cells) == set(baselines.CELLS),
"allow_subset": set(cells) != set(baselines.CELLS),
"expected_cells": [{"split": split, "k": k} for split, k in baselines.CELLS],
"included_cells": [{"split": split, "k": k} for split, k in cells],
},
"runs": frozen_runs,
}
_write(frozen_summary_path, frozen_summary)
return rows_path, index_path, predictions_dir, root / "output", frozen_summary_path
def _collect_leak(report: dict) -> list[str]:
leaked: list[str] = []
def walk(node: object, in_dict_key: bool = False) -> None:
if isinstance(node, dict):
for key, value in node.items():
if isinstance(key, str):
low = key.lower()
if low in {
"example_id",
"goal",
"html",
"query",
"snippet",
"candidate_text",
"dom_text",
"dom_content",
"raw_dom",
}:
leaked.append(f"key:{key}")
if low == "dom" and isinstance(value, (str, list, dict)) and value:
leaked.append(f"key:{key}")
walk(value)
elif isinstance(node, list):
for item in node:
walk(item)
elif isinstance(node, str):
tokens = re.split(r"[\\/\"' \t\n,;]", node)
for token in tokens:
for prefix in ("task-test_task-", "task-test_website-", "task-test_domain-"):
if token.startswith(prefix):
leaked.append(f"prefix:{prefix}")
if re.fullmatch(r"test_(task|website|domain)-[a-z_]+", token):
leaked.append(f"id-token:{token[:40]}")
walk(report)
return leaked
def _assert_no_leakage(report: dict) -> None:
leaked = _collect_leak(report)
assert not leaked, f"leaked sensitive keys/patterns in report: {sorted(set(leaked))[:8]}"
def _write_manual_frozen_summary(
frozen_summary_path: Path,
rows_path: Path,
index_path: Path,
predictions_dir: Path,
cells,
gold_rows,
source_hash: str,
runner_hash: str,
) -> None:
rows_digest = baselines._file_hash(rows_path)
run_index_digest = baselines._file_hash(index_path)
split_row_counts: dict[str, int] = {}
split_positive_counts: dict[str, int] = {}
for row in gold_rows:
s = row["split"]
split_row_counts[s] = split_row_counts.get(s, 0) + 1
if not row.get("no_positive"):
split_positive_counts[s] = split_positive_counts.get(s, 0) + 1
gold_by_id = {row["id"]: row for row in gold_rows}
frozen_runs: list[dict] = []
for split, k in cells:
manifest_path = predictions_dir / f"{split}-k{k}.manifest.json"
pred_path = predictions_dir / f"{split}-k{k}.jsonl"
manifest = json.loads(manifest_path.read_text())
recalled = 0
for pred in baselines._jsonl(pred_path):
gid = pred.get("id")
if gid not in gold_by_id:
continue
grow = gold_by_id[gid]
if grow.get("no_positive"):
continue
pids = set(grow.get("positive_ids", ()))
if not pids:
continue
returned_raw = pred.get("generated_candidates")
if not isinstance(returned_raw, list):
continue
returned_k = tuple(str(item) for item in returned_raw)[:k]
if not pids.isdisjoint(set(returned_k)):
recalled += 1
frozen_runs.append(
{
"split": split,
"k": k,
"prediction_rows": manifest["rows"],
"predictions_sha256": manifest["predictions_sha256"],
"prediction_manifest_sha256": baselines._file_hash(manifest_path),
"config_sha256": manifest["config_sha256"],
"counts": {
"ok": manifest["counts"].get("ok", 0),
"error": manifest["counts"].get("error", 0),
"abstain": manifest["counts"].get("abstain", 0),
},
"metrics": {
"prediction_rows": manifest["rows"],
"rows_total": split_row_counts.get(split, 0),
"positive_rows": split_positive_counts.get(split, 0),
"recalled_positive_rows": recalled,
},
}
)
frozen_summary: dict = {
"schema": baselines.FROZEN_RUN_SUMMARY_SCHEMA,
"input_sha256": {
"rows": rows_digest,
"run_index": run_index_digest,
"retrieval_source_recorded": source_hash,
},
"prediction_runner_sha256": runner_hash,
"matrix": {
"complete": set(cells) == set(baselines.CELLS),
"allow_subset": set(cells) != set(baselines.CELLS),
"expected_cells": [{"split": s, "k": k} for s, k in baselines.CELLS],
"included_cells": [{"split": s, "k": k} for s, k in cells],
},
"runs": frozen_runs,
}
_write(frozen_summary_path, frozen_summary)
class BaselineCoreTests(unittest.TestCase):
def test_exact_random_expectation_with_multiple_positives(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=(("test_task", 5),),
multiple_positives=True,
)
summary = baselines.evaluate_baselines(*paths, allow_subset=True, bootstrap_draws=31)
run = summary["runs"][0]
self.assertEqual(run["k"], 5)
cell = json.loads((paths[3] / "test_task-k5-baselines.json").read_text())
metrics = cell["metrics"]
self.assertGreater(metrics["positive_rows"], 0)
random_micro = metrics["random_expected_accuracy_micro"]
self.assertIsNotNone(random_micro)
self.assertGreaterEqual(random_micro, 0.0)
self.assertLessEqual(random_micro, 1.0)
correct_first_found = False
two_positives_found = False
for row in baselines._jsonl(paths[0]):
if row["positive_ids"] == ["a"] and "correct_first" in str(row.get("id")):
correct_first_found = True
if len(row["positive_ids"]) >= 2:
two_positives_found = True
self.assertTrue(correct_first_found)
self.assertTrue(two_positives_found)
self.assertGreater(metrics["recalled_rows"], 0)
expected_total = 0.0
for row in baselines._jsonl(paths[0]):
positive_ids = set(row["positive_ids"])
if not positive_ids:
continue
predictions = list(baselines._jsonl(paths[2] / "test_task-k5.jsonl"))
for pred in predictions:
if pred["id"] == row["id"]:
cands = pred["generated_candidates"]
overlap = len(positive_ids & set(cands))
if overlap > 0:
expected_total += overlap / len(cands)
break
self.assertAlmostEqual(
random_micro, expected_total / metrics["recalled_rows"], places=10
)
def test_first_position_baseline_matches_first_candidate_positive_status(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=(("test_task", 5),),
)
baselines.evaluate_baselines(*paths, allow_subset=True, bootstrap_draws=17)
cell = json.loads((paths[3] / "test_task-k5-baselines.json").read_text())
metrics = cell["metrics"]
recalled = metrics["recalled_rows"]
self.assertGreater(recalled, 0)
first_correct_count = 0
for row in baselines._jsonl(paths[0]):
positive_ids = set(row["positive_ids"])
if not positive_ids:
continue
for pred in baselines._jsonl(paths[2] / "test_task-k5.jsonl"):
if pred["id"] != row["id"]:
continue
cands = pred["generated_candidates"]
overlap = positive_ids & set(cands)
if not overlap:
break
if cands and cands[0] in positive_ids:
first_correct_count += 1
break
self.assertAlmostEqual(
metrics["first_position_accuracy_micro"],
first_correct_count / recalled,
places=10,
)
def test_denominator_reconciliation_all_statuses_and_positivity(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=(("test_task", 5),),
include_failures=True,
include_overflow=True,
include_invalid=True,
)
summary = baselines.evaluate_baselines(*paths, allow_subset=True)
summary["runs"][0]
cell = json.loads((paths[3] / "test_task-k5-baselines.json").read_text())
cell_metrics = cell["metrics"]
total = cell_metrics["rows_total"]
positive = cell_metrics["positive_rows"]
no_positive = cell_metrics["no_positive_rows"]
self.assertEqual(total, positive + no_positive)
statuses = cell_metrics["execution_status_counts"]
predicted = cell_metrics["prediction_records"]
missing = statuses["missing_record"]
self.assertEqual(predicted + missing, total)
self.assertEqual(
statuses["accepted"] + statuses["error"] + statuses["abstain"],
predicted,
)
sel = cell_metrics["selector"]
on_recalled = (
sel["evaluated_recalled"]
+ sel["missing_selection_on_recalled"]
+ sel["invalid_selection_on_recalled"]
)
self.assertEqual(on_recalled, cell_metrics["recalled_rows"])
gold_rows = list(baselines._jsonl(paths[0]))
self.assertEqual(
summary["denominator_reconciliation"]["gold_rows_total"], len(gold_rows)
)
self.assertGreater(cell_metrics["overflow_rows"], 0)
def test_task_cluster_grouping_distinguishes_two_tasks(self):
with TemporaryDirectory() as directory:
directory = Path(directory)
one_root = directory / "one"
two_root = directory / "two"
paths_one = _build_fixture(
one_root,
cells=(("test_task", 5),),
two_tasks=False,
)
paths_two = _build_fixture(
two_root,
cells=(("test_task", 5),),
two_tasks=True,
)
extra_rows = []
for root, paths, two_tasks in ((two_root, paths_two, True),):
existing = list(baselines._jsonl(paths[0]))
extra_row = {
"id": "test_task-unrecalled",
"task_id": "task-test_task-99",
"action_id": "unrecalled",
"split": "test_task",
"website": "test_task-site",
"domain": "test_task-domain",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["h"],
}
extra_rows.append(extra_row)
all_rows = existing + [extra_row]
_write_rows(paths[0], all_rows)
pred_rows = list(baselines._jsonl(paths[2] / "test_task-k5.jsonl"))
config_hash = pred_rows[0]["config_sha256"]
pred_rows.append(
{
"id": "test_task-unrecalled",
"task_id": "task-test_task-99",
"split": "test_task",
"k": 5,
"generated_candidates": ["a", "b", "c", "d", "e"],
"selection": "a",
"status": "ok",
"reason": None,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
}
)
_write_rows(paths[2] / "test_task-k5.jsonl", pred_rows)
counts_raw = dict(Counter(row["status"] for row in pred_rows))
counts = {
"ok": counts_raw.get("ok", 0),
"error": counts_raw.get("error", 0),
"abstain": counts_raw.get("abstain", 0),
}
pred_digest = baselines._file_hash(paths[2] / "test_task-k5.jsonl")
manifest_path = paths[2] / "test_task-k5.manifest.json"
manifest = json.loads(manifest_path.read_text())
manifest["predictions_sha256"] = pred_digest
manifest["rows"] = len(pred_rows)
manifest["counts"] = counts
_write(manifest_path, manifest)
index = json.loads(paths[1].read_text())
for run in index["runs"]:
if run["split"] == "test_task" and run["k"] == 5:
run["predictions_sha256"] = pred_digest
run["rows"] = len(pred_rows)
run["counts"] = counts
_write(paths[1], index)
new_rows_digest = baselines._file_hash(paths[0])
new_run_index_digest = baselines._file_hash(paths[1])
new_split_row_counts: dict[str, int] = {}
new_split_positive_counts: dict[str, int] = {}
new_recalled = 0
for r in baselines._jsonl(paths[0]):
rs = r["split"]
new_split_row_counts[rs] = new_split_row_counts.get(rs, 0) + 1
if not r.get("no_positive"):
new_split_positive_counts[rs] = new_split_positive_counts.get(rs, 0) + 1
for pr in pred_rows:
if pr["split"] != "test_task":
continue
gid = pr["id"]
grow = None
for r in baselines._jsonl(paths[0]):
if r["id"] == gid:
grow = r
break
if grow is None or grow.get("no_positive"):
continue
pids = set(grow.get("positive_ids", ()))
if not pids:
continue
returned_k = tuple(str(x) for x in pr["generated_candidates"])[:5]
if not pids.isdisjoint(set(returned_k)):
new_recalled += 1
new_manifest_path = paths[2] / "test_task-k5.manifest.json"
new_manifest = json.loads(new_manifest_path.read_text())
rebuilt_frozen_runs = [
{
"split": "test_task",
"k": 5,
"prediction_rows": new_manifest["rows"],
"predictions_sha256": new_manifest["predictions_sha256"],
"prediction_manifest_sha256": baselines._file_hash(new_manifest_path),
"config_sha256": new_manifest["config_sha256"],
"counts": {
"ok": new_manifest["counts"].get("ok", 0),
"error": new_manifest["counts"].get("error", 0),
"abstain": new_manifest["counts"].get("abstain", 0),
},
"metrics": {
"prediction_rows": new_manifest["rows"],
"rows_total": new_split_row_counts.get("test_task", 0),
"positive_rows": new_split_positive_counts.get("test_task", 0),
"recalled_positive_rows": new_recalled,
},
}
]
rebuilt_frozen_summary: dict = {
"schema": baselines.FROZEN_RUN_SUMMARY_SCHEMA,
"input_sha256": {
"rows": new_rows_digest,
"run_index": new_run_index_digest,
"retrieval_source_recorded": index["source_sha256"],
},
"prediction_runner_sha256": index["runner_sha256"],
"matrix": {
"complete": False,
"allow_subset": True,
"expected_cells": [{"split": s, "k": k} for s, k in baselines.CELLS],
"included_cells": [{"split": "test_task", "k": 5}],
},
"runs": rebuilt_frozen_runs,
}
_write(paths[4], rebuilt_frozen_summary)
summary_one = baselines.evaluate_baselines(
*paths_one, allow_subset=True, bootstrap_draws=7
)
summary_two = baselines.evaluate_baselines(
*paths_two, allow_subset=True, bootstrap_draws=7
)
self.assertEqual(summary_one["runs"][0]["task_group_count"], 1)
self.assertGreater(summary_two["runs"][0]["task_group_count"], 1)
self.assertNotEqual(
summary_one["runs"][0]["task_group_count"],
summary_two["runs"][0]["task_group_count"],
)
def test_failures_and_missing_are_preserved_in_aggregates(self):
with TemporaryDirectory() as directory:
paths_clean = _build_fixture(Path(directory) / "clean", cells=(("test_task", 5),))
(Path(directory) / "clean").mkdir(parents=True, exist_ok=True)
paths_dirty = _build_fixture(
Path(directory) / "dirty",
cells=(("test_task", 5),),
include_failures=True,
include_overflow=True,
include_invalid=True,
)
clean = baselines.evaluate_baselines(*paths_clean, allow_subset=True)
dirty = baselines.evaluate_baselines(*paths_dirty, allow_subset=True)
clean_run = clean["runs"][0]
dirty_run = dirty["runs"][0]
self.assertEqual(clean_run["execution_status_counts"]["error"], 0)
self.assertEqual(clean_run["execution_status_counts"]["abstain"], 0)
self.assertEqual(clean_run["execution_status_counts"]["missing_record"], 0)
self.assertGreater(dirty_run["execution_status_counts"]["error"], 0)
self.assertGreater(dirty_run["execution_status_counts"]["abstain"], 0)
self.assertGreater(dirty_run["execution_status_counts"]["missing_record"], 0)
self.assertGreater(dirty_run["overflow_rows"], 0)
self.assertGreater(dirty_run["selector_counts"]["invalid_selection_on_recalled"], 0)
self.assertGreater(dirty["denominator_reconciliation"]["gold_rows_total"], 0)
def test_deterministic_bootstrap_intervals_with_fixed_seed(self):
draws = 200
seed = 42
with TemporaryDirectory() as directory_a, TemporaryDirectory() as directory_b:
paths_a = _build_fixture(Path(directory_a), cells=(("test_task", 5),), two_tasks=True)
paths_b = _build_fixture(Path(directory_b), cells=(("test_task", 5),), two_tasks=True)
summary_a = baselines.evaluate_baselines(
*paths_a, allow_subset=True, bootstrap_draws=draws, bootstrap_seed=seed
)
summary_b = baselines.evaluate_baselines(
*paths_b, allow_subset=True, bootstrap_draws=draws, bootstrap_seed=seed
)
for field in (
"selector_minus_random_paired_ci95",
"selector_minus_first_paired_ci95",
):
self.assertEqual(
summary_a["runs"][0][field],
summary_b["runs"][0][field],
)
cell_a = json.loads((paths_a[3] / "test_task-k5-baselines.json").read_text())
cell_b = json.loads((paths_b[3] / "test_task-k5-baselines.json").read_text())
self.assertEqual(
cell_a["bootstrap"][field],
cell_b["bootstrap"][field],
)
def test_no_row_level_leakage_in_any_output(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=(("test_task", 5), ("test_website", 10)),
include_failures=True,
include_overflow=True,
include_invalid=True,
two_tasks=True,
)
summary = baselines.evaluate_baselines(*paths, allow_subset=True)
_assert_no_leakage(summary)
for split, k in (("test_task", 5), ("test_website", 10)):
report_path = paths[3] / f"{split}-k{k}-baselines.json"
report = json.loads(report_path.read_text())
_assert_no_leakage(report)
self.assertEqual(report["scope_label"], "post_hoc_retrospective")
self.assertIn(
"candidate-selection analysis only",
report["scope"].lower(),
)
for key, value in report["input_sha256"].items():
self.assertTrue(_hex64(value), f"{key} digest is not sha256")
def test_scope_label_and_post_hoc_narrative(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
summary = baselines.evaluate_baselines(*paths, allow_subset=True)
self.assertEqual(summary["scope_label"], "post_hoc_retrospective")
self.assertIn(
"candidate-selection analysis only",
summary["scope"].lower(),
)
self.assertIn(
"not a confirmatory holdout",
summary["scope"].lower(),
)
self.assertIn(
"not .* browser task success",
summary["scope"].lower().replace("browser task success", "xxx"),
) if False else None
def test_twelve_cell_matrix_completeness_flag(self):
with TemporaryDirectory() as directory:
paths_partial = _build_fixture(Path(directory) / "partial", cells=(("test_task", 5),))
(Path(directory) / "partial").mkdir(parents=True, exist_ok=True)
all_cells = tuple((split, k) for split in baselines.SPLITS for k in baselines.KS)
paths_full = _build_fixture(Path(directory) / "full", cells=all_cells)
with self.assertRaises(ValueError):
baselines.evaluate_baselines(*paths_partial)
partial = baselines.evaluate_baselines(*paths_partial, allow_subset=True)
self.assertFalse(partial["matrix"]["complete"])
full = baselines.evaluate_baselines(*paths_full)
self.assertTrue(full["matrix"]["complete"])
self.assertEqual(len(full["runs"]), 12)
def test_output_refuses_existing_directory(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
paths[3].mkdir()
sentinel = paths[3] / "keep.txt"
sentinel.write_text("original")
with self.assertRaisesRegex(ValueError, "overwrite"):
baselines.evaluate_baselines(*paths, allow_subset=True)
self.assertEqual(sentinel.read_text(), "original")
def test_hash_all_source_inputs_present_in_summary(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
summary = baselines.evaluate_baselines(*paths, allow_subset=True)
for name, digest in summary["source_sha256"].items():
self.assertTrue(_hex64(digest), f"source {name} not hashed")
for name in ("rows", "run_index", "retrieval_source_recorded"):
self.assertTrue(
_hex64(summary["input_sha256"][name]),
f"input {name} not hashed",
)
for run in summary["runs"]:
self.assertTrue(_hex64(run["prediction_sha256"]))
self.assertTrue(_hex64(run["manifest_sha256"]))
self.assertTrue(_hex64(run["sha256"]))
def test_selector_deltas_against_both_baselines(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
summary = baselines.evaluate_baselines(*paths, allow_subset=True)
run = summary["runs"][0]
self.assertIsNotNone(run["selector_minus_random_micro"])
self.assertIsNotNone(run["selector_minus_first_micro"])
self.assertAlmostEqual(
run["selector_minus_random_micro"],
(run["selector_accuracy_given_recall_micro"] or 0.0)
- (run["random_expected_accuracy_micro"] or 0.0),
places=10,
)
self.assertAlmostEqual(
run["selector_minus_first_micro"],
(run["selector_accuracy_given_recall_micro"] or 0.0)
- (run["first_position_accuracy_micro"] or 0.0),
places=10,
)
def test_compatible_primary_population_micro_macro_bootstrap(self):
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
index_path = root / "run-index.json"
predictions_dir = root / "predictions"
predictions_dir.mkdir(parents=True, exist_ok=True)
output_dir = root / "output"
frozen_summary_path = root / "frozen-summary.json"
gold = [
{
"id": "r1",
"task_id": "tA",
"action_id": "correct",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["a", "c"],
},
{
"id": "r2",
"task_id": "tA",
"action_id": "missing_sel",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["b"],
},
{
"id": "r3",
"task_id": "tB",
"action_id": "invalid_sel",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["e"],
},
{
"id": "r4",
"task_id": "tB",
"action_id": "correct_alt",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["e"],
},
]
_write_rows(rows_path, gold)
config = {
"split": "test_task",
"k": 5,
"provenance": {"backend": "fixture"},
"adapted": False,
}
config_hash = baselines._object_hash(config)
predictions = [
{
"id": "r1",
"task_id": "tA",
"split": "test_task",
"k": 5,
"generated_candidates": ["a", "b", "c", "d", "e"],
"selection": "a",
"status": "ok",
"reason": None,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
{
"id": "r2",
"task_id": "tA",
"split": "test_task",
"k": 5,
"generated_candidates": ["b", "a", "c", "d", "e"],
"selection": None,
"status": "abstain",
"reason": "skip",
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
{
"id": "r3",
"task_id": "tB",
"split": "test_task",
"k": 5,
"generated_candidates": ["e", "a", "b", "c", "d"],
"selection": "z_not_in_returned_set",
"status": "ok",
"reason": None,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
{
"id": "r4",
"task_id": "tB",
"split": "test_task",
"k": 5,
"generated_candidates": ["a", "b", "c", "d", "e"],
"selection": "e",
"status": "ok",
"reason": None,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
]
prediction_path = predictions_dir / "test_task-k5.jsonl"
_write_rows(prediction_path, predictions)
counts_raw = dict(Counter(row["status"] for row in predictions))
counts = {
"ok": counts_raw.get("ok", 0),
"error": counts_raw.get("error", 0),
"abstain": counts_raw.get("abstain", 0),
}
digest = baselines._file_hash(prediction_path)
manifest = {
"schema": "vons.mind2web-prediction/v1",
"config": config,
"config_sha256": config_hash,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"predictions_sha256": digest,
"rows": len(predictions),
"counts": counts,
"selection_reused_across_k": False,
}
_write(prediction_path.with_suffix(".manifest.json"), manifest)
run_index = {
"schema": baselines.RUN_INDEX_SCHEMA,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"runs": [
{
"split": "test_task",
"k": 5,
"rows": len(predictions),
"counts": counts,
"predictions_sha256": digest,
"config_sha256": config_hash,
}
],
}
_write(index_path, run_index)
_write_manual_frozen_summary(
frozen_summary_path,
rows_path,
index_path,
predictions_dir,
(("test_task", 5),),
gold,
"b" * 64,
"a" * 64,
)
baselines.evaluate_baselines(
rows_path,
index_path,
predictions_dir,
output_dir,
frozen_summary_path,
allow_subset=True,
)
cell = json.loads((output_dir / "test_task-k5-baselines.json").read_text())
metrics = cell["metrics"]
self.assertEqual(metrics["recalled_rows"], 4)
self.assertEqual(metrics["selector"]["correct"], 2)
self.assertEqual(metrics["selector"]["missing_selection_on_recalled"], 1)
self.assertEqual(metrics["selector"]["invalid_selection_on_recalled"], 1)
selector_micro = metrics["selector_accuracy_given_recall_micro"]
self.assertAlmostEqual(selector_micro, 2.0 / 4.0, places=10)
random_micro = metrics["random_expected_accuracy_micro"]
first_micro = metrics["first_position_accuracy_micro"]
self.assertAlmostEqual(
metrics["selector_minus_random_micro"], selector_micro - random_micro, places=10
)
self.assertAlmostEqual(
metrics["selector_minus_first_micro"], selector_micro - first_micro, places=10
)
complete_case = metrics["selector_complete_case_accuracy_micro"]
self.assertIsNotNone(complete_case)
self.assertGreaterEqual(complete_case, selector_micro)
task_macro = metrics["selector_accuracy_given_recall_task_macro"]
rand_task = metrics["random_expected_accuracy_task_macro"]
first_task = metrics["first_position_accuracy_task_macro"]
self.assertAlmostEqual(
metrics["selector_minus_random_task_macro"], task_macro - rand_task, places=10
)
self.assertAlmostEqual(
metrics["selector_minus_first_task_macro"], task_macro - first_task, places=10
)
self.assertIn("primary_denominator_convention", metrics)
self.assertIn("recalled_rows", metrics["primary_denominator_convention"])
def test_generated_candidates_outside_gold_universe_rejected(self):
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
index_path = root / "run-index.json"
predictions_dir = root / "predictions"
predictions_dir.mkdir(parents=True, exist_ok=True)
output_dir = root / "output"
frozen_summary_path = root / "frozen-summary.json"
gold = [
{
"id": "r_bad",
"task_id": "tA",
"action_id": "x",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d"],
"positive_ids": ["a"],
}
]
_write_rows(rows_path, gold)
config = {
"split": "test_task",
"k": 5,
"provenance": {"backend": "fixture"},
"adapted": False,
}
config_hash = baselines._object_hash(config)
predictions = [
{
"id": "r_bad",
"task_id": "tA",
"split": "test_task",
"k": 5,
"generated_candidates": ["a", "Z_NOT_IN_GOLD", "c"],
"selection": "a",
"status": "ok",
"reason": None,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
}
]
prediction_path = predictions_dir / "test_task-k5.jsonl"
_write_rows(prediction_path, predictions)
counts_raw = dict(Counter(row["status"] for row in predictions))
counts = {
"ok": counts_raw.get("ok", 0),
"error": counts_raw.get("error", 0),
"abstain": counts_raw.get("abstain", 0),
}
digest = baselines._file_hash(prediction_path)
manifest = {
"schema": "vons.mind2web-prediction/v1",
"config": config,
"config_sha256": config_hash,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"predictions_sha256": digest,
"rows": len(predictions),
"counts": counts,
"selection_reused_across_k": False,
}
_write(prediction_path.with_suffix(".manifest.json"), manifest)
run_index = {
"schema": baselines.RUN_INDEX_SCHEMA,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"runs": [
{
"split": "test_task",
"k": 5,
"rows": len(predictions),
"counts": counts,
"predictions_sha256": digest,
"config_sha256": config_hash,
}
],
}
_write(index_path, run_index)
_write_manual_frozen_summary(
frozen_summary_path,
rows_path,
index_path,
predictions_dir,
(("test_task", 5),),
gold,
"b" * 64,
"a" * 64,
)
with self.assertRaisesRegex(ValueError, "outside gold candidate universe"):
baselines.evaluate_baselines(
rows_path,
index_path,
predictions_dir,
output_dir,
frozen_summary_path,
allow_subset=True,
)
def test_rows_and_run_index_digests_present_in_per_cell_reports(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=(("test_task", 5),),
include_failures=True,
)
summary = baselines.evaluate_baselines(*paths, allow_subset=True)
cell_path = paths[3] / "test_task-k5-baselines.json"
cell = json.loads(cell_path.read_text())
self.assertIn("rows", cell["input_sha256"])
self.assertIn("run_index", cell["input_sha256"])
self.assertTrue(_hex64(cell["input_sha256"]["rows"]))
self.assertTrue(_hex64(cell["input_sha256"]["run_index"]))
run = summary["runs"][0]
self.assertIn("rows_input_sha256", run)
self.assertIn("run_index_input_sha256", run)
self.assertTrue(_hex64(run["rows_input_sha256"]))
self.assertTrue(_hex64(run["run_index_input_sha256"]))
self.assertEqual(run["rows_input_sha256"], cell["input_sha256"]["rows"])
self.assertEqual(run["run_index_input_sha256"], cell["input_sha256"]["run_index"])
def test_run_index_manifest_file_mismatch_rejected(self):
cells = (("test_task", 5),)
def base_fixture():
directory = TemporaryDirectory()
root = Path(directory.name)
paths = _build_fixture(root, cells=cells)
return directory, paths
def reload(paths):
index = json.loads(paths[1].read_text())
manifest_path = paths[2] / "test_task-k5.manifest.json"
manifest = json.loads(manifest_path.read_text())
pred_path = paths[2] / "test_task-k5.jsonl"
predictions = list(baselines._jsonl(pred_path))
return index, manifest, predictions, pred_path, manifest_path
def save(index, manifest, predictions, paths, pred_path, manifest_path):
_write_rows(pred_path, predictions)
counts = dict(Counter(row["status"] for row in predictions))
digest = baselines._file_hash(pred_path)
manifest["rows"] = len(predictions)
manifest["counts"] = counts
manifest["predictions_sha256"] = digest
_write(manifest_path, manifest)
index["runs"][0]["rows"] = len(predictions)
index["runs"][0]["counts"] = counts
index["runs"][0]["predictions_sha256"] = digest
index["runs"][0]["config_sha256"] = manifest["config_sha256"]
_write(paths[1], index)
directory, paths = base_fixture()
index, manifest, predictions, pred_path, manifest_path = reload(paths)
save(index, manifest, predictions, paths, pred_path, manifest_path)
index = json.loads(paths[1].read_text())
index["runs"][0]["predictions_sha256"] = "0" * 64
_write(paths[1], index)
with self.assertRaises(ValueError):
baselines.evaluate_baselines(*paths, allow_subset=True)
directory.cleanup()
directory, paths = base_fixture()
index, manifest, predictions, pred_path, manifest_path = reload(paths)
save(index, manifest, predictions, paths, pred_path, manifest_path)
index = json.loads(paths[1].read_text())
index["runs"][0]["rows"] = 9999
_write(paths[1], index)
with self.assertRaises(ValueError):
baselines.evaluate_baselines(*paths, allow_subset=True)
directory.cleanup()
directory, paths = base_fixture()
index, manifest, predictions, pred_path, manifest_path = reload(paths)
save(index, manifest, predictions, paths, pred_path, manifest_path)
index = json.loads(paths[1].read_text())
index["runs"][0]["counts"] = {"ok": 9999}
_write(paths[1], index)
with self.assertRaises(ValueError):
baselines.evaluate_baselines(*paths, allow_subset=True)
directory.cleanup()
directory, paths = base_fixture()
index, manifest, predictions, pred_path, manifest_path = reload(paths)
save(index, manifest, predictions, paths, pred_path, manifest_path)
index = json.loads(paths[1].read_text())
index["runs"][0]["config_sha256"] = "f" * 64
_write(paths[1], index)
with self.assertRaises(ValueError):
baselines.evaluate_baselines(*paths, allow_subset=True)
directory.cleanup()
def test_overflow_from_error_reason_accounted(self):
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
index_path = root / "run-index.json"
predictions_dir = root / "predictions"
predictions_dir.mkdir(parents=True, exist_ok=True)
output_dir = root / "output"
frozen_summary_path = root / "frozen-summary.json"
gold = [
{
"id": "ok_row",
"task_id": "t1",
"action_id": "ok",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["a"],
},
{
"id": "overflow_input",
"task_id": "t1",
"action_id": "ov_in",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["b"],
},
{
"id": "overflow_candidate_input",
"task_id": "t2",
"action_id": "ov_cand",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["c"],
},
{
"id": "plain_error",
"task_id": "t2",
"action_id": "plain_err",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f", "g", "h"],
"positive_ids": ["d"],
},
]
_write_rows(rows_path, gold)
k = 5
config = {
"split": "test_task",
"k": k,
"provenance": {"backend": "fixture"},
"adapted": False,
}
config_hash = baselines._object_hash(config)
predictions = [
{
"id": "ok_row",
"task_id": "t1",
"split": "test_task",
"k": k,
"generated_candidates": ["a", "b", "c", "d", "e"],
"selection": "a",
"status": "ok",
"reason": None,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
{
"id": "overflow_input",
"task_id": "t1",
"split": "test_task",
"k": k,
"generated_candidates": ["a", "b", "c", "d", "e"],
"selection": None,
"status": "error",
"reason": "input_overflow",
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
{
"id": "overflow_candidate_input",
"task_id": "t2",
"split": "test_task",
"k": k,
"generated_candidates": ["a", "b", "c", "d", "e"],
"selection": None,
"status": "error",
"reason": "candidate_input_overflow",
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
{
"id": "plain_error",
"task_id": "t2",
"split": "test_task",
"k": k,
"generated_candidates": ["a", "b", "c", "d", "e"],
"selection": None,
"status": "error",
"reason": "timeout",
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
]
for pred in predictions:
self.assertLessEqual(len(pred["generated_candidates"]), k)
prediction_path = predictions_dir / f"test_task-k{k}.jsonl"
_write_rows(prediction_path, predictions)
counts_raw = dict(Counter(row["status"] for row in predictions))
counts = {
"ok": counts_raw.get("ok", 0),
"error": counts_raw.get("error", 0),
"abstain": counts_raw.get("abstain", 0),
}
digest = baselines._file_hash(prediction_path)
manifest = {
"schema": "vons.mind2web-prediction/v1",
"config": config,
"config_sha256": config_hash,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"predictions_sha256": digest,
"rows": len(predictions),
"counts": counts,
"selection_reused_across_k": False,
}
_write(prediction_path.with_suffix(".manifest.json"), manifest)
run_index = {
"schema": baselines.RUN_INDEX_SCHEMA,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"runs": [
{
"split": "test_task",
"k": k,
"rows": len(predictions),
"counts": counts,
"predictions_sha256": digest,
"config_sha256": config_hash,
}
],
}
_write(index_path, run_index)
_write_manual_frozen_summary(
frozen_summary_path,
rows_path,
index_path,
predictions_dir,
(("test_task", 5),),
gold,
"b" * 64,
"a" * 64,
)
summary = baselines.evaluate_baselines(
rows_path,
index_path,
predictions_dir,
output_dir,
frozen_summary_path,
allow_subset=True,
)
run = summary["runs"][0]
self.assertEqual(run["overflow_rows"], 2)
self.assertEqual(run["overflow_reason_counts"]["input_overflow"], 1)
self.assertEqual(run["overflow_reason_counts"]["candidate_input_overflow"], 1)
self.assertEqual(run["overflow_reason_counts"]["returned_count_exceeds_k"], 0)
cell = json.loads((output_dir / "test_task-k5-baselines.json").read_text())
self.assertEqual(cell["metrics"]["execution_status_counts"]["error"], 3)
self.assertEqual(
cell["metrics"]["rows_total"],
cell["metrics"]["prediction_records"]
+ cell["metrics"]["execution_status_counts"]["missing_record"],
)
self.assertIn("primary_denominator_convention", run)
def test_frozen_summary_valid_accepts_and_propagates_digests(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
summary = baselines.evaluate_baselines(*paths, allow_subset=True)
self.assertIn("frozen_summary", summary["input_sha256"])
self.assertTrue(_hex64(summary["input_sha256"]["frozen_summary"]))
expected_frozen_digest = baselines._file_hash(paths[4])
self.assertEqual(summary["input_sha256"]["frozen_summary"], expected_frozen_digest)
for run in summary["runs"]:
self.assertIn("frozen_summary_input_sha256", run)
self.assertTrue(_hex64(run["frozen_summary_input_sha256"]))
self.assertEqual(run["frozen_summary_input_sha256"], expected_frozen_digest)
cell_path = paths[3] / "test_task-k5-baselines.json"
cell = json.loads(cell_path.read_text())
self.assertIn("frozen_summary", cell["input_sha256"])
self.assertEqual(cell["input_sha256"]["frozen_summary"], expected_frozen_digest)
def test_frozen_summary_wrong_top_level_rows_hash_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["rows"] = "0" * 64
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "rows input_sha256 disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_retrieval_source_recorded_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["retrieval_source_recorded"] = "c" * 64
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "retrieval_source_recorded disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_runner_sha256_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["prediction_runner_sha256"] = "c" * 64
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "prediction_runner_sha256 disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_missing_requested_cell_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5), ("test_website", 10)))
frozen = json.loads(paths[4].read_text())
frozen["matrix"]["included_cells"] = [{"split": "test_task", "k": 5}]
frozen["runs"] = [
r for r in frozen["runs"] if not (r["split"] == "test_website" and r["k"] == 10)
]
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "missing runs for requested cells"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_duplicate_cell_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
run0 = dict(frozen["runs"][0])
frozen["runs"].append(run0)
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "duplicate split/k runs"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_cell_predictions_sha256_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["predictions_sha256"] = "f" * 64
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "predictions_sha256 disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_cell_manifest_sha256_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["prediction_manifest_sha256"] = "f" * 64
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "prediction_manifest_sha256 disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_cell_counts_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
ok_key = (
"ok"
if "ok" in frozen["runs"][0]["counts"]
else next(iter(frozen["runs"][0]["counts"]))
)
frozen["runs"][0]["counts"][ok_key] = 9999
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "status counts sum to"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_cell_rows_total_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["metrics"]["rows_total"] = 9999
_write(paths[4], frozen)
with self.assertRaisesRegex(
ValueError, "rows_total.*disagrees with expected aggregate"
):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_schema_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["schema"] = "vons.wrong-schema/v99"
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "frozen summary schema must be"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_cell_prediction_rows_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["prediction_rows"] = 9999
if "metrics" in frozen["runs"][0]:
if "prediction_rows" in frozen["runs"][0]["metrics"]:
frozen["runs"][0]["metrics"]["prediction_rows"] = 9999
frozen["runs"][0]["metrics"].pop("rows", None)
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "prediction_rows.*disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_summary_wrong_cell_config_sha256_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["config_sha256"] = "e" * 64
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "config_sha256 disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_sparse_all_ok_actual_counts_zero_fill_and_preserve_match(self):
norm = baselines._normalize_status_counts(
{"ok": 7}, "sparse actuals", total_rows=7, zero_fill_absent=True
)
self.assertEqual(norm, {"ok": 7, "error": 0, "abstain": 0})
def test_declared_counts_strict_missing_status_rejected(self):
with self.assertRaisesRegex(ValueError, "omits required status class 'error'"):
baselines._normalize_status_counts(
{"ok": 5, "abstain": 0},
"declared strict",
total_rows=5,
)
def test_counts_unknown_class_rejected_regardless_of_zero_fill(self):
for zero_fill in (False, True):
with self.assertRaisesRegex(ValueError, "unknown status class 'bogus'"):
baselines._normalize_status_counts(
{"ok": 3, "bogus": 1},
"tst",
total_rows=4,
zero_fill_absent=zero_fill,
)
def test_counts_sum_mismatch_rejected(self):
with self.assertRaisesRegex(ValueError, "sum to 9, expected 5"):
baselines._normalize_status_counts(
{"ok": 5, "error": 3, "abstain": 1},
"tst",
total_rows=5,
)
def test_run_index_missing_schema_value_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
index = json.loads(paths[1].read_text())
del index["schema"]
_write(paths[1], index)
new_run_index_digest = baselines._file_hash(paths[1])
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["run_index"] = new_run_index_digest
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "run index schema must be"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_run_index_unknown_schema_value_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
index = json.loads(paths[1].read_text())
index["schema"] = "vons.wrong-index/v99"
_write(paths[1], index)
new_run_index_digest = baselines._file_hash(paths[1])
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["run_index"] = new_run_index_digest
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "run index schema must be"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_duplicate_expected_cells_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["matrix"]["expected_cells"].append({"split": "test_task", "k": 5})
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "expected_cells contains duplicate cell"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_matrix_complete_false_with_all_cells_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=tuple(baselines.CELLS),
)
frozen = json.loads(paths[4].read_text())
frozen["matrix"]["complete"] = False
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "matrix.complete is inconsistent"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_incomplete_without_allow_subset_flag_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["matrix"]["allow_subset"] = False
_write(paths[4], frozen)
still_false_frozen = json.loads(paths[4].read_text())
still_false_frozen["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], still_false_frozen)
with self.assertRaisesRegex(
ValueError,
"both --allow-subset and matrix.allow_subset=true are required",
):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_incomplete_cli_flag_false_and_frozen_true_requires_both(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=(("test_task", 5),),
)
frozen = json.loads(paths[4].read_text())
frozen["matrix"]["allow_subset"] = True
_write(paths[4], frozen)
fixed_digest = json.loads(paths[4].read_text())
fixed_digest["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed_digest)
with self.assertRaisesRegex(
ValueError,
"all 12 official split/k cells are required unless --allow-subset",
):
baselines.evaluate_baselines(*paths, allow_subset=False)
def test_frozen_matrix_allow_subset_missing_required_type_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
del frozen["matrix"]["allow_subset"]
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(TypeError, "matrix.allow_subset must be a boolean"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_complete_matrix_with_allow_subset_true_is_valid_from_producer_contract(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=tuple(baselines.CELLS),
)
frozen = json.loads(paths[4].read_text())
self.assertTrue(frozen["matrix"]["complete"])
frozen["matrix"]["allow_subset"] = True
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
summary = baselines.evaluate_baselines(*paths, allow_subset=False)
first_12 = sorted((run["split"], run["k"]) for run in summary["runs"])
expected_12 = sorted(set(baselines.CELLS))
self.assertEqual(first_12, expected_12)
self.assertTrue(frozen["matrix"]["complete"])
def test_k_plus_one_excess_candidate_cannot_earn_recall_or_accuracy(self):
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
index_path = root / "run-index.json"
predictions_dir = root / "predictions"
predictions_dir.mkdir(parents=True, exist_ok=True)
output_dir = root / "output"
frozen_summary_path = root / "frozen-summary.json"
gold = [
{
"id": "excess_only_positive",
"task_id": "t_excess",
"action_id": "excess",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c", "d", "e", "f"],
"positive_ids": ["f"],
}
]
_write_rows(rows_path, gold)
k = 5
config = {
"split": "test_task",
"k": k,
"provenance": {"backend": "fixture"},
"adapted": False,
}
config_hash = baselines._object_hash(config)
predictions = [
{
"id": "excess_only_positive",
"task_id": "t_excess",
"split": "test_task",
"k": k,
"generated_candidates": ["a", "b", "c", "d", "e", "f"],
"selection": "f",
"status": "ok",
"reason": None,
"config_sha256": config_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
}
]
prediction_path = predictions_dir / f"test_task-k{k}.jsonl"
_write_rows(prediction_path, predictions)
counts = {"ok": 1, "error": 0, "abstain": 0}
digest = baselines._file_hash(prediction_path)
manifest = {
"schema": "vons.mind2web-prediction/v1",
"config": config,
"config_sha256": config_hash,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"predictions_sha256": digest,
"rows": len(predictions),
"counts": counts,
"selection_reused_across_k": False,
}
_write(prediction_path.with_suffix(".manifest.json"), manifest)
run_index = {
"schema": baselines.RUN_INDEX_SCHEMA,
"source_sha256": "b" * 64,
"runner_sha256": "a" * 64,
"runs": [
{
"split": "test_task",
"k": k,
"rows": len(predictions),
"counts": counts,
"predictions_sha256": digest,
"config_sha256": config_hash,
}
],
}
_write(index_path, run_index)
_write_manual_frozen_summary(
frozen_summary_path,
rows_path,
index_path,
predictions_dir,
(("test_task", 5),),
gold,
"b" * 64,
"a" * 64,
)
baselines.evaluate_baselines(
rows_path,
index_path,
predictions_dir,
output_dir,
frozen_summary_path,
allow_subset=True,
)
cell = json.loads((output_dir / "test_task-k5-baselines.json").read_text())
metrics = cell["metrics"]
self.assertEqual(metrics["recalled_rows"], 0)
self.assertEqual(metrics["selector"]["correct"], 0)
self.assertGreater(metrics["overflow_rows"], 0)
self.assertGreater(metrics["overflow_reason_counts"]["returned_count_exceeds_k"], 0)
def test_empty_gold_candidate_ids_universe_accepted_skip(self):
pass
def test_gold_empty_candidate_ids_accepted_by_index_validator(self):
with TemporaryDirectory() as directory:
rows_path = Path(directory) / "rows.jsonl"
gold = [
{
"id": "r_empty_universe",
"task_id": "t1",
"action_id": "x",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": [],
"positive_ids": ["only_target"],
}
]
_write_rows(rows_path, gold)
metadata, index = baselines._validate_and_index_gold(rows_path)
self.assertEqual(metadata["rows_total"], 1)
self.assertEqual(index["r_empty_universe"][1], ("only_target",))
def test_prediction_ids_outside_empty_universe_accepted_but_still_rejected_when_universe_populated(
self,
):
config_for_cell = {
"split": "test_task",
"k": 5,
"provenance": {"backend": "fixture"},
"adapted": False,
}
cfg_hash = baselines._object_hash(config_for_cell)
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
gold_empty = [
{
"id": "r_no_universe",
"task_id": "t1",
"action_id": "x",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": [],
"positive_ids": ["alpha"],
}
]
_write_rows(rows_path, gold_empty)
_, gold_idx = baselines._validate_and_index_gold(rows_path)
structural = baselines._validate_prediction_row(
{
"id": "r_no_universe",
"split": "test_task",
"k": 5,
"generated_candidates": ["alpha", "beta", "gamma"],
"selection": "alpha",
"status": "ok",
"config_sha256": cfg_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
split="test_task",
k=5,
gold_index=gold_idx,
config_hash=cfg_hash,
)
self.assertTrue(structural["recalled"])
self.assertTrue(structural["selector_correct"])
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
gold_populated = [
{
"id": "r_with_universe",
"task_id": "t1",
"action_id": "x",
"split": "test_task",
"website": "w",
"domain": "d",
"candidate_ids": ["a", "b", "c"],
"positive_ids": ["a"],
}
]
_write_rows(rows_path, gold_populated)
_, gold_idx = baselines._validate_and_index_gold(rows_path)
with self.assertRaisesRegex(ValueError, "outside gold candidate universe"):
baselines._validate_prediction_row(
{
"id": "r_with_universe",
"split": "test_task",
"k": 5,
"generated_candidates": ["a", "Z_NOT_IN_GOLD"],
"selection": "a",
"status": "ok",
"config_sha256": cfg_hash,
"request_sha256": "c" * 64,
"retrieval_input_sha256": "d" * 64,
},
split="test_task",
k=5,
gold_index=gold_idx,
config_hash=cfg_hash,
)
def test_frozen_conflicting_prediction_rows_top_vs_metrics_rows_alias_fail_closed(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["prediction_rows"] = 4
if "metrics" not in frozen["runs"][0]:
frozen["runs"][0]["metrics"] = {}
frozen["runs"][0]["metrics"]["rows"] = 9999
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(
ValueError,
"'prediction_rows'.*conflicting declared values",
):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_conflicting_prediction_rows_top_vs_nested_metrics_prediction_rows_fail_closed(
self,
):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["prediction_rows"] = 4
frozen["runs"][0]["metrics"]["prediction_rows"] = 9999
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(
ValueError,
"'prediction_rows'.*conflicting declared values",
):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_complete_case_task_macro_empty_when_no_evaluated_groups(self):
_m = baselines.CellAccumulator("test_task", 5)
_m.feed_row(
row_index=0,
task_key="t_single",
no_positive=False,
positive_ids_in_candidates=1,
candidate_count=2,
has_prediction_record=True,
execution_status="ok",
recalled=True,
selector_present=False,
selector_in_candidates=False,
selector_correct=False,
first_candidate_correct=True,
overflow=False,
)
reduced = _m.reduce()
self.assertEqual(reduced["selector"]["evaluated_recalled"], 0)
self.assertIsNone(reduced["selector_complete_case_accuracy_task_macro"])
def test_frozen_sparse_counts_ok_only_zero_fills_error_and_abstain(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["counts"] = {"ok": 4}
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_counts_unknown_class_strictly_rejected_despite_zero_fill_policy(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["counts"] = {"ok": 4, "bogus": 0}
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "unknown status class 'bogus'"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_counts_nonint_value_rejected_despite_zero_fill_policy(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["counts"] = {"ok": "four"}
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "must be a nonnegative integer"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_counts_sum_mismatch_strictly_rejected_despite_zero_fill_policy(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
frozen["runs"][0]["counts"] = {"ok": 9999}
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "status counts sum to"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_gold_positive_row_without_candidate_ids_key_accepted_as_empty_universe(self):
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
gold = [
{
"id": "r_no_candidate_ids_field",
"task_id": "t1",
"action_id": "x",
"split": "test_task",
"website": "w",
"domain": "d",
"positive_ids": ["A"],
}
]
_write_rows(rows_path, gold)
metadata, index = baselines._validate_and_index_gold(rows_path)
self.assertEqual(metadata["rows_total"], 1)
no_positive, positives, universe, _task = index["r_no_candidate_ids_field"]
self.assertFalse(no_positive)
self.assertEqual(positives, ("A",))
self.assertEqual(universe, ())
def test_gold_positive_row_with_tuple_universe_not_accepted_list_type_guard(self):
with TemporaryDirectory() as directory:
root = Path(directory)
rows_path = root / "rows.jsonl"
rows_path.write_text(
'{"id":"r_raw_tuple_bypass","task_id":"t2","action_id":"x",'
'"split":"test_task","website":"w","domain":"d",'
'"candidate_ids":["not-a-list","A"],"positive_ids":["A"]}\n'.replace(
'"candidate_ids":["not-a-list","A"]', '"candidate_ids":("not-a-list","A")'
)
if False
else '{"id":"r_raw_tuple_bypass","task_id":"t2","action_id":"x",'
'"split":"test_task","website":"w","domain":"d",'
'"candidate_ids":3,"positive_ids":["A"]}\n',
encoding="utf-8",
)
with self.assertRaisesRegex(TypeError, "gold candidate_ids must be a list"):
baselines._validate_and_index_gold(rows_path)
def test_frozen_metrics_helper_maps_metrics_rows_to_prediction_rows_alias(self):
run_obj = {
"split": "test_task",
"k": 5,
"metrics": {
"rows": 10,
"rows_total": 20,
"positive_rows": 15,
"recalled_positive_rows": 8,
},
}
merged = baselines._frozen_metrics(run_obj)
self.assertEqual(merged["prediction_rows"], 10)
self.assertEqual(merged["rows_total"], 20)
self.assertEqual(merged["positive_rows"], 15)
self.assertEqual(merged["recalled_positive_rows"], 8)
def test_coerce_denominator_int_fail_closed_bool_rejected(self):
with self.assertRaisesRegex(TypeError, "must be a nonnegative integer"):
baselines._coerce_denominator_int(True, "rows_total")
def test_coerce_denominator_int_fail_closed_string_rejected(self):
with self.assertRaisesRegex(TypeError, "must be a nonnegative integer"):
baselines._coerce_denominator_int("4", "rows_total")
def test_coerce_denominator_int_fail_closed_non_integer_float_rejected(self):
with self.assertRaisesRegex(TypeError, "must be a nonnegative integer"):
baselines._coerce_denominator_int(3.5, "rows_total")
def test_coerce_denominator_int_fail_closed_negative_rejected(self):
with self.assertRaisesRegex(ValueError, "must be a nonnegative integer"):
baselines._coerce_denominator_int(-1, "rows_total")
def test_coerce_denominator_int_integer_valued_float_accepted(self):
self.assertEqual(baselines._coerce_denominator_int(4.0, "rows_total"), 4)
self.assertEqual(baselines._coerce_denominator_int(0, "rows_total"), 0)
def test_frozen_metrics_rows_alias_reconciles_to_prediction_rows(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
del frozen["runs"][0]["prediction_rows"]
if "metrics" not in frozen["runs"][0]:
frozen["runs"][0]["metrics"] = {}
frozen["runs"][0]["metrics"]["rows"] = 4
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_metrics_rows_alias_wrong_value_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
del frozen["runs"][0]["prediction_rows"]
if "metrics" not in frozen["runs"][0]:
frozen["runs"][0]["metrics"] = {}
frozen["runs"][0]["metrics"]["rows"] = 9999
frozen["runs"][0]["metrics"].pop("prediction_rows", None)
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "prediction_rows.*disagrees"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_denominator_nonneg_int_string_fail_closed_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
if "metrics" not in frozen["runs"][0]:
frozen["runs"][0]["metrics"] = {}
frozen["runs"][0]["metrics"]["rows_total"] = "four"
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "must be a nonnegative integer"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_denominator_negative_value_fail_closed_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
if "metrics" not in frozen["runs"][0]:
frozen["runs"][0]["metrics"] = {}
frozen["runs"][0]["metrics"]["positive_rows"] = -1
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "must be a nonnegative integer"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_denominator_bool_value_fail_closed_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
if "metrics" not in frozen["runs"][0]:
frozen["runs"][0]["metrics"] = {}
frozen["runs"][0]["metrics"]["recalled_positive_rows"] = True
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "must be a nonnegative integer"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_frozen_denominator_non_integer_float_fail_closed_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
frozen = json.loads(paths[4].read_text())
if "metrics" not in frozen["runs"][0]:
frozen["runs"][0]["metrics"] = {}
frozen["runs"][0]["metrics"]["rows_total"] = 3.5
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "must be a nonnegative integer"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_producer_run_index_without_config_sha256_accepted_evaluate_passes(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=tuple(baselines.CELLS),
)
index = json.loads(paths[1].read_text())
for run in index["runs"]:
run.pop("config_sha256", None)
_write(paths[1], index)
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["run_index"] = baselines._file_hash(paths[1])
_write(paths[4], frozen)
summary = baselines.evaluate_baselines(*paths, allow_subset=False)
output_cells = sorted((r["split"], r["k"]) for r in summary["runs"])
self.assertEqual(output_cells, sorted(set(baselines.CELLS)))
for run in json.loads(paths[1].read_text())["runs"]:
self.assertNotIn("config_sha256", run)
def test_producer_run_index_optional_config_sha256_present_but_invalid_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
index = json.loads(paths[1].read_text())
index["runs"][0]["config_sha256"] = "short-not-hex64"
_write(paths[1], index)
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["run_index"] = baselines._file_hash(paths[1])
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "run index config_sha256 must be a"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_producer_run_index_optional_config_sha256_mismatch_manifest_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
index = json.loads(paths[1].read_text())
index["runs"][0]["config_sha256"] = "f" * 64
_write(paths[1], index)
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["run_index"] = baselines._file_hash(paths[1])
_write(paths[4], frozen)
with self.assertRaisesRegex(
ValueError, "run index config_sha256 disagrees with manifest"
):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_producer_sparse_run_index_counts_zero_fill_matches_actuals(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=tuple(baselines.CELLS),
)
index = json.loads(paths[1].read_text())
for run in index["runs"]:
counts = run["counts"]
run["counts"] = {"ok": counts.get("ok", 0)}
_write(paths[1], index)
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["run_index"] = baselines._file_hash(paths[1])
_write(paths[4], frozen)
summary = baselines.evaluate_baselines(*paths, allow_subset=False)
self.assertEqual(len(summary["runs"]), len(baselines.CELLS))
def test_producer_sparse_manifest_counts_zero_fill_matches_actuals(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(
Path(directory),
cells=tuple(baselines.CELLS),
)
hashes_to_update = {}
for split, k in baselines.CELLS:
manifest_path = paths[2] / f"{split}-k{k}.manifest.json"
manifest = json.loads(manifest_path.read_text())
counts = manifest["counts"]
manifest["counts"] = {"ok": counts.get("ok", 0)}
_write(manifest_path, manifest)
hashes_to_update[(split, k)] = (
baselines._file_hash(manifest_path),
manifest["predictions_sha256"],
)
frozen = json.loads(paths[4].read_text())
for run in frozen["runs"]:
cell = (run["split"], run["k"])
if cell in hashes_to_update:
run["prediction_manifest_sha256"] = hashes_to_update[cell][0]
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
summary = baselines.evaluate_baselines(*paths, allow_subset=False)
self.assertEqual(len(summary["runs"]), len(baselines.CELLS))
def test_producer_sparse_counts_unknown_class_strictly_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
index = json.loads(paths[1].read_text())
raw_counts = index["runs"][0]["counts"]
ok_val = raw_counts.get("ok", 0)
err_val = raw_counts.get("error", 0)
abs_val = raw_counts.get("abstain", 0)
index["runs"][0]["counts"] = {
"ok": ok_val,
"error": err_val,
"abstain": abs_val,
"bogus": 0,
}
_write(paths[1], index)
frozen = json.loads(paths[4].read_text())
frozen["input_sha256"]["run_index"] = baselines._file_hash(paths[1])
_write(paths[4], frozen)
with self.assertRaisesRegex(ValueError, "unknown status class 'bogus'"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_producer_sparse_counts_sum_mismatch_strictly_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
manifest_path = paths[2] / "test_task-k5.manifest.json"
manifest = json.loads(manifest_path.read_text())
manifest["counts"] = {"ok": manifest["counts"]["ok"] + 1}
_write(manifest_path, manifest)
frozen = json.loads(paths[4].read_text())
for run in frozen["runs"]:
if run["split"] == "test_task" and run["k"] == 5:
run["prediction_manifest_sha256"] = baselines._file_hash(manifest_path)
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "sum to .* expected"):
baselines.evaluate_baselines(*paths, allow_subset=True)
def test_producer_sparse_counts_negative_value_strictly_rejected(self):
with TemporaryDirectory() as directory:
paths = _build_fixture(Path(directory), cells=(("test_task", 5),))
manifest_path = paths[2] / "test_task-k5.manifest.json"
manifest = json.loads(manifest_path.read_text())
manifest["counts"] = {"ok": -1}
_write(manifest_path, manifest)
frozen = json.loads(paths[4].read_text())
for run in frozen["runs"]:
if run["split"] == "test_task" and run["k"] == 5:
run["prediction_manifest_sha256"] = baselines._file_hash(manifest_path)
_write(paths[4], frozen)
fixed = json.loads(paths[4].read_text())
fixed["input_sha256"]["frozen_summary"] = baselines._file_hash(paths[4])
_write(paths[4], fixed)
with self.assertRaisesRegex(ValueError, "must be a nonnegative integer"):
baselines.evaluate_baselines(*paths, allow_subset=True)
if __name__ == "__main__":
unittest.main()