fsi-anomaly / tests /test_eval_summary.py
FerrellSyntheticIntelligence's picture
backup all: 100 files (batch)
76b78ee verified
Raw
History Blame Contribute Delete
1.56 kB
"""eval_summary parses persisted per-probe lines into the honest scorecard."""
from pathlib import Path
from research.eval_summary import scored_ids, summarize
FIXTURE = """\
== eval ckpt/hybrid50m_v22_lora/best.pt ==
[p01] 0.00 | verdict: true | conf: HIGH
[p05] 1.00 | verdict: false | conf: HIGH
[p07] 1.00 | verdict: true | conf: HIGH
[p38] qual | verdict: false | conf: HIGH
[verdict-00] 1.00 | verdict: true | conf: HIGH
[discrepancy-06] 0.00 | verdict: false | conf: HIGH
"""
def test_summarize_exact(tmp_path):
log = tmp_path / "battery.log"
log.write_text(FIXTURE)
out = summarize(str(log))
# p01 expected refutes -> 0; p05 false -> 1; p07 true -> 1; p38 qualitative
# verdict-00 true -> 1; discrepancy-06 false -> 1
assert out["n"] == 5
assert out["qualitative"] == 1
assert out["accuracy"] == 0.8
assert out["by_category"]["generic"] == 2 / 3
assert out["by_category"]["discrepancy"] == 1.0
def test_summarize_dedupes_resume_sections(tmp_path):
log = tmp_path / "battery.log"
log.write_text(FIXTURE + "== eval ckpt/hybrid50m_v22_lora/best.pt ==\n"
"[p01] 0.00 | verdict: true | conf: HIGH\n")
out = summarize(str(log))
assert out["n"] == 5 # p01 counted once
def test_scored_ids_whole_file(tmp_path):
log = tmp_path / "battery.log"
log.write_text(FIXTURE)
ids = scored_ids(str(log))
assert "p01" in ids and "verdict-00" in ids and "discrepancy-06" in ids
assert "resume" not in ids # attempt headers / resume lines are ignored