File size: 1,555 Bytes
76b78ee
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
"""eval_summary parses persisted per-probe lines into the honest scorecard."""

from pathlib import Path

from research.eval_summary import scored_ids, summarize

FIXTURE = """\
== eval ckpt/hybrid50m_v22_lora/best.pt ==
[p01] 0.00 | verdict: true | conf: HIGH
[p05] 1.00 | verdict: false | conf: HIGH
[p07] 1.00 | verdict: true | conf: HIGH
[p38] qual | verdict: false | conf: HIGH
[verdict-00] 1.00 | verdict: true | conf: HIGH
[discrepancy-06] 0.00 | verdict: false | conf: HIGH
"""


def test_summarize_exact(tmp_path):
    log = tmp_path / "battery.log"
    log.write_text(FIXTURE)
    out = summarize(str(log))
    # p01 expected refutes -> 0; p05 false -> 1; p07 true -> 1; p38 qualitative
    # verdict-00 true -> 1; discrepancy-06 false -> 1
    assert out["n"] == 5
    assert out["qualitative"] == 1
    assert out["accuracy"] == 0.8
    assert out["by_category"]["generic"] == 2 / 3
    assert out["by_category"]["discrepancy"] == 1.0


def test_summarize_dedupes_resume_sections(tmp_path):
    log = tmp_path / "battery.log"
    log.write_text(FIXTURE + "== eval ckpt/hybrid50m_v22_lora/best.pt ==\n"
                              "[p01] 0.00 | verdict: true | conf: HIGH\n")
    out = summarize(str(log))
    assert out["n"] == 5  # p01 counted once


def test_scored_ids_whole_file(tmp_path):
    log = tmp_path / "battery.log"
    log.write_text(FIXTURE)
    ids = scored_ids(str(log))
    assert "p01" in ids and "verdict-00" in ids and "discrepancy-06" in ids
    assert "resume" not in ids  # attempt headers / resume lines are ignored