File size: 6,421 Bytes
c14ceee
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
"""harness/evals.py β€” the Analyst eval gate (OM-4, 2026-07-11).

Runs the golden set (evals/analyst_golden.yml β€” truths from the parity-proven semantic layer over
CLOSED windows) against the Analyst and scores DETERMINISTICALLY: expected number found in the
answer (tolerance-aware, handles $/commas/% forms), required substrings, required tools in the
trace, required artifacts. No LLM judge in v1 β€” the validate() culture applied to AI answers.

Gate rule (from the plan): run before any deploy that touches model/ or the Analyst; ship only on
pass-rate >= the bar (start 80%, tighten as the model/skills improve).

Usage:
  python -X utf8 -c "import ssl; ssl._create_default_https_context=ssl._create_unverified_context;
                     import harness.evals as E; E.run()"        # live (needs OPENROUTER_API_KEY)
  E.run(chat_fn=scripted)                                        # offline harness test
"""
import re
from pathlib import Path

import yaml

GOLDEN = Path(__file__).resolve().parents[1] / "evals" / "analyst_golden.yml"


def _numbers(text):
    """Every number in the answer, normalized: $1,234.56 -> 1234.56; '59%'/'59.0 %' -> 0.59 too."""
    out = []
    for m in re.finditer(r"\$?([\d,]+(?:\.\d+)?)\s*(%?)", text):
        try:
            v = float(m.group(1).replace(",", ""))
        except ValueError:
            continue
        out.append(v)
        if m.group(2):                       # percent form: also offer the fraction
            out.append(v / 100.0)
    return out


def _norm_ws(s):
    """Collapse ALL unicode whitespace to single spaces. Models emit U+202F (narrow no-break
    space) and friends inside names β€” 'Poppy\\u202fFlowers' failed a plain substring check on
    two different models (the 'top_customer flake', diagnosed 2026-07-12)."""
    return re.sub(r"\s+", " ", s)


def _check(item, result):
    exp = item.get("expect") or {}
    answer = _norm_ws(result.get("answer") or "")
    reasons = []
    if result.get("exhausted"):
        reasons.append("tool budget exhausted")
    if "value" in exp:
        want = float(exp["value"])
        tol = exp.get("abs", exp.get("rel", 0.001) * abs(want) or 0.01)
        nums = _numbers(answer)
        if exp.get("allow_pct_form"):
            nums += [n for n in list(nums)]
        if not any(abs(n - want) <= tol for n in nums):
            reasons.append(f"expected {want:,.2f} (Β±{tol:,.2f}) not in answer")
    for s in exp.get("contains", []):
        if _norm_ws(s).lower() not in answer.lower():
            reasons.append(f"missing substring {s!r}")
    # WRONG-ANSWER guard (2026-07-28). `contains` can only prove an answer said the right thing;
    # some failures are about saying a WRONG thing that a right answer never says β€” e.g. naming an
    # internal salesperson as an agent ([[invoice-line-agent-commission]]). Those need a negative
    # assertion or the eval passes on an answer that is confidently incorrect.
    for s in exp.get("not_contains", []):
        if _norm_ws(s).lower() in answer.lower():
            reasons.append(f"forbidden substring {s!r} present")
    used = [t["tool"] for t in result.get("tool_trace", [])]
    for t in exp.get("tools", []):
        if t not in used:
            reasons.append(f"tool {t!r} not used (trace: {used})")
    for t in exp.get("not_tools", []):      # over-refusal guard: these must NOT appear
        if t in used:
            reasons.append(f"tool {t!r} used but forbidden here (trace: {used})")
    if exp.get("artifact") == "chart":
        if not any("chart" in a for a in result.get("artifacts", [])):
            reasons.append("no chart artifact produced")
    return (not reasons), reasons


def run(chat_fn=None, model=None, only=None, bar=0.8, verbose=True):
    """Run the golden set. Returns {passed, total, pass_rate, gate_ok, results}."""
    import harness.analyst as A
    items = yaml.safe_load(GOLDEN.read_text(encoding="utf-8"))["items"]
    if only:
        items = [i for i in items if i["id"] in only]
    results, passed = [], 0
    import os as _os
    import time as _time
    _os.environ["ANALYST_PATIENT"] = "1"    # batch mode: wait out rate windows, don't fail fast
    infra_fails = 0
    aborted = False
    for item in items:
        _time.sleep(6)      # pace the free-tier per-minute windows (Groq/Cerebras)
        kw = {"chat_fn": chat_fn} if chat_fn else {}
        if model:
            kw["model"] = model
        try:
            res = A.ask(item["ask"], telemetry_kind="analyst_eval", **kw)
            ok, reasons = _check(item, res)
            infra_fails = 0
        except Exception as e:
            res, ok, reasons = {"answer": "", "tool_trace": []}, False, [f"RUN ERROR: {e}"]
            # Provider exhaustion is NOT a model failure: two in a row means the ladder is out
            # of quota β€” abort and report INCONCLUSIVE instead of burning the remaining items
            # into a misleading FAIL (learned 2026-07-16: 14 wasted items, 4/18 "22%").
            if "all LLM providers failed" in str(e):
                infra_fails += 1
                if infra_fails >= 2:
                    aborted = True
        passed += ok
        results.append({"id": item["id"], "ok": ok, "reasons": reasons,
                        "usage": res.get("usage"), "iterations": res.get("iterations")})
        if verbose:
            mark = "OK " if ok else "XX "
            print(f"  {mark}{item['id']:18s}" + ("" if ok else f" β€” {'; '.join(reasons)[:90]}"))
        if aborted:
            break
    if aborted:
        ran = [r for r in results if not str(r["reasons"])[:60].startswith("['RUN ERROR")]
        if verbose:
            print(f"\nEVAL ABORTED β€” LLM providers exhausted after {len(results)} item(s); "
                  f"{passed}/{len(ran) or 1} of the items that actually ran passed. "
                  "Gate is INCONCLUSIVE (not a model failure) β€” re-run when quota recovers.")
        return {"passed": passed, "total": len(items), "pass_rate": None, "gate_ok": False,
                "inconclusive": True, "results": results}
    rate = passed / len(items) if items else 0.0
    gate = rate >= bar
    if verbose:
        print(f"\nEVAL: {passed}/{len(items)} = {rate:.0%}  GATE({bar:.0%}): "
              f"{'PASS' if gate else 'FAIL'}")
    return {"passed": passed, "total": len(items), "pass_rate": rate, "gate_ok": gate,
            "results": results}