shl-recommender / eval /report.py
Eshit's picture
HF Space deploy snapshot
5733f37
Raw
History Blame Contribute Delete
3.89 kB
"""Write the evaluation report (CLAUDE.md §6).
Aggregates the three measurements into ``eval/REPORT.md``:
1. Retrieval Recall@10 (deterministic, no LLM) — from ``eval.recall_eval``.
2. Conversational Recall@10 + groundedness — from ``eval.replay`` (multi-turn,
one router LLM call per turn).
3. Behavior probes — from ``eval.probes``.
python eval/report.py # scripted replay
python eval/report.py --llm-user # LLM-simulated user for the replay
LLM-dependent sections degrade gracefully: if calls fail (e.g. rate limit), the
affected traces/probes are flagged, and the deterministic retrieval section is
always populated.
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
from eval import probes as probes_mod # noqa: E402
from eval import recall_eval, replay # noqa: E402
REPORT_PATH = ROOT / "eval" / "REPORT.md"
def build_report(llm_user: bool = False) -> str:
# 1. Deterministic retrieval recall (assembly + router query).
retr = recall_eval.evaluate(k=10, use_router=False, assemble=True)
# 2. Conversational replay (multi-turn through the real agent).
conv = replay.run(mode="llm" if llm_user else "scripted")
# 3. Behavior probes.
probe_results = probes_mod.run_probes()
lines: list[str] = []
lines += ["# Evaluation Report", ""]
lines += [
"Generated by `python eval/report.py`. Three views: deterministic "
"retrieval recall, multi-turn conversational recall, and behavior probes.",
"",
"## Headline",
"",
f"- **Retrieval Recall@10 (deterministic):** {retr['mean_recall']:.4f}",
f"- **Conversational Recall@10 ({conv['mode']} user):** "
f"{conv['mean_recall']:.4f} "
f"(scored {conv['n_scored']}/{len(conv['rows'])}; "
f"{conv['n_errored']} LLM-errored)",
f"- **Groundedness:** {conv['groundedness']:.4f} (URLs in catalog)",
f"- **Probes:** {sum(p.passed for p in probe_results)}/{len(probe_results)} passed",
"",
]
lines += ["## Retrieval Recall@10 (per trace)", ""]
lines += ["| Trace | Recall | Hits/Exp | Missed |", "|--|--:|--:|--|"]
for row in retr["rows"]:
missed = ", ".join(row["missed"]) if row["missed"] else "—"
lines.append(f"| {row['id']} | {row['recall']:.2f} | "
f"{row['hits']}/{row['expected']} | {missed} |")
lines.append("")
lines += ["## Conversational Recall@10 (multi-turn replay)", ""]
lines += ["| Trace | Recall | Hits/Exp | Turns | Ended | Recs | Note |",
"|--|--:|--:|--:|--|--:|--|"]
for row in conv["rows"]:
note = ("LLM error (excluded)" if row["excluded"]
else "recovered after transient error" if row["llm_error"] else "")
lines.append(f"| {row['id']} | {row['recall']:.2f} | "
f"{row['hits']}/{row['expected']} | {row['turns']} | "
f"{row['ended']} | {row['n_recs']} | {note} |")
lines.append("")
lines += ["## Behavior probes", ""]
lines += ["| Probe | Result | Detail |", "|--|--|--|"]
for p in probe_results:
lines.append(f"| {p.name} | {'PASS' if p.passed else 'FAIL'} | {p.detail} |")
lines.append("")
lines += [
"See `eval/TUNING_LOG.md` for the before/after that produced these "
"numbers, and `APPROACH.md` for the method.",
"",
]
return "\n".join(lines)
def main() -> None:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--llm-user", action="store_true")
args = ap.parse_args()
report = build_report(llm_user=args.llm_user)
REPORT_PATH.write_text(report, encoding="utf-8")
print(report)
print(f"\nWrote {REPORT_PATH}")
if __name__ == "__main__":
main()