Rifqi Hafizuddin
[NOTICKET] eval: repair readiness harness; commit intent/help/readiness baselines
08a0a4a | """Report-readiness eval runner. | |
| Feeds each golden case in `readiness_dataset.json` to the deterministic | |
| `is_report_ready` signal (`src/agents/report/readiness.py`) via injectable FAKE | |
| stores — no LLM, no DB — then scores both the boolean `ready` and the `missing` | |
| gaps. Prints a per-case detail table + aggregate summary and writes a timestamped | |
| JSON report under `results/` (never overwritten — one file per run, diffable). | |
| Two metrics matter: | |
| - FLOOR correctness (ready + missing exact) — should be ~100%; this is the | |
| regression guard as the criteria evolve. | |
| - ALIGNMENT GAP — cases the floor calls ready=true but whose analyses are NOT | |
| aligned to the problem statement (`aligned=false`). The floor can't see this; | |
| the gap count is the evidence for/against adding the deferred LLM-judge. | |
| Invoke as a module so `src` imports resolve: | |
| uv run python -m eval.readiness.run_eval | |
| uv run python -m eval.readiness.run_eval --limit 5 | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import asyncio | |
| import json | |
| import statistics | |
| import time | |
| from dataclasses import asdict, dataclass, field | |
| from datetime import UTC, datetime, timedelta | |
| from pathlib import Path | |
| from typing import Any | |
| from src.agents.gate import stub_analysis_state | |
| from src.agents.report.readiness import ( | |
| _MISSING_ANALYSIS, | |
| _MISSING_DELTA, | |
| is_report_ready, | |
| ) | |
| _HERE = Path(__file__).resolve().parent | |
| DATASET = _HERE / "readiness_dataset.json" | |
| RESULTS_DIR = _HERE / "results" | |
| GROUPS = ["floor", "delta", "edge", "alignment"] | |
| # Dataset short codes -> the exact `missing` strings is_report_ready emits. Imported | |
| # from the module so the dataset stays readable and survives wording changes. The | |
| # `problem` code was retired with the problem_validated gate (KM-652, 2026-06-24). | |
| _CODE_TO_MISSING = { | |
| "analysis": _MISSING_ANALYSIS, | |
| "delta": _MISSING_DELTA, | |
| } | |
| class _FakeTask: | |
| """Mirrors slow_path.schemas.TaskSummary (the bits is_report_ready reads).""" | |
| status: str # success | partial | failure | |
| tools_used: list[str] | |
| class _FakeOutput: | |
| """Mirrors tools.contracts.ToolOutput (the bits `_produced_rows` reads).""" | |
| tool: str | |
| kind: str # scalar | table | stats | ... | |
| rows: list[list[Any]] | None | |
| class _FakeResult: | |
| """Mirrors slow_path.schemas.TaskResult (the bits `_produced_rows` reads).""" | |
| status: str | |
| outputs: list[_FakeOutput] | |
| class _FakeRecord: | |
| findings: list[Any] | |
| created_at: datetime | |
| tasks_run: list[_FakeTask] | |
| # Added 2026-07-23: the floor extension (#34) reads row counts from | |
| # `results_snapshot`, not `tasks_run` — TaskSummary carries no row counts. | |
| results_snapshot: dict[str, _FakeResult] | |
| class _FakeReport: | |
| generated_at: datetime | |
| class _FakeStore: | |
| """Stand-in for the Postgres record/report store — returns canned rows.""" | |
| def __init__(self, rows: list[Any]) -> None: | |
| self._rows = rows | |
| async def list_for_analysis( | |
| self, _analysis_id: str, _user_id: str | None = None | |
| ) -> list[Any]: | |
| # `_user_id` is accepted because `report_floor` passes it positionally as of | |
| # the tenant-scoping change (#38, 2026-07-23); the report-store call site | |
| # still passes one arg, so it stays optional. The fake is unscoped by design | |
| # — tenant scoping is covered by tests/catalog/test_tenant_scoping.py. | |
| return self._rows | |
| class CaseResult: | |
| id: str | |
| group: str | |
| expected_ready: bool | |
| got_ready: bool | |
| expected_missing: list[str] | |
| got_missing: list[str] | |
| correct: bool | |
| aligned: bool | |
| gap: bool # floor said ready but analyses not aligned to the problem statement | |
| latency_ms: float | |
| def load_cases(path: Path) -> list[dict[str, Any]]: | |
| data = json.loads(path.read_text(encoding="utf-8")) | |
| return list(data["cases"]) | |
| def _build_tasks(analysis: str) -> list[_FakeTask]: | |
| """Realistic tasks_run: data-access always succeeds; the analyze_* task varies. | |
| analysis = 'success' (analyze_* succeeded) | 'failure' (analyze_* failed) | | |
| 'none' (no analyze task at all — only check/retrieve succeeded). | |
| """ | |
| tasks = [ | |
| _FakeTask(status="success", tools_used=["check_data"]), | |
| _FakeTask(status="success", tools_used=["retrieve_data"]), | |
| ] | |
| if analysis == "success": | |
| tasks.append(_FakeTask(status="success", tools_used=["analyze_aggregate"])) | |
| elif analysis == "failure": | |
| tasks.append(_FakeTask(status="failure", tools_used=["analyze_aggregate"])) | |
| return tasks | |
| def _build_results(rows: int) -> dict[str, _FakeResult]: | |
| """The `results_snapshot` half of a record — how many rows the retrieve returned. | |
| `_build_tasks` always emits a SUCCESSFUL `retrieve_data` task, but the floor | |
| extension (#34, 2026-07-23) asks a question `tasks_run` cannot answer: did that | |
| retrieval actually return rows? `rows=0` means it succeeded and came back empty, | |
| which still fails the floor. Defaulting to 0 keeps every pre-#34 case's expected | |
| value exactly as it was — only cases that opt in with `rows` exercise the | |
| extension. | |
| """ | |
| return { | |
| "t_retrieve": _FakeResult( | |
| status="success", | |
| outputs=[ | |
| _FakeOutput( | |
| tool="retrieve_data", | |
| kind="table", | |
| rows=[[i] for i in range(rows)], | |
| ) | |
| ], | |
| ) | |
| } | |
| def _build_records(specs: list[dict[str, Any]], now: datetime) -> list[_FakeRecord]: | |
| return [ | |
| _FakeRecord( | |
| findings=["f"] * int(spec.get("findings", 0)), | |
| created_at=now - timedelta(minutes=int(spec["age_min"])), | |
| tasks_run=_build_tasks(str(spec.get("analysis", "success"))), | |
| results_snapshot=_build_results(int(spec.get("rows", 0))), | |
| ) | |
| for spec in specs | |
| ] | |
| def _build_reports(specs: list[dict[str, Any]], now: datetime) -> list[_FakeReport]: | |
| return [ | |
| _FakeReport(generated_at=now - timedelta(minutes=int(spec["age_min"]))) | |
| for spec in specs | |
| ] | |
| async def run_case(case: dict[str, Any]) -> CaseResult: | |
| now = datetime.now(UTC) | |
| # The problem_validated gate was removed (KM-652); readiness no longer reads the goal, | |
| # so a bare stub state + report_id is all is_report_ready needs. | |
| state = stub_analysis_state() | |
| if case.get("report_id"): | |
| state = state.model_copy(update={"report_id": case["report_id"]}) | |
| record_store = _FakeStore(_build_records(case.get("records", []), now)) | |
| report_store = _FakeStore(_build_reports(case.get("reports", []), now)) | |
| expected_missing = sorted(_CODE_TO_MISSING[c] for c in case["expected_missing"]) | |
| start = time.perf_counter() | |
| rr = await is_report_ready( | |
| case["id"], state, record_store=record_store, report_store=report_store | |
| ) | |
| latency_ms = round((time.perf_counter() - start) * 1000, 1) | |
| got_missing = sorted(rr.missing) | |
| ready_ok = rr.ready == bool(case["expected_ready"]) | |
| missing_ok = got_missing == expected_missing | |
| return CaseResult( | |
| id=case["id"], | |
| group=case["group"], | |
| expected_ready=bool(case["expected_ready"]), | |
| got_ready=rr.ready, | |
| expected_missing=expected_missing, | |
| got_missing=got_missing, | |
| correct=ready_ok and missing_ok, | |
| aligned=bool(case["aligned"]), | |
| gap=rr.ready and not bool(case["aligned"]), | |
| latency_ms=latency_ms, | |
| ) | |
| def _group_accuracy(results: list[CaseResult]) -> dict[str, dict[str, Any]]: | |
| out: dict[str, dict[str, Any]] = {} | |
| for g in GROUPS: | |
| sub = [r for r in results if r.group == g] | |
| if not sub: | |
| continue | |
| passed = sum(r.correct for r in sub) | |
| out[g] = {"n": len(sub), "passed": passed, "accuracy": round(passed / len(sub), 3)} | |
| return out | |
| def summarize(results: list[CaseResult]) -> dict[str, Any]: | |
| n = len(results) | |
| passed = sum(r.correct for r in results) | |
| gaps = [r for r in results if r.gap] | |
| latencies = [r.latency_ms for r in results] | |
| return { | |
| "total": n, | |
| "passed": passed, | |
| "accuracy": round(passed / n, 3) if n else 0.0, | |
| "runtime_avg_ms": round(statistics.mean(latencies), 2) if latencies else 0, | |
| "alignment_gap": {"count": len(gaps), "ids": [r.id for r in gaps]}, | |
| "by_group": _group_accuracy(results), | |
| } | |
| def _fmt_bool(value: bool) -> str: | |
| return "T" if value else "F" | |
| def _truncate(text: str, width: int) -> str: | |
| return text if len(text) <= width else text[: width - 3] + "..." | |
| def format_table(results: list[CaseResult]) -> str: | |
| header = ( | |
| f"{'ID':<12} {'GROUP':<10} {'RDY e/g':<8} " | |
| f"{'MISSING (got)':<40} {'OK':<3} {'GAP':<4}" | |
| ) | |
| rule = "-" * len(header) | |
| lines = [rule, header, rule] | |
| for r in results: | |
| rdy = f"{_fmt_bool(r.expected_ready)}/{_fmt_bool(r.got_ready)}" | |
| missing = ", ".join(r.got_missing) or "-" | |
| ok = "ok" if r.correct else "X" | |
| gap = "GAP" if r.gap else "" | |
| lines.append( | |
| f"{r.id:<12} {r.group:<10} {rdy:<8} " | |
| f"{_truncate(missing, 40):<40} {ok:<3} {gap:<4}" | |
| ) | |
| lines.append(rule) | |
| return "\n".join(lines) | |
| def format_summary(summary: dict[str, Any], results: list[CaseResult]) -> str: | |
| lines = ["SUMMARY"] | |
| lines.append( | |
| f" Floor {summary['passed']}/{summary['total']} correct" | |
| f" ({summary['accuracy'] * 100:.1f}%) avg {summary['runtime_avg_ms']} ms" | |
| ) | |
| gap = summary["alignment_gap"] | |
| lines.append( | |
| f" Align gap {gap['count']} case(s) ready-but-misaligned" | |
| + (f" -> {', '.join(gap['ids'])}" if gap["ids"] else "") | |
| ) | |
| lines.append(" (floor can't catch these; this count is the LLM-judge justification)") | |
| lines.append("") | |
| lines.append(" By group") | |
| for g, m in summary["by_group"].items(): | |
| lines.append(f" {g:<12} {m['passed']}/{m['n']} {m['accuracy'] * 100:.0f}%") | |
| failures = [r for r in results if not r.correct] | |
| lines.append("") | |
| lines.append(f" FAILURES ({len(failures)})") | |
| for r in failures: | |
| lines.append( | |
| f" {r.id:<12} ready {_fmt_bool(r.expected_ready)}->{_fmt_bool(r.got_ready)}" | |
| f" missing {r.expected_missing} -> {r.got_missing}" | |
| ) | |
| return "\n".join(lines) | |
| def build_report( | |
| results: list[CaseResult], summary: dict[str, Any], meta: dict[str, Any] | |
| ) -> dict[str, Any]: | |
| run = {**meta, **{k: summary[k] for k in ("total", "passed", "accuracy", "runtime_avg_ms")}} | |
| return { | |
| "run": run, | |
| "alignment_gap": summary["alignment_gap"], | |
| "by_group": summary["by_group"], | |
| "cases": [asdict(r) for r in results], | |
| } | |
| class _Args: | |
| dataset: Path = DATASET | |
| limit: int = 0 | |
| no_table: bool = False | |
| extra: dict[str, Any] = field(default_factory=dict) | |
| async def main() -> None: | |
| parser = argparse.ArgumentParser(description="Report-readiness eval") | |
| parser.add_argument("--dataset", type=Path, default=DATASET) | |
| parser.add_argument("--limit", type=int, default=0, help="run first N cases only") | |
| parser.add_argument("--no-table", action="store_true", help="skip the detail table") | |
| args = parser.parse_args() | |
| cases = load_cases(args.dataset) | |
| if args.limit: | |
| cases = cases[: args.limit] | |
| started = datetime.now() | |
| print(f"Report-Readiness Eval -- {started:%Y-%m-%d %H:%M:%S}") | |
| print(f"dataset: {args.dataset.name} ({len(cases)} cases) target: is_report_ready") | |
| results = [await run_case(case) for case in cases] | |
| summary = summarize(results) | |
| if not args.no_table: | |
| print(format_table(results)) | |
| print(format_summary(summary, results)) | |
| meta = { | |
| "timestamp": started.isoformat(timespec="seconds"), | |
| "dataset": args.dataset.name, | |
| "target": "src/agents/report/readiness.is_report_ready", | |
| } | |
| report = build_report(results, summary, meta) | |
| RESULTS_DIR.mkdir(parents=True, exist_ok=True) | |
| out_path = RESULTS_DIR / f"readiness_result_{started:%Y-%m-%d_%H%M%S}.json" | |
| out_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8") | |
| print(f"\n-> saved: {out_path.relative_to(_HERE.parent.parent)}") | |
| if __name__ == "__main__": | |
| asyncio.run(main()) | |