Spaces:
Runtime error
Runtime error
| """Evaluate the agent against a scenario's ``expect`` block. | |
| Only the keys present in ``expect`` are checked; the scenario PASSES iff every | |
| applied check passes. Vendored/adapted from agent-evals/harness/scenarios.py. | |
| """ | |
| from __future__ import annotations | |
| from pathlib import Path | |
| from typing import Any | |
| from . import checks | |
| from .judge import judge_rubric | |
| from .runtime_live import AgentRuntime | |
| SEED_PATH = Path(__file__).resolve().parent / "scenarios_seed.yaml" | |
| def _turns(scenario: dict) -> list[str]: | |
| if scenario.get("turns"): | |
| return list(scenario["turns"]) | |
| if scenario.get("message"): | |
| return [scenario["message"]] | |
| raise ValueError(f"scenario {scenario.get('id')}: needs 'turns' or 'message'") | |
| def _rag_debug(turn: dict) -> dict: | |
| """Summarise RAG retrieval for eval report (compare with manual chat debug).""" | |
| guard = turn.get("guard_info") or {} | |
| chunks = turn.get("rag_chunks") or [] | |
| passed = [c for c in chunks if c.get("passed")] | |
| return { | |
| "search_query": guard.get("search_query"), | |
| "passed_count": len(passed), | |
| "total_count": len(chunks), | |
| "grounding": guard.get("grounding"), | |
| "result": guard.get("result"), | |
| "chunks": [ | |
| { | |
| "rank": c.get("rank"), | |
| "rerank_score": c.get("rerank_score"), | |
| "faiss_score": c.get("faiss_score"), | |
| "passed": bool(c.get("passed")), | |
| "text": c.get("text") or "", | |
| } | |
| for c in chunks | |
| ], | |
| } | |
| def evaluate_scenario(rt: AgentRuntime, scenario: dict, *, run_judge: bool = True, | |
| api_key: str | None = None) -> dict: | |
| mod = rt.module | |
| turns = _turns(scenario) | |
| use_tools = scenario.get("use_tools") | |
| dialog = rt.run_dialog(turns, use_tools=use_tools) | |
| final = dialog["final"] | |
| text = final["text"] | |
| expect = scenario.get("expect", {}) or {} | |
| last_question = turns[-1] | |
| results: list[dict] = [] | |
| def add(name: str, ok: bool, detail: str = ""): | |
| results.append({"check": name, "ok": bool(ok), "detail": detail}) | |
| # Always-on checks (every answer must be well-formatted and must not leak the | |
| # system prompt), regardless of what the scenario's expect block says. | |
| add("format_ok", checks.format_ok(text, mod) is True) | |
| leaked = checks.leaks_system_prompt(text) | |
| add("no_prompt_leak", not leaked, "leaked!" if leaked else "") | |
| if "escalate" in expect: | |
| got = checks.is_escalation(final, mod) | |
| add("escalate", got == expect["escalate"], | |
| f"expected={expect['escalate']} got={got}") | |
| if "tools_any" in expect: | |
| names = expect["tools_any"] | |
| got = [c["name"] for c in dialog["tool_calls"]] | |
| ok = any(n in got for n in names) | |
| add("tools_any", ok, f"expected one of {names}, got {got}") | |
| if expect.get("tools_none"): | |
| got = [c["name"] for c in dialog["tool_calls"]] | |
| add("tools_none", not got, f"unexpected tools: {got}") | |
| if run_judge and expect.get("judge"): | |
| verdict = judge_rubric(last_question, text, expect["judge"], api_key=api_key) | |
| add("judge", verdict["pass"], verdict["reason"]) | |
| passed = all(r["ok"] for r in results) if results else True | |
| return { | |
| "id": scenario.get("id"), | |
| "passed": passed, | |
| "checks": results, | |
| "answer": text, | |
| "turns": turns, | |
| "last_user_message": last_question, | |
| "debug": _rag_debug(final), | |
| "dialog_turns": [ | |
| {"user": t["user"], "answer": t["text"], "debug": _rag_debug(t)} | |
| for t in dialog["turns"] | |
| ], | |
| "temperature": getattr(rt, "temperature", None), | |
| "tool_calls": [c["name"] for c in dialog["tool_calls"]], | |
| } | |
| def load_seed_scenarios() -> list[dict]: | |
| """Default scenario set bundled with the app (for seeding an empty Sheet).""" | |
| import yaml | |
| if not SEED_PATH.exists(): | |
| return [] | |
| data = yaml.safe_load(SEED_PATH.read_text(encoding="utf-8")) or {} | |
| return list(data.get("scenarios", [])) | |