Spaces:
Running
Running
| """Every recorded failure, replayed, so none of them can happen twice. | |
| `dee/data/golden/*.json` holds real incidents — things Turing did to a real | |
| user that were wrong. The mining rollup (dee/core/learn_signal.py) is what | |
| finds them; this is what makes finding them once sufficient. Adding a newly | |
| observed failure is a JSON file, not a new test, which is the only version of | |
| this that survives contact with a busy week. | |
| The model is scripted, so these do not test tool CHOICE — that needs a live | |
| call and is non-deterministic. They test everything downstream of the choice, | |
| which is where the harm actually landed every time: a fabricated sequence | |
| reaching the user, a tool that should have run not running, the engine | |
| treating "described it" as "did it". | |
| Plus one static assertion that does move the live model: the descriptions of | |
| the two tools users confuse must actually disambiguate. The pCAMBIA mis-pick | |
| was fixed by what `fetch_sequence` says about vectors, not by anything at | |
| runtime. | |
| """ | |
| import json | |
| import time | |
| import pytest | |
| from dee.core import agent_tools as _tools | |
| from dee.core import golden | |
| from dee.core import llm as _llm | |
| from dee.core import orchestrator as orch | |
| SCENARIOS = golden.load_all() | |
| def _configured(monkeypatch): | |
| monkeypatch.setenv("OPENROUTER_API_KEY", "test-key") | |
| monkeypatch.setenv("AGENT_MAX_COST_USD", "0.5") | |
| def _run_to_rest(run, timeout=5.0): | |
| deadline = time.time() + timeout | |
| while time.time() < deadline: | |
| if run.status in ("done", "error", "awaiting_input", "stopped"): | |
| return run.status | |
| time.sleep(0.01) | |
| raise AssertionError(f"run never settled (status={run.status})") | |
| def _replay(scenario, monkeypatch): | |
| """Drive a real run with the scenario's scripted model and tool results.""" | |
| msgs = golden.to_messages(scenario) | |
| seq = list(msgs) | |
| def fake_call(config, body): | |
| # After the script runs out the model just stops talking, which is | |
| # what a real final turn looks like. | |
| return (seq.pop(0) if seq else {"content": "", "tool_calls": []}), 0.001, 100, 0 | |
| monkeypatch.setattr(_llm, "call", fake_call) | |
| called = [] | |
| results = scenario.get("tool_results") or {} | |
| def fake_exec(name, args, anon, **kw): | |
| called.append((name, args)) | |
| return results.get(name, {"ok": True}) | |
| monkeypatch.setattr(orch._tools, "execute_tool", fake_exec) | |
| run = orch.create_run(owner="golden-user", anonymous=False) | |
| if scenario.get("workspace"): | |
| run.workspace = scenario["workspace"] | |
| orch.start(run, scenario["user"]) | |
| _run_to_rest(run) | |
| # A gated tool (edit_sequence, log_outcome, blast_sequence) parks the run | |
| # on a confirm card instead of executing. That is correct and is itself | |
| # one of the protections — but a scenario about what happens AFTER the | |
| # user approves has to actually approve, or it silently tests the gate | |
| # and nothing else. | |
| if scenario.get("approve") and run.status == "awaiting_input": | |
| orch.reply(run, "yes") | |
| _run_to_rest(run) | |
| return run, [n for n, _a in called] | |
| def _final_text(run): | |
| return "\n".join(e.get("text") or "" for e in run.events if e["kind"] == "text") | |
| def _ids(scenarios): | |
| return [s["name"] for s in scenarios] | |
| # --------------------------------------------------------------------------- # | |
| # the corpus itself | |
| # --------------------------------------------------------------------------- # | |
| def test_there_are_scenarios_and_they_are_well_formed(): | |
| """A validation error is raised, not skipped: an unloadable fixture reads | |
| as coverage while asserting nothing.""" | |
| assert SCENARIOS, "no golden scenarios on disk" | |
| for s in SCENARIOS: | |
| golden.validate(s, source=s["name"]) | |
| assert s["incident"].strip(), f"{s['name']}: no incident recorded" | |
| def test_a_scenario_that_asserts_nothing_is_rejected(): | |
| """The failure mode this format is most prone to: a note pretending to be | |
| a test.""" | |
| with pytest.raises(golden.GoldenError, match="asserts nothing"): | |
| golden.validate({"name": "x", "incident": "i", "first_seen": "2026-01-01", | |
| "user": "u", "model_script": [{"text": "hi"}]}) | |
| def test_a_typo_in_an_expectation_is_an_error_not_a_silent_pass(): | |
| with pytest.raises(golden.GoldenError, match="unknown expect"): | |
| golden.validate({"name": "x", "incident": "i", "first_seen": "2026-01-01", | |
| "user": "u", "model_script": [{"text": "hi"}], | |
| "expect": {"must_kall": ["a"]}}) | |
| # --------------------------------------------------------------------------- # | |
| # replay | |
| # --------------------------------------------------------------------------- # | |
| def test_recorded_incident_does_not_recur(scenario, monkeypatch): | |
| run, called = _replay(scenario, monkeypatch) | |
| expect = scenario["expect"] | |
| text = _final_text(run) | |
| why = scenario["incident"][:120] | |
| for name in expect.get("must_call", []): | |
| assert name in called, f"{scenario['name']}: {name} never ran — {why}" | |
| for name in expect.get("must_not_call", []): | |
| assert name not in called, f"{scenario['name']}: {name} ran — {why}" | |
| for name in expect.get("must_confirm", []): | |
| confirms = [e for e in run.events if e["kind"] == "confirm"] | |
| assert any(name in json.dumps(e) for e in confirms), ( | |
| f"{scenario['name']}: {name} ran without stopping to ask — {why}") | |
| for needle in expect.get("final_text_contains", []): | |
| assert needle in text, f"{scenario['name']}: missing {needle!r} — {why}" | |
| for needle in expect.get("final_text_lacks", []): | |
| assert needle not in text, \ | |
| f"{scenario['name']}: leaked {needle!r} to the user — {why}" | |
| if expect.get("status"): | |
| assert run.status == expect["status"] | |
| if expect.get("no_unsourced_sequence"): | |
| from dee.core import provenance as prov | |
| pool = orch._sourced_pool(run) | |
| leaked = prov.unsourced(text, pool) | |
| assert not leaked, ( | |
| f"{scenario['name']}: {len(leaked)} sequence(s) reached the user " | |
| f"without coming from a tool result, the user, or the bench — {why}") | |
| def test_tool_descriptions_still_disambiguate(scenario): | |
| """Static, and the one assertion that actually steers the live model. | |
| Runtime guards stop the harm; the description is what stops the wrong | |
| call being made in the first place. | |
| """ | |
| wanted = (scenario["expect"].get("tool_descriptions_say") or {}) | |
| if not wanted: | |
| pytest.skip("scenario makes no description claim") | |
| specs = {s["function"]["name"]: s["function"]["description"] | |
| for s in orch.TOOL_SPECS} | |
| for tool, needles in wanted.items(): | |
| assert tool in specs, f"{tool} is not a registered tool" | |
| desc = specs[tool].lower() | |
| for needle in needles: | |
| assert needle.lower() in desc, ( | |
| f"{tool}'s description no longer mentions {needle!r}; " | |
| f"the {scenario['name']} incident becomes possible again") | |
| # --------------------------------------------------------------------------- # | |
| # the guard the scenarios lean on | |
| # --------------------------------------------------------------------------- # | |
| def test_the_fabrication_scenario_actually_exercises_the_guard(monkeypatch): | |
| """Guards against the worst outcome here: the scenario passing because | |
| the model was scripted to behave, rather than because the engine caught | |
| it. This one scripts a fabrication and asserts it was CAUGHT.""" | |
| scenario = next(s for s in SCENARIOS if s["name"] == "pcambia-fabricated-sequence") | |
| run, _called = _replay(scenario, monkeypatch) | |
| events = [e for e in run.events if e["kind"] == "provenance"] | |
| assert events, "the guard never fired — this scenario proves nothing" | |
| assert events[0]["withheld"], "fired but withheld nothing" | |
| def test_a_fabrication_sends_the_run_back_to_fetch_it_exactly_once(monkeypatch): | |
| """Ending on a redacted reply leaves the user holding a hole. So the run | |
| gets one more turn to go and look the sequence up — one, not a loop: if | |
| the model writes an unsourced sequence twice, the redaction stands and | |
| the run ends rather than burning steps.""" | |
| fake = "GGGGCCCCGGGGCCCCGGGGCCCCGGGGCCCCTTTTAAAA" | |
| scenario = { | |
| "name": "retry", "incident": "n/a", "first_seen": "2026-08-05", | |
| "user": "what is pUC19", | |
| "model_script": [{"text": f"It is {fake}"}, | |
| {"text": f"Actually it is {fake}"}], | |
| "expect": {"no_unsourced_sequence": True}, | |
| } | |
| run, _ = _replay(scenario, monkeypatch) | |
| assert run.provenance_retried is True | |
| # Fired on both replies, and the run still terminated. | |
| assert len([e for e in run.events if e["kind"] == "provenance"]) == 2 | |
| assert run.status == "done" | |
| assert fake not in _final_text(run) | |
| def test_a_sourced_sequence_is_left_alone(monkeypatch): | |
| """The other half. A guard that redacts real retrieved sequences would be | |
| turned off within a day, and then nothing is protected.""" | |
| real = ("ATGACCATGATTACGCCAAGCTTGCATGCCTGCAGGTCGACTCTAGAGG" | |
| "ATCCCCGGGTACCGAGCTCGAATTCACTGGC") | |
| scenario = { | |
| "name": "sourced", "incident": "n/a", "first_seen": "2026-08-05", | |
| "user": "get pUC19", | |
| "tool_results": {"lookup_vector": {"ok": True, "sequence": real}}, | |
| "model_script": [ | |
| {"tool": "lookup_vector", "args": {"name": "pUC19"}}, | |
| {"text": f"Here is the region you asked for: {real}"}, | |
| ], | |
| "expect": {"no_unsourced_sequence": True}, | |
| } | |
| run, _ = _replay(scenario, monkeypatch) | |
| text = _final_text(run) | |
| assert real in text, "a legitimately retrieved sequence was redacted" | |
| assert not [e for e in run.events if e["kind"] == "provenance"] | |