syntheogenesis / tests /test_compounding.py
github-actions[bot]
Deploy e90fe57
0fea6ef
Raw
History Blame Contribute Delete
9.94 kB
"""Two ways the engine gets better with use, neither of which is a model.
The founder's ask was that "both engine and turing chat panel get smarter each
day". The tempting answer is to train something. The honest answer, given that
platform labs = 0 and there is no measured data to train on, is to make use
itself compound:
resolution cache the thousandth person to ask for TP53 gets an instant
answer BECAUSE nine hundred and ninety-nine asked first.
worked examples a tool chain N different users completed is evidence
about how this engine is actually driven, and it can go
straight back into the agent's prompt.
Neither can be quietly wrong. The cached value is byte-identical to what the
database returned, and the promoted chain is a count. That is the whole appeal
— a learned component that degrades silently is the failure this product is
positioned against.
"""
import datetime as _dt
import json
import pytest
from dee.core import resolution_cache as rc
from dee.core import worked_examples as we
from dee.core.aggregate import EFFECTIVE_DATE, AggregationGateError
@pytest.fixture(autouse=True)
def _clean():
rc.clear()
yield
rc.clear()
def ev(kind, seq, **kw):
return dict(kind=kind, seq=seq, at=1754000000.0 + seq, **kw)
def clean_run(chain, user="u1", status="done", extra=()):
"""A run that called `chain` in order and finished."""
events = []
for i, name in enumerate(chain):
events.append(ev("tool_call", i * 2 + 1, id=f"c{i}", name=name))
events.append(ev("tool_result", i * 2 + 2, id=f"c{i}", name=name, ok=True))
events.extend(extra)
return {"run_id": f"r{user}{'-'.join(chain)}", "user_id": user,
"status": status, "events": events}
# --------------------------------------------------------------------------- #
# resolution cache
# --------------------------------------------------------------------------- #
RECORD = {"ok": True, "kind": "refseq", "sequence": "ATGC" * 50,
"label": "NM_000546 · 200 nt", "source": "ncbi", "gene_symbol": ""}
def test_a_second_lookup_is_served_from_the_first():
assert rc.get("refseq", "NM_000546") is None
rc.put("refseq", "NM_000546", RECORD)
hit = rc.get("refseq", "NM_000546")
assert hit is not None
assert hit["sequence"] == RECORD["sequence"], "cached value must be identical"
assert hit["cached"] is True
def test_a_pasted_sequence_is_never_stored_in_a_shared_cache():
"""The standing rule is that a user's own sequence never leaves the Space.
A cross-user cache is very much leaving."""
pasted = {"ok": True, "kind": "sequence", "sequence": "ACGT" * 40}
assert rc.cacheable("sequence") is False
assert rc.put("sequence", "ACGT" * 40, pasted) is False
assert rc.get("sequence", "ACGT" * 40) is None
assert rc.stats()["stores"] == 0
def test_the_refusal_is_on_kind_not_on_a_list_of_approved_types():
"""A whitelist would mean each new public identifier type silently
bypasses the cache until someone remembers to add it."""
assert rc.cacheable("refseq") and rc.cacheable("uniprot")
assert rc.cacheable("some_new_database_added_next_year") is True
assert rc.cacheable("sequence") is False
def test_a_failure_is_not_cached():
"""Ensembl down for ten seconds must not become 'this gene does not
exist' for the next fortnight."""
assert rc.put("symbol", "TP53", {"ok": False, "error": "timeout"}, "human") is False
assert rc.get("symbol", "TP53", "human") is None
def test_organism_is_part_of_the_key():
"""TP53 exists in dozens of species and they are different sequences.
Serving human TP53 to someone who asked for zebrafish is worse than a
miss."""
rc.put("symbol", "TP53", RECORD, "human")
assert rc.get("symbol", "TP53", "human") is not None
assert rc.get("symbol", "TP53", "zebrafish") is None
def test_keys_are_case_and_whitespace_insensitive():
rc.put("symbol", "TP53", RECORD, "human")
assert rc.get("symbol", " tp53 ", "Human") is not None
def test_a_caller_mutating_the_result_cannot_poison_the_cache():
"""The hard-to-trace failure mode of any shared cache."""
rc.put("refseq", "NM_1", RECORD)
first = rc.get("refseq", "NM_1")
first["sequence"] = "TAMPERED"
assert rc.get("refseq", "NM_1")["sequence"] == RECORD["sequence"]
def test_entries_expire_so_a_reannotated_record_is_refetched(monkeypatch):
rc.put("refseq", "NM_1", RECORD)
assert rc.get("refseq", "NM_1") is not None
monkeypatch.setattr(rc, "TTL_SECONDS", -1)
assert rc.get("refseq", "NM_1") is None
def test_the_cache_is_bounded():
monkey = rc.MAX_ENTRIES
try:
rc.MAX_ENTRIES = 5
for i in range(20):
rc.put("refseq", f"NM_{i}", RECORD)
assert rc.stats()["entries"] <= 5
assert rc.stats()["evictions"] >= 15
finally:
rc.MAX_ENTRIES = monkey
def test_the_resolver_serves_repeats_without_going_back_out(monkeypatch):
"""End to end through resolve_target, which is where it has to work."""
from dee.core import resolve
calls = []
def fake(text, organism, kind, val):
calls.append(val)
return {"ok": True, "kind": "refseq", "sequence": "ATGC" * 30,
"label": "x", "source": "ncbi", "gene_symbol": ""}
monkeypatch.setattr(resolve, "_resolve_uncached", fake)
a = resolve.resolve_target("NM_000546")
b = resolve.resolve_target("NM_000546")
assert calls == ["NM_000546"], "the second lookup went back to the network"
assert a["sequence"] == b["sequence"]
assert b.get("cached") is True
def test_hit_rate_is_reported_because_it_is_the_whole_claim():
rc.put("refseq", "NM_1", RECORD)
rc.get("refseq", "NM_1")
rc.get("refseq", "NM_2")
assert rc.stats()["hit_rate"] == 0.5
# --------------------------------------------------------------------------- #
# worked examples
# --------------------------------------------------------------------------- #
BUILD = ("lookup_vector", "simulate_assembly", "check_synthesis")
def test_a_chain_completed_by_enough_users_is_promoted():
runs = [clean_run(BUILD, user=f"u{i}") for i in range(4)]
out = we.promote(runs, min_users=3)
assert out["examples"][0]["chain"] == list(BUILD)
assert out["examples"][0]["users"] == 4
def test_a_corrected_run_is_not_a_worked_example():
"""A run the user had to steer is a near-miss. Promoting it teaches the
agent the route that needed fixing."""
runs = [clean_run(BUILD, user=f"u{i}") for i in range(3)]
runs.append(clean_run(BUILD, user="u9", extra=[
ev("steer", 99, text="no, that's the wrong backbone")]))
out = we.promote(runs, min_users=1)
assert out["examples"][0]["users"] == 3, "the corrected run was counted"
def test_a_run_with_a_failed_tool_is_not_promoted():
bad = {"run_id": "b", "user_id": "u1", "status": "done", "events": [
ev("tool_call", 1, id="c1", name="fetch_sequence"),
ev("tool_result", 2, id="c1", name="fetch_sequence", ok=False, error="x"),
ev("tool_call", 3, id="c2", name="lookup_vector"),
ev("tool_result", 4, id="c2", name="lookup_vector", ok=True)]}
assert we.chain_of(bad) is None
def test_an_unfinished_run_is_not_promoted():
assert we.chain_of(clean_run(BUILD, status="awaiting_input")) is None
assert we.chain_of(clean_run(BUILD, status="error")) is None
def test_a_single_tool_call_is_not_a_path():
assert we.chain_of(clean_run(("fetch_sequence",))) is None
def test_consecutive_repeats_collapse():
"""Fetching three genes is the same PATH as fetching one. Keeping the
repetition fragments the counts across chains that mean the same thing."""
chain = we.chain_of(clean_run(
("fetch_sequence", "fetch_sequence", "fetch_sequence", "fold_structure")))
assert chain == ("fetch_sequence", "fold_structure")
def test_one_lab_s_unusual_workflow_is_not_published_to_everyone():
runs = [clean_run(BUILD, user=f"u{i}") for i in range(3)]
runs.append(clean_run(("design_crispr_guides", "check_prior_art"), user="solo"))
out = we.promote(runs, min_users=3)
chains = [tuple(e["chain"]) for e in out["examples"]]
assert BUILD in chains
assert ("design_crispr_guides", "check_prior_art") not in chains
def test_no_user_content_survives_promotion():
"""Same rule as the field report: a worked example carrying the question
that produced it is a cross-user transcript excerpt."""
secret = "ZZQXSECRETZZ"
runs = [clean_run(BUILD, user=f"u{i}", extra=[
ev("user", 90, text=f"engineer {secret}"),
ev("text", 91, text=f"done with {secret}")]) for i in range(4)]
out = we.promote(runs, min_users=1)
assert secret not in json.dumps(out)
assert secret not in we.as_prompt_section(out)
def test_promotion_obeys_the_same_date_gate():
runs = [clean_run(BUILD, user=f"u{i}") for i in range(4)]
with pytest.raises(AggregationGateError):
we.promote(runs, today=EFFECTIVE_DATE - _dt.timedelta(days=1))
assert we.promote(runs, today=EFFECTIVE_DATE)["ok"] is True
def test_the_prompt_section_is_empty_when_nothing_qualifies():
"""An empty 'PATHS THAT WORK' heading reads as the engine having no idea
what works — worse than saying nothing."""
assert we.as_prompt_section(we.promote([], min_users=3)) == ""
assert we.as_prompt_section({}) == ""
def test_the_prompt_section_frames_them_as_evidence_not_rules():
"""Presented as law, an observed chain becomes a cage: the agent stops
solving tasks that need a different route."""
runs = [clean_run(BUILD, user=f"u{i}") for i in range(4)]
text = we.as_prompt_section(we.promote(runs, min_users=3))
assert "lookup_vector -> simulate_assembly" in text
assert "not rules" in text and "deviate" in text
assert "4 users" in text