B2D-agentic-ai / tests /test_optimization.py
AMRYB's picture
Upload 91 files
287f3d3 verified
Raw
History Blame Contribute Delete
19.1 kB
"""Regression tests for the token/latency optimization work.
These lock in the guarantees the optimization preserved: compact schema
embedding, deterministic digest handoffs, decision-dense prompts, and honest
telemetry that never conflates estimated vs provider-reported usage.
"""
from __future__ import annotations
import json
import pytest
from agentic_core.agents import (
APIAgent,
ArchitectureAgent,
DatabaseAgent,
DevOpsAgent,
RequirementsAgent,
digest_requirements,
)
from agentic_core.llm import LLMService, StructuredOutputError
from agentic_core.llm.service import _strip_schema_titles, extract_json_object
from agentic_core.prompts import api, architecture, database, devops, discovery, requirements
from agentic_core.schemas import (
APIOutput,
ArchitectureOutput,
DatabaseOutput,
DevopsOutput,
DiscoveryOutput,
RequirementsOutput,
ReviewOutput,
)
from tests.helpers import (
api_output,
architecture_output,
database_output,
devops_output,
requirements_output,
build_handler,
detect_agent,
)
# ---------------------------------------------------------------- schema size
def test_embedded_schema_has_no_title_boilerplate():
"""The schema shown to the LLM has no Pydantic title boilerplate (pure
token overhead). It stays human-readable (indented) on purpose — compact
whitespace-free schemas measurably increased repair rates in real runs."""
spec = _strip_schema_titles(RequirementsOutput.model_json_schema())
serialized = json.dumps(spec, separators=(",", ":"))
assert "title" not in serialized
assert len(serialized) < 500 # previously ~940 chars
def test_embedded_schema_has_no_dangling_refs():
"""Every $ref in the embedded schema resolves to a definition that is still
present. Dropping $defs left the model guessing the shape of DBEntity,
SystemComponent, APIEndpoint and the severity/importance enums, which is the
class of failure recorded in the run logs as
`missing_information.N.importance - Input should be ...`."""
from agentic_core.llm.service import _referenced_defs
for schema in (
DiscoveryOutput, RequirementsOutput, ArchitectureOutput,
DatabaseOutput, APIOutput, DevopsOutput, ReviewOutput,
):
spec = LLMService._schema_for(schema)
referenced = _referenced_defs(spec, set())
defined = set(spec.get("$defs") or {})
assert referenced <= defined, f"{schema.__name__} references undefined {referenced - defined}"
def test_embedded_schema_carries_nested_enum_values():
"""The enums the pre-validators exist to repair must be visible up front."""
arch = json.dumps(LLMService._schema_for(ArchitectureOutput))
assert "infrastructure" in arch and "frontend" in arch
review = json.dumps(LLMService._schema_for(ReviewOutput))
assert "blocking" in review and "suggestion" in review
discovery = json.dumps(LLMService._schema_for(DiscoveryOutput))
assert "not_applicable" in discovery
def test_unreferenced_defs_are_pruned():
"""Excluding a derived field must not leave its definitions behind."""
spec = LLMService._schema_for(APIOutput)
assert "openapi_spec" not in json.dumps(spec)
assert set(spec.get("$defs") or {}) == {"APIEndpoint"}
def test_schema_excludes_derived_fields():
from agentic_core.llm.service import LLMService
class _Fake:
pass
api_spec = LLMService._schema_for(APIAgent.output_schema)
assert "openapi_spec" not in json.dumps(api_spec)
db_spec = LLMService._schema_for(DatabaseAgent.output_schema)
assert "sql_schema" not in json.dumps(db_spec)
assert "erd_mermaid" not in json.dumps(db_spec)
async def test_schema_chars_telemetry_recorded(provider, llm_service, make_context):
agent = RequirementsAgent(llm_service)
provider.set_responses([json.dumps(requirements_output())])
result = await agent.run(make_context("Food delivery."))
assert result.status == "success"
assert result.schema_chars > 0
assert result.input_chars > result.schema_chars
assert result.schema_chars == len(
json.dumps(
_strip_schema_titles(agent.output_schema.model_json_schema()),
indent=2,
)
)
async def test_schema_chars_persisted_to_tracker(
provider, llm_service, make_context, tracker, settings
):
from agentic_core.orchestrator import Orchestrator
agent = RequirementsAgent(llm_service, tracker)
provider.set_responses([json.dumps(requirements_output())])
await agent.run(make_context("Food delivery."))
records = tracker.list("test_proj")
assert records
assert all(r.schema_chars > 0 for r in records if r.status == "success")
# ---------------------------------------------------------------- prompt density
def test_discovery_prompt_instructs_early_stop_and_no_dupes():
text = discovery.SYSTEM_PROMPT
assert "STOP AGGRESSIVELY" in text
assert "NEVER re-ask" in text
assert "LATEST answer wins" in text
assert "architectural forks" in text
def test_discovery_prompt_targets_at_most_two_rounds():
text = discovery.SYSTEM_PROMPT
assert "Target at most TWO question rounds" in text
assert "recording lower-priority unknowns as assumptions" in text
def test_discovery_importance_vocabulary_present():
"""Discovery must use the exact critical/optional/not_applicable vocabulary —
replacing it with synonyms like "high" caused a real validation crash."""
text = discovery.SYSTEM_PROMPT
for word in ('"critical"', '"optional"', '"not_applicable"'):
assert word in text
from agentic_core.schemas.discovery import MissingInfo
assert set(MissingInfo.model_fields["importance"].annotation.__args__) == {
"critical",
"optional",
"not_applicable",
}
def test_anti_overengineering_guidance_present():
assert "ANTI-OVERENGINEERING" in architecture.SYSTEM_PROMPT
assert "modular monolith" in architecture.SYSTEM_PROMPT
assert "speculative entities or redundant tables" in database.SYSTEM_PROMPT
assert "No hypothetical/future endpoints" in api.SYSTEM_PROMPT
assert "ANTI-OVERENGINEERING" in devops.SYSTEM_PROMPT
assert "Docker Compose" in devops.SYSTEM_PROMPT
def test_requirements_prompt_bounds_output():
text = requirements.SYSTEM_PROMPT
assert "concise, bounded and testable" in text
assert "4-6" in text
# ---------------------------------------------------------------- context hygiene
async def test_requirements_agent_condenses_context(provider, llm_service, make_context):
agent = RequirementsAgent(llm_service)
context = make_context("Food delivery.")
context.target_users = [f"user-{i}" for i in range(20)]
context.business_goals = [f"goal-{i}" for i in range(30)]
provider.set_responses([json.dumps(requirements_output())])
await agent.run(context)
user_prompt = provider.calls[0][1]
# Lists are capped by condense_context (12 per list) — the prompt is small.
assert "user-12" not in user_prompt
assert "goal-12" not in user_prompt
assert "more items omitted" in user_prompt
async def test_downstream_agents_get_digests_not_raw_artifacts(
provider, llm_service, make_context
):
"""Database receives the architecture digest (components), never the raw
architecture artifact with prose-heavy fields like communication."""
agent = DatabaseAgent(llm_service)
context = make_context("Food delivery.")
context.requirements = requirements_output()
context.architecture = architecture_output()
provider.set_responses([json.dumps(database_output())])
await agent.run(context)
user_prompt = provider.calls[0][1]
assert '"system_components"' in user_prompt
assert "mermaid_diagram" not in user_prompt
assert "flowchart" not in user_prompt
async def test_reviewer_never_sees_project_context(provider, llm_service, make_context):
from agentic_core.agents import ReviewAgent
agent = ReviewAgent(llm_service)
context = make_context("Food delivery.")
context.requirements = requirements_output()
context.architecture = architecture_output()
context.database = database_output()
context.api = api_output()
context.devops = devops_output()
provider.set_responses([json.dumps({"status": "approved", "score": 0.9, "issues": [], "artifacts_to_regenerate": []})])
await agent.run(context)
user_prompt = provider.calls[0][1]
assert "Food delivery." not in user_prompt
assert "PROJECT CONTEXT" not in user_prompt
assert "business_idea" not in user_prompt
# ---------------------------------------------------------------- digest integrity
def test_digests_preserve_cross_artifact_contracts():
"""Digests keep the exact names downstream agents must match."""
req = requirements_output()
arch = architecture_output()
db = database_output()
api = api_output()
dev = devops_output()
req_digest = json.dumps(digest_requirements(req))
assert "FR1" in req_digest
assert "user_stories" not in req_digest # derived prose, not a contract
from agentic_core.agents import digest_architecture
arch_digest = json.dumps(digest_architecture(arch))
assert "PostgreSQL" in arch_digest
assert "mermaid_diagram" not in arch_digest
from agentic_core.agents import digest_database
db_digest = json.dumps(digest_database(db))
assert '"orders"' in db_digest
assert "sql_schema" not in db_digest
from agentic_core.agents import digest_api
api_digest = json.dumps(digest_api(api))
assert "/api/orders" in api_digest
assert "openapi" not in api_digest
from agentic_core.agents import digest_devops
dev_digest = json.dumps(digest_devops(dev))
assert "Docker Compose" in dev_digest
# ---------------------------------------------------------------- orchestration
async def test_opt_in_summarizer_does_not_break_default_path(
provider, make_orchestrator, make_context
):
"""With summarize_with_llm on, handoffs use LLM summaries; the default off
path remains fully digest-based. Both converge."""
provider.set_handler(build_handler())
orchestrator = make_orchestrator(summarize_with_llm=True)
context = make_context("Food delivery.")
context.status = "ready_for_confirmation"
orchestrator.confirm(context)
await orchestrator.generate(context)
order = [detect_agent(c[0]) for c in provider.calls]
assert order.count("summarizer") == 5
assert context.status == "approved"
assert context.requirements_summary.startswith("Requirements summary")
def test_execution_levels_never_starve_an_agent(make_orchestrator):
"""api and devops still run concurrently, but only after the database design
they are told to match actually exists. Sharing a level with database meant
both received an empty object where the schema should have been."""
import agentic_core.orchestrator.orchestrator as orch_mod
orchestrator = make_orchestrator()
levels = orchestrator._execution_levels(orch_mod.ENGINEERING_ORDER)
assert levels == [
["requirements"],
["architecture"],
["database"],
["api", "devops"],
]
# Every declared dependency is produced by a strictly earlier level.
produced: set[str] = set()
for level in levels:
for name in level:
assert set(orch_mod.DEPENDENCIES[name]) <= produced
produced.update(level)
# ---------------------------------------------------------------- speed / token resilience
def test_schema_pre_normalizers_prevent_repair_roundtrips():
"""Synonymous LLM output strings are normalized in memory to avoid 60-120s repair roundtrips."""
from agentic_core.schemas import (
ArchitectureOutput,
DBField,
DiscoveryOutput,
MissingInfo,
ReviewIssue,
ReviewOutput,
SystemComponent,
)
# 1. Discovery importance & status
info = MissingInfo(field="target_users", importance="high", reason="essential")
assert info.importance == "critical"
info_opt = MissingInfo(field="theme", importance="nice_to_have", reason="aesthetic")
assert info_opt.importance == "optional"
disc = DiscoveryOutput(
status="done", confidence=0.95, summary="All set",
missing_information=[info],
)
assert disc.status == "ready"
# 2. Architecture component type
comp_api = SystemComponent(name="API", type="api", description="Core API", technology="Python")
assert comp_api.type == "backend"
comp_db = SystemComponent(name="DB", type="db", description="Main DB", technology="PostgreSQL")
assert comp_db.type == "database"
comp_k8s = SystemComponent(name="K8s", type="infra", description="Cluster", technology="Kubernetes")
assert comp_k8s.type == "infrastructure"
# 3. Review issue artifact & severity & status
issue = ReviewIssue(artifact="db", severity="blocker", problem="Mismatch", expected="PostgreSQL", actual="MySQL")
assert issue.artifact == "database"
assert issue.severity == "blocking"
review = ReviewOutput(status="pass", score=1.0, issues=[issue])
assert review.status == "approved"
# 4. DBField boolean & foreign key normalization
f = DBField(name="id", type="UUID", primary_key="true", foreign_key="false")
assert f.primary_key is True
assert f.foreign_key is None
def test_extract_json_tolerates_comments_and_trailing_commas():
from agentic_core.llm.service import extract_json_object
text_with_comments = """
Here is your JSON response:
```json
{
// Primary status
"status": "ready",
"confidence": 0.95, // high confidence
"summary": "Ready to build",
}
```
"""
extracted = extract_json_object(text_with_comments)
assert extracted["status"] == "ready"
assert extracted["confidence"] == 0.95
def test_truncated_response_is_rejected_not_salvaged():
"""A response cut off by max_tokens must fail loudly.
The balanced-region fallback used to hand back an inner fragment — one entity
out of a database design — which then validated into an artifact with zero
entities and was committed as a successful run."""
full = json.dumps(database_output(), indent=2)
truncated = full[: len(full) // 2]
with pytest.raises(StructuredOutputError) as excinfo:
extract_json_object(truncated)
assert "cut off" in str(excinfo.value)
def test_primary_fields_are_required_so_fragments_cannot_validate():
"""Even if a fragment reaches validation, an artifact missing its whole
payload is not a success."""
for schema, field in (
(DatabaseOutput, "entities"),
(APIOutput, "endpoints"),
(ArchitectureOutput, "system_components"),
(DevopsOutput, "dockerfile"),
):
assert field in LLMService._schema_for(schema)["required"]
with pytest.raises(Exception):
schema.model_validate({})
def test_output_budgets_are_published_as_maxitems():
"""The ceilings live in the schema, not only in prose: measured against a
real run, prose ceilings were overshot 3-6x (71 endpoints for "max 8-12")."""
assert LLMService._schema_for(APIOutput)["properties"]["endpoints"]["maxItems"] == 12
db = LLMService._schema_for(DatabaseOutput)
assert db["properties"]["entities"]["maxItems"] == 8
assert db["$defs"]["DBEntity"]["properties"]["fields"]["maxItems"] == 10
arch = LLMService._schema_for(ArchitectureOutput)
assert arch["properties"]["system_components"]["maxItems"] == 6
req = LLMService._schema_for(RequirementsOutput)
assert req["properties"]["functional_requirements"]["maxItems"] == 8
def test_over_budget_output_is_trimmed_not_rejected():
"""Overshooting is trimmed in memory. Rejecting it would cost a full repair
round-trip for what is a purely cosmetic overrun."""
api_out = APIOutput.model_validate(
{"endpoints": [{"method": "GET", "path": f"/r{i}", "summary": "s"} for i in range(71)]}
)
assert len(api_out.endpoints) == 12
db = database_output()
db["entities"] = [
{"name": f"e{i}", "description": "d", "fields": [
{"name": f"f{j}", "type": "text"} for j in range(21)
]}
for i in range(22)
]
parsed = DatabaseOutput.model_validate(db)
assert len(parsed.entities) == 8
assert all(len(e.fields) <= 10 for e in parsed.entities)
def test_downstream_agents_get_a_narrower_context():
"""The same context block is embedded in every engineering prompt, so the
agents that also read the requirements digest get only the fields that still
carry decisions."""
from agentic_core.agents.digest import condense_context
payload = {
"project_id": "p1",
"business_idea": "An idea",
"target_users": ["someone"],
"business_goals": ["grow"],
"assumptions": ["assumed"],
"problem": "the problem",
"user_roles": ["admin"],
"technology_preferences": ["Postgres"],
}
full = condense_context(payload)
downstream = condense_context(payload, downstream=True)
assert "business_idea" in full and "assumptions" in full
assert set(downstream) == {"problem", "user_roles", "technology_preferences"}
assert len(json.dumps(downstream)) < len(json.dumps(full))
async def test_revision_payload_drops_locally_derived_fields(
provider, make_orchestrator, make_context
):
"""A revision resends the existing artifact; the fields the system derives
itself are excluded from the schema, so echoing them back is dead weight."""
orchestrator = make_orchestrator()
context = make_context("Food delivery.")
context.database = database_output()
payload = orchestrator._revision_payload(context, "database")
assert "entities" in payload
for derived in ("sql_schema", "erd_mermaid", "indexes", "constraints"):
assert derived not in payload
def test_dumps_prunes_empty_and_null_to_save_tokens():
from agentic_core.agents.digest import dumps
payload = {
"name": "Service",
"description": "Short",
"empty_list": [],
"none_val": None,
"empty_str": "",
"nested": {"valid": 1, "empty": {}},
}
serialized = dumps(payload)
assert "empty_list" not in serialized
assert "none_val" not in serialized
assert "empty_str" not in serialized
assert serialized == '{"name":"Service","description":"Short","nested":{"valid":1}}'