RICS / app /tests /test_inspector_loop.py
StormShadow308's picture
feat: async pipeline, job queue, generation hardening, and docs
732b14f
Raw
History Blame Contribute Delete
32.5 kB
"""Sanity checks for OpenAI inspector tool definitions (no live API calls)."""
import pytest
from httpx import ASGITransport, AsyncClient
from app.agentic.agents import RiskAssessmentAgent, _risk_priority_key
from app.agentic.inspector_loop import (
_enforce_verify_submit_payload,
_openai_inspector_tool_definitions,
_section_c_requires_rating_summary,
_validate_condition_rating_summary,
_verify_audit_payload,
_verify_condition_payload,
_verify_plan_payload,
_verify_risks,
)
from app.agentic.models import RiskItem
from app.agentic.runtime_status import inspector_public_status, is_openai_inspector_live
from app.templates.registry import get_survey_pack
def test_inspector_openai_tools_include_core_names() -> None:
tools = _openai_inspector_tool_definitions()
names = [t["function"]["name"] for t in tools]
assert "retrieve_survey_rag" in names
assert "retrieve_rics_kb" in names
assert "scan_note_duplicates" in names
assert "submit_extraction_audit" in names
assert "submit_section_plan" in names
assert "submit_condition_rating_summary" in names
assert "submit_inspection_section" in names
assert len(tools) == 7
def test_section_c_rating_summary_gate_depends_on_pack_not_only_level_3() -> None:
"""Level 1 pack still includes rated element sections; Section C should require the summary tool."""
pack1 = get_survey_pack(1)
assert _section_c_requires_rating_summary(section_code="C", pack=pack1) is True
assert _section_c_requires_rating_summary(section_code="E1", pack=pack1) is False
def test_section_c_rating_summary_gate_level_2_and_3_packs() -> None:
"""HomeBuyer (2) and Building Survey (3) packs include rated elements — Section C gate applies."""
for level in (2, 3):
pack = get_survey_pack(level)
assert _section_c_requires_rating_summary(section_code="C", pack=pack) is True
assert _section_c_requires_rating_summary(section_code="A", pack=pack) is False
def test_validate_condition_rating_summary_accepts_markdown_table() -> None:
table = "| Cat | NI | 3 | 2 | 1 |\n| --- | --- | --- | --- | --- |\n| Roof | 0 | 1 | 2 | 3 |\n"
err, payload = _validate_condition_rating_summary({"ratings_not_applicable": False, "summary_markdown_table": table})
assert err == ""
assert payload is not None
assert payload["ratings_not_applicable"] is False
def test_inspector_public_status_shape(monkeypatch: pytest.MonkeyPatch) -> None:
from app.config import settings
monkeypatch.setattr(settings, "openai_api_key", "")
monkeypatch.setattr(settings, "inspector_tool_agent", True)
assert is_openai_inspector_live() is False
st = inspector_public_status()
assert st["effective_mode"] == "legacy_pipeline"
assert st["openai_api_key_configured"] is False
assert "tool_agent_endpoints" in st
monkeypatch.setattr(settings, "openai_api_key", "sk-test")
monkeypatch.setattr(settings, "inspector_tool_agent", True)
assert is_openai_inspector_live() is True
assert inspector_public_status()["effective_mode"] == "openai_tool_agent"
monkeypatch.setattr(settings, "inspector_tool_agent", False)
assert is_openai_inspector_live() is False
def test_risk_priority_key_sorts_critical_above_low() -> None:
"""Regression: the LLM-based RiskAssessmentAgent prompt allows severity="Critical",
but the old _risk_priority_key dict only mapped High/Medium/Low and fell back to a
sort key of 3 — sinking Critical *below* Low. That broke `consolidated[:5]` in
`generate_full_report`, dropping the most severe defects from the executive summary.
"""
items = [
RiskItem(category="A", risk="low item", severity="Low", likelihood="Low", action=""),
RiskItem(category="B", risk="critical item", severity="Critical", likelihood="High", action=""),
RiskItem(category="C", risk="high item", severity="High", likelihood="Medium", action=""),
RiskItem(category="D", risk="medium item", severity="Medium", likelihood="Low", action=""),
RiskItem(category="E", risk="garbage severity", severity="bogus", likelihood="Low", action=""),
]
ordered = sorted(items, key=_risk_priority_key)
assert [r.severity for r in ordered] == ["Critical", "High", "Medium", "Low", "bogus"], (
"Critical must sort first; unknown severities must fall to the bottom, never above Low."
)
def test_risk_priority_key_is_case_insensitive() -> None:
a = RiskItem(category="x", risk="y", severity="CRITICAL", likelihood="HIGH", action="")
b = RiskItem(category="x", risk="y", severity="critical", likelihood="high", action="")
assert _risk_priority_key(a) == _risk_priority_key(b) == (0, 0)
def test_coerce_risk_item_drops_unknown_keys() -> None:
"""Regression: previously `RiskItem(**item)` raised TypeError on any extra
LLM key (e.g. "description"), and the broad except silently degraded the
entire batch to keyword heuristics — discarding the contextual analysis.
"""
raw = {
"category": "Moisture",
"risk": "Damp ingress",
"severity": "Medium",
"likelihood": "Medium",
"action": "Investigate source.",
"description": "Visible staining on north wall.",
"details": {"location": "kitchen"},
"reasoning": "Based on the photographs.",
}
item = RiskAssessmentAgent._coerce_risk_item(raw)
assert item is not None
assert item.category == "Moisture"
assert item.risk == "Damp ingress"
assert item.severity == "Medium"
assert item.likelihood == "Medium"
assert item.action == "Investigate source."
def test_coerce_risk_item_skips_when_required_key_missing() -> None:
"""A single item missing a required key must be skipped, not raise — so the
rest of the batch survives instead of falling back wholesale.
"""
assert (
RiskAssessmentAgent._coerce_risk_item(
{
"category": "Electrical",
"risk": "Old wiring",
"severity": "High",
"likelihood": "Medium",
# action intentionally omitted
}
)
is None
)
def test_coerce_risk_item_rejects_non_dict() -> None:
assert RiskAssessmentAgent._coerce_risk_item("not a dict") is None
assert RiskAssessmentAgent._coerce_risk_item(None) is None
assert RiskAssessmentAgent._coerce_risk_item(["category", "risk"]) is None
def test_coerce_risk_item_coerces_non_string_values_and_strips() -> None:
item = RiskAssessmentAgent._coerce_risk_item(
{
"category": " Structural ",
"risk": "Cracking",
"severity": "high",
"likelihood": "Medium",
"action": None, # null in JSON — must become empty string, not crash
}
)
assert item is not None
assert item.category == "Structural"
assert item.action == ""
@pytest.mark.asyncio
async def test_assess_survives_partial_llm_drift(monkeypatch) -> None:
"""End-to-end: when the LLM returns one drifted item alongside three valid
ones, the assessor returns the three valid items rather than collapsing to
the keyword fallback.
"""
monkeypatch.setattr("app.config.settings.openai_api_key", "sk-test")
drifted_payload = (
'['
' {"category":"Moisture","risk":"Damp","severity":"Medium",'
' "likelihood":"Medium","action":"Investigate.","description":"extra"},'
' {"category":"Electrical","risk":"Old wiring","severity":"High",'
' "likelihood":"Medium"},' # missing "action" — must be skipped, not fatal
' {"category":"Structural","risk":"Cracking","severity":"Critical",'
' "likelihood":"Medium","action":"Engineer review.","reasoning":"x"},'
' {"category":"Gas","risk":"Old boiler","severity":"High",'
' "likelihood":"Low","action":"Service.","evidence":"photo-1"}'
']'
)
async def _fake_chat(**kwargs):
return drifted_payload
monkeypatch.setattr("app.llm.openai_chat.chat_completions_create", _fake_chat)
agent = RiskAssessmentAgent()
risks = await agent.assess(section_code="E2", bullets=["roof shows damp staining"])
assert len(risks) == 3, (
f"expected 3 valid items kept, got {len(risks)} — drifted item likely "
"collapsed the whole batch to keyword fallback"
)
categories = {r.category for r in risks}
assert categories == {"Moisture", "Structural", "Gas"}
assert {r.severity for r in risks} == {"Medium", "Critical", "High"}
@pytest.mark.asyncio
async def test_enforce_verify_submit_payload_strips_invented_address(monkeypatch) -> None:
"""Regression: the agentic ``submit_inspection_section`` payload bypassed the
non-invention guard, so invented identity facts ("10 Kingsley Avenue, Sutton",
"robust steel frame") shipped straight to the user-visible report.
Verify the helper now runs the regex pass on every field — invented postcodes
and capitalised place-name pairs not present in bullets/snippets must be
replaced with an approved missing-information phrase. The OpenAI key is set
to empty so only the cheap regex layer fires (deterministic, no live API).
"""
monkeypatch.setattr("app.config.settings.openai_api_key", "")
payload = {
"executive_summary": "The property at 10 Kingsley Avenue is in fair order.",
"property_description": "Main walls are constructed of a robust steel frame.",
"condition_assessment": "Condition rating 3 noted to roof timbers.",
"defects_and_risks": "Damp staining at SW1A 1AA postcode.",
"recommendations": "Engage a structural engineer.",
}
bullets = [
"Roof timbers — condition rating 3, evidence of damp staining",
"Recommend structural engineer review",
]
out = await _enforce_verify_submit_payload(
payload,
bullets=bullets,
snippets=[],
peer_sections=None,
)
# Invented address ("Kingsley Avenue") and postcode ("SW1A 1AA") must be
# stripped; the bullets-grounded "structural engineer" recommendation must
# survive untouched.
assert "Kingsley Avenue" not in out["executive_summary"]
assert "SW1A 1AA" not in out["defects_and_risks"]
assert "structural engineer" in out["recommendations"].lower()
@pytest.mark.asyncio
async def test_enforce_verify_submit_payload_preserves_grounded_facts() -> None:
"""Facts that DO appear in bullets/snippets must pass through unchanged.
The regex pass is deliberately conservative — only specific facts not found
verbatim (allowing pluralisation / case differences) get replaced. A
grounded postcode and a grounded named entity must both survive.
"""
payload = {
"executive_summary": "The property at SW1A 1AA was inspected.",
"property_description": "Hampstead Heath is nearby.",
"condition_assessment": "",
"defects_and_risks": "",
"recommendations": "",
}
out = await _enforce_verify_submit_payload(
payload,
bullets=["Address: SW1A 1AA", "Property near Hampstead Heath"],
snippets=[],
peer_sections=None,
)
assert "SW1A 1AA" in out["executive_summary"]
assert "Hampstead Heath" in out["property_description"]
@pytest.mark.asyncio
async def test_enforce_verify_submit_payload_treats_peer_sections_as_trusted() -> None:
"""Peer-section drafts (the surveyor's own notes for sibling sections) are
legitimate cross-section context and must be merged into the trusted
source set, not flagged as invention.
"""
payload = {
"executive_summary": "Damp evident throughout the rear elevation.",
"property_description": "",
"condition_assessment": "",
"defects_and_risks": "",
"recommendations": "",
}
out = await _enforce_verify_submit_payload(
payload,
bullets=[], # current section has no bullets
snippets=[],
peer_sections={
"F1": "Damp readings of 22% noted at the rear elevation in section F1.",
},
)
assert "rear elevation" in out["executive_summary"]
def test_verify_audit_payload_strips_invented_postcode_and_name() -> None:
"""Regression: extraction_audit fields ship in `inspector_meta`, so an
invented surveyor name or postcode there is just as user-visible as the
five body fields. The regex pass must clean those without an LLM call.
"""
audit = {
"property_address": "10 Kingsley Avenue, Sutton, SW1A 1AA",
"surveyor_name": "Jonathan Pemberton MRICS",
"rics_number": "Information not provided in source document.",
"company_name": "Ridge & Partners",
"inspection_date": "2024-03-12",
"report_reference": "RPT-001",
"property_type": "Detached house",
"construction_details": "Cavity brick wall.",
"services": "Mains gas, mains water.",
"client_names": ["Sophia Wright"],
"observed_defects": ["Damp staining to rear elevation"],
}
bullets = [
"Cavity brick wall construction; mains gas and mains water connected",
"Damp staining noted at rear elevation",
]
out = _verify_audit_payload(audit, bullets=bullets, snippets=[])
assert out is not None
# Invented capitalised entities not in source must be stripped.
assert "Kingsley Avenue" not in out["property_address"]
assert "SW1A 1AA" not in out["property_address"]
assert "Jonathan Pemberton" not in out["surveyor_name"]
assert "Sophia Wright" not in (out.get("client_names") or [None])[0]
# Grounded facts survive.
assert "cavity" in out["construction_details"].lower()
assert "rear elevation" in out["observed_defects"][0].lower()
def test_verify_plan_payload_cleans_claim_strings() -> None:
"""Section plan claims are LLM narrative — must run through the regex pass."""
plan = {
"claims": [
{
"claim": "Inspection at 10 Kingsley Avenue revealed damp to rear elevation.",
"evidence_refs": ["bullet_1"],
"risk": "medium",
}
],
"non_claims": ["Property valued at £999,999 by Acme Surveyors Ltd"],
}
bullets = ["Damp staining noted at rear elevation"]
out = _verify_plan_payload(plan, bullets=bullets, snippets=[])
assert out is not None
assert "Kingsley Avenue" not in out["claims"][0]["claim"]
assert "rear elevation" in out["claims"][0]["claim"].lower()
# Invented company name in a non_claim is stripped too.
assert "Acme Surveyors" not in out["non_claims"][0]
def test_verify_condition_payload_cleans_justification_and_table() -> None:
"""Section C summary table + justification can carry invented facts —
e.g. an LLM might cite a postcode or named contractor in the markdown.
"""
cond = {
"ratings_not_applicable": False,
"justification": "",
"summary_markdown_table": (
"| Element | NI | 3 | 2 | 1 |\n"
"| --- | --- | --- | --- | --- |\n"
"| Roof at 10 Kingsley Avenue | 0 | 1 | 0 | 0 |\n"
),
}
bullets = ["Roof condition rating 3"]
out = _verify_condition_payload(cond, bullets=bullets, snippets=[])
assert out is not None
assert "Kingsley Avenue" not in out["summary_markdown_table"]
# Genuine table tokens (NI, 3, 2, 1) must survive.
assert "NI" in out["summary_markdown_table"]
def test_verify_risks_strips_invented_firm_in_action() -> None:
"""The LLM RiskAssessmentAgent emits {risk, action} strings — both can
invent named contractors or postcodes. Regex pass must clean both.
"""
risks = [
RiskItem(
category="Structural",
risk="Movement evident at flank wall",
severity="High",
likelihood="Medium",
action="Engage Acme Structural Engineers Ltd of Sutton SW1A 1AA.",
),
]
bullets = ["Crack to flank wall — recommend structural engineer review"]
verified = _verify_risks(risks, bullets=bullets, snippets=[])
assert len(verified) == 1
r = verified[0]
# Real category + severity untouched (constrained enum values).
assert r.category == "Structural"
assert r.severity == "High"
# Invented firm name + invented postcode stripped from `action`.
assert "Acme Structural Engineers" not in r.action
assert "SW1A 1AA" not in r.action
# Grounded claim survives.
assert "flank wall" in r.risk.lower()
def test_inspector_system_prompt_demands_depth_not_summary() -> None:
"""The user reported the agentic output read like a one-paragraph summary.
Root cause: the system prompt told the LLM to ground facts but never to
produce *detailed* analysis, and gpt-4o-mini defaulted to terse output.
Lock the anti-summarisation directive AND the per-tier depth directive
into the prompt so a future edit can't silently regress either.
"""
from app.agentic.inspector_loop import _inspector_system_prompt
from app.templates.registry import get_survey_pack
prompt = _inspector_system_prompt(
tenant_id="t1",
primary_document_id="d1",
survey_level=3,
pack=get_survey_pack(3),
section_code="E1",
section_title="Roof structure",
style_profile=None,
)
lower = prompt.lower()
assert "not summaries" in lower or "not a summary" in lower
assert "comprehensive" in lower
# L3 depth directive: cause / implications / options layering.
assert "diagnostic mode" in lower
assert "cause" in lower and "implications" in lower and "options" in lower
# Per-tier word targets must be present in the prompt. We calibrate these
# per-section (not whole-report) so L3 condition_assessment is a smaller
# band than the prior Option-B "very long per section" configuration.
assert "60–120" in prompt or "60-120" in prompt
def test_inspector_system_prompt_l1_not_directive() -> None:
"""Level 1 condition reports must not be told to advise/recommend repairs."""
from app.agentic.inspector_loop import _inspector_system_prompt
from app.templates.registry import get_survey_pack
prompt = _inspector_system_prompt(
tenant_id="t1",
primary_document_id="d1",
survey_level=1,
pack=get_survey_pack(1),
section_code="C",
section_title="Overall assessment",
style_profile=None,
)
lower = prompt.lower()
# L1 directive must call out observation-only mode.
assert "observation mode" in lower or "do not advise" in lower
# L3-only language must NOT leak into L1.
assert "diagnostic mode" not in lower
def test_inspector_system_prompt_pins_identity_block() -> None:
"""Identity facts (extracted from the surveyor's bullets/uploaded survey)
must be carried into the inspector system prompt as NON-NEGOTIABLE so the
LLM can never invent a different address/postcode/property type."""
from app.agentic.inspector_loop import _inspector_system_prompt
from app.templates.registry import get_survey_pack
identity_block = (
"- Address: 2 Cunliffe Road, Epsom (use exactly this; do not introduce any other address)\n"
"- Postcode: KT19 0RL (use exactly this; do not invent postcodes)"
)
prompt = _inspector_system_prompt(
tenant_id="t1",
primary_document_id="d1",
survey_level=3,
pack=get_survey_pack(3),
section_code="A",
section_title="Introduction",
style_profile=None,
identity_block=identity_block,
)
assert "2 Cunliffe Road" in prompt
assert "KT19 0RL" in prompt
assert "non-negotiable" in prompt.lower()
def test_inspector_submit_schema_includes_per_field_word_targets() -> None:
"""The submit schema's field descriptions must carry length guidance,
otherwise the LLM treats each field as "as short as possible". The
schema now defers exact ranges to the per-tier system prompt
word_targets, so we assert on the deferral language rather than on
a hard-coded number that's wrong for two of the three tiers.
"""
tools = _openai_inspector_tool_definitions()
submit = next(
t for t in tools if t["function"]["name"] == "submit_inspection_section"
)
props = submit["function"]["parameters"]["properties"]
for field in (
"executive_summary",
"property_description",
"condition_assessment",
"defects_and_risks",
"recommendations",
):
desc = props[field].get("description", "").lower()
assert "word" in desc, f"{field} description missing word target"
assert "tier" in desc, (
f"{field} description must defer length to tier (otherwise L3 ranges "
f"baked here will misguide L1/L2 generations)"
)
def test_inspector_submit_schema_no_baked_l3_ranges() -> None:
"""Schema field descriptions must NOT bake in concrete L3 word ranges
like "200-500 words" — those are correct for L3 but cause the model to
over-generate at L1 (40-90 word target) and approach L3 length at L2.
The per-tier system prompt is the single source of truth.
"""
tools = _openai_inspector_tool_definitions()
submit = next(
t for t in tools if t["function"]["name"] == "submit_inspection_section"
)
props = submit["function"]["parameters"]["properties"]
for field, desc in (
(name, props[name].get("description", "")) for name in props
):
# Hard-coded L3-leaning ranges that previously appeared in this schema.
assert "60–180" not in desc, f"{field} schema still bakes L3 60-180 range"
assert "150–400" not in desc, f"{field} schema still bakes L3 150-400 range"
assert "200–500" not in desc, f"{field} schema still bakes L3 200-500 range"
assert "100–300" not in desc, f"{field} schema still bakes L3 100-300 range"
def test_inspector_system_prompt_l2_proportionate_directive() -> None:
"""Level 2 (HomeBuyer) must receive a *proportionate* depth directive —
not the L3 cause/implications/options diagnostic frame, and not the L1
observation-only frame. Lock the contract so future edits can't
silently collapse L2 into one of the neighbouring tiers."""
from app.agentic.inspector_loop import _inspector_system_prompt
from app.templates.registry import get_survey_pack
prompt = _inspector_system_prompt(
tenant_id="t1",
primary_document_id="d1",
survey_level=2,
pack=get_survey_pack(2),
section_code="E1",
section_title="Roof structure",
style_profile=None,
)
lower = prompt.lower()
# L2 marker — proportionate / HomeBuyer / buyer-focused.
assert "homebuyer" in lower or "proportionate" in lower
# L3-only diagnostic markers must NOT leak into L2.
assert "diagnostic mode" not in lower
# L1-only observation-only markers must NOT leak into L2.
assert "observation mode" not in lower
# Per-tier word ranges visible in the prompt must be the L2 set, not L3.
# Calibrated-to-total profile: Inspector L2 condition_assessment = ~45–90.
assert "45–90" in prompt or "45-90" in prompt
def test_inspector_word_ranges_match_total_aware_profile() -> None:
"""Condition-assessment ranges should reflect full-report totals.
With only 5 sections in L1, per-section L1 can be larger than L2.
"""
from app.agentic.inspector_loop import _inspector_system_prompt
from app.templates.registry import get_survey_pack
import re
def _condition_upper_bound(level: int) -> int:
prompt = _inspector_system_prompt(
tenant_id="t1",
primary_document_id="d1",
survey_level=level,
pack=get_survey_pack(level),
section_code="E1",
section_title="Roof",
style_profile=None,
)
# Match: condition_assessment ~LOW–HIGH words
m = re.search(
r"condition_assessment\s*~\s*\d+[\u2013\-]\s*(\d+)\s*words",
prompt,
)
assert m, f"condition_assessment word range not found in L{level} prompt"
return int(m.group(1))
l1_upper = _condition_upper_bound(1)
l2_upper = _condition_upper_bound(2)
l3_upper = _condition_upper_bound(3)
assert l1_upper > l2_upper, (
f"Expected L1 per-section budget > L2 (fewer total sections); got {l1_upper}, {l2_upper}"
)
assert l3_upper > l2_upper, (
f"Expected L3 diagnostic budget > L2; got {l3_upper}, {l2_upper}"
)
def test_strip_l1_advice_payload_used_only_when_l1() -> None:
"""`strip_l1_advice_payload` is the agentic loop's L1 sanitiser.
L2 and L3 outputs intentionally include recommendations and must NOT
pass through it. This test pins the contract by checking the import
surface — :mod:`postprocess` exposes the helper, and the inspector
loop imports it. If a future refactor accidentally calls the helper
in the L2/L3 paths the test in
test_l1_advice_strips_advisory_phrases_in_payload below catches the
behaviour, but this one keeps the public surface tidy.
"""
from app.generator import postprocess
assert hasattr(postprocess, "strip_l1_advice")
assert hasattr(postprocess, "strip_l1_advice_payload")
def test_l1_advice_strips_advisory_phrases_in_payload() -> None:
"""End-to-end behaviour of the payload-level sanitiser used by the
agentic inspector loop on L1 outputs: advice is dropped, observation
survives, and a fully-advisory recommendations field becomes the L1
placeholder so the renderer never shows an empty heading."""
from app.generator.postprocess import strip_l1_advice_payload
payload = {
"executive_summary": (
"The two-bedroom flat is in serviceable condition. "
"We recommend further investigation of the rear elevation."
),
"property_description": "A two-bedroom first-floor flat with double-glazed windows.",
"condition_assessment": (
"The kitchen units showed light wear consistent with age. "
"The boiler should be replaced ahead of completion."
),
"defects_and_risks": (
"Hairline cracks were noted at the rear elevation. "
"Obtain a structural engineer's report before proceeding."
),
"recommendations": "We recommend a full electrical inspection.",
}
cleaned = strip_l1_advice_payload(payload)
# Advisory sentences are stripped from each field.
assert "we recommend" not in cleaned["executive_summary"].lower()
assert "should be replaced" not in cleaned["condition_assessment"].lower()
assert "structural engineer" not in cleaned["defects_and_risks"].lower()
# All-advisory fields surface the L1 placeholder.
assert "level 1" in cleaned["recommendations"].lower()
# Observation-only field is preserved verbatim.
assert (
cleaned["property_description"].strip()
== payload["property_description"].strip()
)
@pytest.mark.asyncio
async def test_generate_full_report_l1_fallback_does_not_leak_advice() -> None:
"""Regression for the L1 exec-summary exception fallback.
The previous fallback inside ``except Exception`` in
:func:`app.agentic.agents.generate_full_report` was a single hardcoded
string that contained "recommended next steps" — a phrase
`_L1_ADVICE_KEYWORDS_RE` flags as forbidden L1 advice. Worse, the L1
sanitiser was called only inside the try block, so on adapter failure
an L1 product would ship directive phrasing in its top-level
executive summary.
This test forces the adapter to raise, runs ``generate_full_report``
with ``survey_level=1`` and an empty bullets map (so each section
short-circuits to its NA placeholder and we don't need a real DB or
inspector loop), and asserts the resulting ``executive_summary`` is
free of the advice phrases the rest of the L1 enforcement removes.
"""
from unittest.mock import MagicMock, patch
from app.agentic.agents import generate_full_report
from app.generator.postprocess import _L1_ADVICE_KEYWORDS_RE
from app.models.schemas import WritingStyleProfile
# Adapter that raises for the exec-summary call. The per-section path
# also calls the adapter via the inspector loop or legacy pipeline,
# but we route around that by passing an empty bullets_by_section so
# every section uses its NA-placeholder branch (no LLM call).
raising_adapter = MagicMock()
raising_adapter.generate_section.side_effect = RuntimeError(
"simulated adapter outage"
)
style = WritingStyleProfile(
tone="formal",
formality_level="professional",
avg_sentence_complexity="moderate",
vocabulary_level="technical",
common_phrases=[],
structural_patterns=[],
writing_style_summary="L1 fallback test",
example_paragraphs=[],
)
with patch(
"app.generator.adapter.get_llm_adapter", return_value=raising_adapter
):
result = await generate_full_report(
db=None,
tenant_id="t1",
primary_document_id="doc1",
bullets_by_section={},
style_profile=style,
ai_percent=50,
survey_level=1,
)
exec_text = result.get("executive_summary", "")
# The fallback path was reached (no LLM-generated text exists in this run).
assert exec_text, "expected exception fallback to populate executive_summary"
# Critical: no advice / directive phrasing in an L1 product, even on
# adapter failure. Run the same regex the agentic loop uses to detect
# leakage so this catches any future regression that re-introduces
# "recommended", "we suggest", "should be replaced", etc.
assert not _L1_ADVICE_KEYWORDS_RE.search(exec_text), (
f"L1 fallback exec_text leaks advice phrasing: {exec_text!r}"
)
# And specifically the phrase the user-supplied bug report cited.
assert "recommended next steps" not in exec_text.lower()
@pytest.mark.asyncio
async def test_generate_full_report_l3_fallback_keeps_recommendations() -> None:
"""Counter-test for the L1 fallback fix: the L3 fallback may legitimately
reference "recommended next steps" because L3 (Building Survey) is an
advice product. This locks in tier-aware fallback behaviour so a future
edit can't accidentally apply the L1 sanitiser to L3 / strip
legitimate advice from a Building Survey on adapter failure.
"""
from unittest.mock import MagicMock, patch
from app.agentic.agents import generate_full_report
from app.models.schemas import WritingStyleProfile
raising_adapter = MagicMock()
raising_adapter.generate_section.side_effect = RuntimeError("outage")
style = WritingStyleProfile(
tone="formal", formality_level="professional",
avg_sentence_complexity="moderate", vocabulary_level="technical",
common_phrases=[], structural_patterns=[],
writing_style_summary="L3 fallback test", example_paragraphs=[],
)
with patch(
"app.generator.adapter.get_llm_adapter", return_value=raising_adapter
):
result = await generate_full_report(
db=None,
tenant_id="t1",
primary_document_id="doc1",
bullets_by_section={},
style_profile=style,
ai_percent=50,
survey_level=3,
)
exec_text = result.get("executive_summary", "")
assert exec_text
# L3 advice phrasing IS allowed and expected.
assert "recommended" in exec_text.lower() or "next steps" in exec_text.lower()
@pytest.mark.asyncio
async def test_health_includes_rics_inspector_block() -> None:
from app.main import create_app
app = create_app()
async with AsyncClient(transport=ASGITransport(app=app), base_url="http://test") as client:
res = await client.get("/health")
assert res.status_code == 200
data = res.json()
assert "rics_inspector" in data
assert data["rics_inspector"]["effective_mode"] in ("openai_tool_agent", "legacy_pipeline")
assert "summary" in data["rics_inspector"]