amplegest / tests /test_post_synthesis.py
Viney's picture
feat: multi-provider LLM support, prominent chat, design pass, and new analytics
7880373
Raw
History Blame Contribute Delete
6.75 kB
"""Unit tests for conservative post-synthesis reliability scoring.
Verifies that apply_reliability:
- never promotes transcript corroboration above MEDIUM
- keeps a 10-K MD&A fact at HIGH when not corroborated
- downgrades a Risk Factors fact (heuristic on snippet) to MEDIUM
- keeps a stale lone news fact at LOW and notes it
- appends auto evidence_notes capped at 3
"""
from __future__ import annotations
from datetime import datetime, timedelta
import pytest
from agent.post_synthesis import apply_reliability
def _fact(text, source, snippet, reliability="HIGH", verification_status="VERIFIED"):
return {
"text": text,
"source": source,
"reliability": reliability,
"evidence_snippet": snippet,
"verification_status": verification_status,
}
def test_transcript_corroboration_does_not_promote_above_medium():
"""Cross-source similarity is explanatory, never a reliability promotion."""
brief = {
"filing_date": "2026-04-01",
"bull_points": [
_fact(
"Services revenue accelerated meaningfully",
"transcript",
"services revenue grew twenty four percent driven by subscriptions",
reliability="MEDIUM",
),
],
"what_changed": [
_fact(
"Services segment posted a strong quarter",
"10-Q",
"services revenue increased twenty four percent year over year subscriptions",
reliability="HIGH",
),
],
}
out = apply_reliability(brief)
assert out["bull_points"][0]["reliability"] == "MEDIUM"
notes = out.get("evidence_notes") or []
assert not any("uplift" in n.lower() for n in notes), notes
def test_lone_news_stays_low_and_gets_note():
"""News with no corroboration AND >30d old → LOW + auto note."""
old_date = (datetime.now() - timedelta(days=45)).strftime("%Y-%m-%d")
brief = {
"filing_date": old_date,
"bear_points": [
_fact(
"Analyst downgrade",
"news",
"downgraded to neutral after results citing macro uncertainty",
reliability="MEDIUM",
),
],
}
out = apply_reliability(brief)
assert out["bear_points"][0]["reliability"] == "LOW"
notes = out.get("evidence_notes") or []
assert any("only from news" in n.lower() for n in notes), notes
def test_filing_mda_uncorroborated_stays_high():
"""A 10-K MD&A driver with no peer in another source → stays HIGH."""
brief = {
"filing_date": "2026-04-01",
"mda_summary": {
"drivers": [
_fact(
"Operating leverage from cloud platform",
"10-K",
"operating leverage continued as cloud platform scaled across verticals",
reliability="HIGH",
),
],
"headwinds": [],
"language_shift": "",
"key_quote": _fact(
"We expect continued strength",
"10-K",
"we expect continued strength in our cloud business throughout fiscal year",
reliability="HIGH",
),
},
}
out = apply_reliability(brief)
assert out["mda_summary"]["drivers"][0]["reliability"] == "HIGH"
assert out["mda_summary"]["key_quote"]["reliability"] == "HIGH"
def test_risk_factors_heuristic_downgrades_to_medium():
"""A 10-K fact whose snippet contains 'risk', 'litigation', etc → MEDIUM."""
brief = {
"filing_date": "2026-04-01",
"risks_categorized": [
{
**_fact(
"Cybersecurity exposure remains material",
"10-K",
"cybersecurity risks could materially harm operations litigation exposure remains",
reliability="HIGH",
),
"category": "Cybersecurity",
},
],
}
out = apply_reliability(brief)
assert out["risks_categorized"][0]["reliability"] == "MEDIUM"
notes = out.get("evidence_notes") or []
assert any("risk factors" in n.lower() for n in notes), notes
def test_auto_notes_capped_at_three():
"""Many lone-news facts → only 3 auto notes appended."""
old_date = (datetime.now() - timedelta(days=60)).strftime("%Y-%m-%d")
brief = {
"filing_date": old_date,
"bear_points": [
_fact(f"News claim {i}", "news",
f"unique news content number {i} with several distinctive words here",
reliability="MEDIUM")
for i in range(8)
],
"evidence_notes": [],
}
out = apply_reliability(brief)
notes = out.get("evidence_notes") or []
assert len(notes) <= 3
def test_preserves_existing_evidence_notes():
"""Existing LLM-authored notes are kept, auto notes are appended."""
brief = {
"filing_date": "2026-04-01",
"bull_points": [
_fact("X", "transcript", "services revenue grew twenty four percent driven by",
reliability="MEDIUM"),
],
"what_changed": [
_fact("Y", "10-Q", "services revenue increased twenty four percent year over year",
reliability="HIGH"),
],
"evidence_notes": ["Pre-existing LLM note about X"],
}
out = apply_reliability(brief)
notes = out.get("evidence_notes") or []
assert "Pre-existing LLM note about X" in notes
def test_handles_empty_brief():
out = apply_reliability({})
assert out == {}
def test_handles_none_brief():
assert apply_reliability(None) is None
def test_short_snippet_no_corroboration():
"""Snippet with <4 meaningful tokens cannot corroborate anything."""
brief = {
"filing_date": "2026-04-01",
"bull_points": [
_fact("Short snippet", "transcript", "yes good", reliability="MEDIUM"),
],
"what_changed": [
_fact("Same idea", "10-Q", "yes good results", reliability="HIGH"),
],
}
out = apply_reliability(brief)
# transcript stays MEDIUM (no real corroboration)
assert out["bull_points"][0]["reliability"] == "MEDIUM"
def test_missing_verification_status_is_always_low():
brief = {
"bull_points": [
_fact(
"Unverified filing claim",
"10-K",
"a plausible but unverified filing statement",
verification_status=None,
)
]
}
out = apply_reliability(brief)
assert out["bull_points"][0]["reliability"] == "LOW"