"""Deterministic contradiction & duplicate audit (STEP 7). Pure-Python verifier that compares extracted findings and flags logical inconsistencies, tracing each back to the findings (and therefore source spans) that produced it. No network dependency — fully unit-testable. Detected classes: * ``RATING_CONFLICT`` — same element given two different condition ratings. * ``CONDITION`` — same element described both satisfactory and defective. * ``OPERATIONAL`` — same element described both operational and non-operational. * ``DUPLICATE`` — near-identical findings for the same element. * ``MUTUALLY_EXCLUSIVE`` — explicit antonym pairs on the same element. Resolution policy: the verifier never fabricates a merged truth. It keeps the finding with the stronger evidence support and records the conflict for audit. """ from __future__ import annotations import re from collections import defaultdict from app.extraction.citation_validator import normalize_for_match from app.extraction.schemas import ( Contradiction, ContradictionKind, ConditionRating, SupportLevel, SurveyFinding, ) # Lexical polarity cues. Kept conservative to avoid false positives. _DEFECTIVE_CUES = ( "defective", "defect", "failed", "failing", "broken", "cracked", "leak", "leaking", "damp", "rot", "rotten", "corroded", "deteriorated", "missing", "damaged", "unsafe", "not satisfactory", "poor condition", "in need of repair", ) _SATISFACTORY_CUES = ( "satisfactory", "good condition", "sound", "no defect", "no defects", "well maintained", "intact", "serviceable", "in good order", "no repair", ) _NON_OPERATIONAL_CUES = ( "not operational", "non-operational", "inoperable", "not working", "not functioning", "out of order", "does not work", "non operational", "not in working order", ) _OPERATIONAL_CUES = ( "operational", "working", "functioning", "in working order", "operates correctly", "fully functional", "in good working order", ) _WORD_RE = re.compile(r"[a-z0-9]+") def _element_key(f: SurveyFinding) -> str: """Normalized (section, element) key for grouping findings about one thing.""" return f"{normalize_for_match(f.section)}::{normalize_for_match(f.element)}" def _has_any(text: str, cues: tuple[str, ...]) -> bool: norm = normalize_for_match(text) return any(cue in norm for cue in cues) def _polarity(text: str) -> tuple[bool, bool, bool, bool]: """Return (defective, satisfactory, operational, non_operational) cue presence.""" return ( _has_any(text, _DEFECTIVE_CUES), _has_any(text, _SATISFACTORY_CUES), _has_any(text, _OPERATIONAL_CUES), _has_any(text, _NON_OPERATIONAL_CUES), ) def _jaccard(a: str, b: str) -> float: ta = set(_WORD_RE.findall(normalize_for_match(a))) tb = set(_WORD_RE.findall(normalize_for_match(b))) if not ta or not tb: return 0.0 return len(ta & tb) / len(ta | tb) def _support_rank(level: SupportLevel) -> int: return {SupportLevel.SUPPORTED: 2, SupportLevel.PARTIAL: 1, SupportLevel.NOT_FOUND: 0}[level] def _keep_stronger(a: tuple[int, SurveyFinding], b: tuple[int, SurveyFinding]) -> tuple[int, int]: """Return (keep_index, drop_index) preferring stronger evidence support.""" (ia, fa), (ib, fb) = a, b if _support_rank(fa.support) >= _support_rank(fb.support): return ia, ib return ib, ia def audit_contradictions( findings: list[SurveyFinding], *, duplicate_threshold: float = 0.9, ) -> tuple[list[SurveyFinding], list[Contradiction]]: """Detect contradictions/duplicates and return (resolved_findings, reports). Resolution keeps the evidence-stronger finding for hard contradictions and collapses near-duplicates. Every action is recorded in the returned :class:`Contradiction` list for the audit trail. The input list is not mutated; a filtered copy is returned. """ contradictions: list[Contradiction] = [] drop: set[int] = set() groups: dict[str, list[int]] = defaultdict(list) for i, f in enumerate(findings): groups[_element_key(f)].append(i) for _key, idxs in groups.items(): for pos_a in range(len(idxs)): for pos_b in range(pos_a + 1, len(idxs)): ia, ib = idxs[pos_a], idxs[pos_b] if ia in drop or ib in drop: continue fa, fb = findings[ia], findings[ib] # Rating conflict (both real ratings, different values). real = {ConditionRating.CR1, ConditionRating.CR2, ConditionRating.CR3} if ( fa.condition_rating in real and fb.condition_rating in real and fa.condition_rating != fb.condition_rating ): keep, dropped = _keep_stronger((ia, fa), (ib, fb)) drop.add(dropped) contradictions.append(Contradiction( kind=ContradictionKind.RATING_CONFLICT, element=fa.element, detail=( f"Condition rating {fa.condition_rating.value} vs " f"{fb.condition_rating.value} for the same element." ), finding_indices=[ia, ib], resolution=f"kept finding #{keep} (stronger evidence), dropped #{dropped}", )) continue da, sa, oa, na = _polarity(fa.finding) db, sb, ob, nb = _polarity(fb.finding) # Satisfactory vs defective. if (da and sb) or (sa and db): keep, dropped = _keep_stronger((ia, fa), (ib, fb)) drop.add(dropped) contradictions.append(Contradiction( kind=ContradictionKind.CONDITION, element=fa.element, detail="One finding describes the element as defective, the other as satisfactory.", finding_indices=[ia, ib], resolution=f"kept finding #{keep} (stronger evidence), dropped #{dropped}", )) continue # Operational vs non-operational. if (oa and nb) or (na and ob): keep, dropped = _keep_stronger((ia, fa), (ib, fb)) drop.add(dropped) contradictions.append(Contradiction( kind=ContradictionKind.OPERATIONAL, element=fa.element, detail="One finding states operational, the other non-operational.", finding_indices=[ia, ib], resolution=f"kept finding #{keep} (stronger evidence), dropped #{dropped}", )) continue # Near-duplicate. if _jaccard(fa.finding, fb.finding) >= duplicate_threshold: keep, dropped = _keep_stronger((ia, fa), (ib, fb)) drop.add(dropped) contradictions.append(Contradiction( kind=ContradictionKind.DUPLICATE, element=fa.element, detail="Near-identical findings for the same element.", finding_indices=[ia, ib], resolution=f"collapsed to finding #{keep}, dropped #{dropped}", )) resolved = [f for i, f in enumerate(findings) if i not in drop] return resolved, contradictions