| """Multi-modal consensus and disagreement gating between the LLM |
| extraction pass (llm_extraction.py, issue #7) and the ColPali visual |
| retrieval pass (colpali_retrieval.py, issue #8), per docs/prd.md section 7: |
| "Requiring the LLM extraction and an embedding-based check to agree |
| lowers the chance of a hallucinated value passing through unexamined." |
| |
| What "agreement" means here, given the current data model: the LLM pass |
| produces a value with char-offset text evidence; ColPali's visual pass |
| produces no value at all, only a page-level match for a field's query |
| (sub-page localization is not built yet: #49, docs/prd.md section 4.6 -- |
| so there is no character-level location to cross-check the LLM's evidence |
| offsets against). So this gate cannot ask "do both signals point to the same |
| value" or "the same location", only "did each signal find something at |
| all". That is a real, acknowledged limitation, not an oversight -- docs/prd.md |
| names this gate's blind spot directly: "It does not catch errors where |
| both fail for the same underlying reason, most likely on the worst |
| physically degraded scans... That specific case still needs a human, not |
| a second model vote." A field where the LLM confidently extracts a wrong |
| value AND ColPali confidently (wrongly) retrieves a page that happens to |
| look relevant will still gate as agree/high-confidence here. This module |
| narrows the more common failure -- one signal hallucinating while the |
| other finds nothing -- it does not and cannot catch a correlated failure. |
| """ |
|
|
| from __future__ import annotations |
|
|
| from enum import Enum |
| from typing import Any, Optional |
|
|
| from endopath.schema import ChecklistField, FieldStatus |
|
|
| |
| |
| |
| |
| |
| |
| DEFAULT_VISUAL_MARGIN_THRESHOLD = 1.0 |
|
|
| |
| |
| |
| AGREEMENT_CONFIDENCE_FLOOR = 0.85 |
|
|
| |
| |
| |
| |
| DISAGREEMENT_CONFIDENCE_CEILING = 0.2 |
|
|
|
|
| class ConsensusOutcome(str, Enum): |
| AGREE = "agree" |
| DISAGREE = "disagree" |
| BOTH_EMPTY = "both_empty" |
|
|
|
|
| def gate( |
| llm_value: Any, |
| llm_confidence: Optional[float], |
| visual_margin: Optional[float], |
| margin_threshold: float = DEFAULT_VISUAL_MARGIN_THRESHOLD, |
| ) -> tuple[ConsensusOutcome, Optional[float]]: |
| """Compares one field's LLM result against its visual-retrieval margin. |
| |
| Returns (outcome, confidence). confidence is None only for BOTH_EMPTY, |
| where there is nothing for a confidence score to describe -- that's |
| the not-found path (issue #10's job to escalate further), not a low |
| number. |
| """ |
| llm_found = llm_value is not None |
| visual_found = visual_margin is not None and visual_margin >= margin_threshold |
|
|
| if llm_found and visual_found: |
| base = llm_confidence if llm_confidence is not None else AGREEMENT_CONFIDENCE_FLOOR |
| return ConsensusOutcome.AGREE, max(base, AGREEMENT_CONFIDENCE_FLOOR) |
|
|
| if not llm_found and not visual_found: |
| return ConsensusOutcome.BOTH_EMPTY, None |
|
|
| |
| return ConsensusOutcome.DISAGREE, DISAGREEMENT_CONFIDENCE_CEILING |
|
|
|
|
| def apply_consensus_gate( |
| field: ChecklistField, |
| visual_margin: Optional[float], |
| margin_threshold: float = DEFAULT_VISUAL_MARGIN_THRESHOLD, |
| ) -> ConsensusOutcome: |
| """Runs the gate against one already-extracted ChecklistField and |
| mutates its confidence/search_log in place. Never sets status to |
| CONFIRMED -- every field still gets a human look regardless of |
| agreement (llm_extraction.py's own framing: "Not autonomous"); this |
| only changes how urgently it's flagged for review. Never sets FLAGGED |
| either -- that's issue #10's escalation state, earned only after the |
| full LLM/embedding/combined search comes back empty, not after a |
| single paired comparison. |
| """ |
| outcome, confidence = gate(field.value, field.confidence, visual_margin, margin_threshold) |
| field.confidence = confidence |
| field.status = FieldStatus.NEEDS_REVIEW |
| field.search_log.append(f"consensus gate: {outcome.value}") |
| return outcome |
|
|