File size: 3,663 Bytes
eea689d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
"""The hosted-model adjunct to the B1 validation pass (#97).

The deterministic checks in `validation.py` (type, enum, cardinality, derivation recompute) run
with no model call. This module adds the one judgment the deterministic pass cannot make on its own:
whether the extracted evidence span actually supports the extracted value. It asks the validation
model to adjudicate one field through a forced tool call, the same schema-constrained pattern
`llm_extraction.py` uses, so the model returns a structured verdict rather than free text.

The model is configured explicitly here (`VALIDATION_MODEL`), not inherited from the extraction
model: the extraction and the validation of that extraction are deliberately separate calls. The
fast test suite mocks the client, so no API key is needed; the live call is exercised behind a
`@pytest.mark.slow` test.
"""

from __future__ import annotations

from typing import Optional

import anthropic

from endopath.llm_extraction import get_client

# The validation model, set explicitly (claude-api skill: default to Claude Opus 4.8). Kept
# distinct from llm_extraction.MODEL so the validation pass and the extraction it validates are
# separately configurable.
VALIDATION_MODEL = "claude-sonnet-5"

# The forced tool the model must call: a structured verdict on one field, never free text.
FIELD_REVIEW_TOOL = {
    "name": "record_field_review",
    "description": (
        "Record whether the extracted evidence span supports the extracted value for one "
        "checklist concept."
    ),
    "input_schema": {
        "type": "object",
        "properties": {
            "agrees": {
                "type": "boolean",
                "description": "True if the evidence span supports the extracted value.",
            },
            "note": {
                "type": "string",
                "description": "A one-sentence reason, citing the evidence where it disagrees.",
            },
        },
        "required": ["agrees", "note"],
    },
}

_PROMPT_TEMPLATE = """\
You are validating one field of a structured extraction from a pathology report against the \
evidence that justified it. The deterministic checks (type, enum, cardinality, and the derivation \
recompute) have already run; your job is the judgment they cannot make: does the cited evidence \
actually support this value?

Call record_field_review with your verdict.

concept: {concept_id}
extracted value: {value!r}
evidence span: {evidence!r}
"""


def review_field(
    client: Optional[anthropic.Anthropic],
    *,
    concept_id: str,
    value: object,
    evidence: Optional[str] = None,
    model: str = VALIDATION_MODEL,
) -> dict:
    """Ask the validation model whether the evidence supports the value. Returns the tool input
    (`{"agrees": bool, "note": str}`). Raises on a truncated tool call, the same failure
    `llm_extraction.extract_raw` guards against."""
    client = client or get_client()
    response = client.messages.create(
        model=model,
        max_tokens=1024,
        tools=[FIELD_REVIEW_TOOL],
        tool_choice={"type": "tool", "name": "record_field_review"},
        messages=[
            {
                "role": "user",
                "content": _PROMPT_TEMPLATE.format(
                    concept_id=concept_id, value=value, evidence=evidence
                ),
            }
        ],
    )
    if response.stop_reason == "max_tokens":
        raise RuntimeError(
            "field review hit max_tokens before completing the tool call; raise max_tokens."
        )
    tool_call = next(block for block in response.content if block.type == "tool_use")
    return tool_call.input