Rifqi Hafizuddin
[NOTICKET] feat(knowledge_extraction): paid extraction stage + validate, diff, queue
ab5ea78 | """Pydantic contracts for the knowledge-extraction pipeline. | |
| Three invariants are encoded here rather than described in prose, because every | |
| one of them is a control that a later change could quietly remove: | |
| 1. **All content fields are Optional.** A model that cannot answer null will | |
| fabricate one. Abstention is correct behaviour, never an error. | |
| 2. **`subdomain_tags` is an enum.** Classification, not generation. | |
| 3. **`Provenance.span` is mandatory and verbatim-checked.** It is the primary | |
| anti-hallucination control and the thing that makes expert review | |
| finishable β the reviewer checks a quote against a page, not a claim | |
| against their memory. | |
| `Chunk` here is the pipeline's **internal** unit, deliberately narrower than the | |
| parsed-document artifact being agreed with Sofhia (the seam). Stages depend only | |
| on this subset; `adapter.py` maps the seam type onto it, so seam churn lands in | |
| one file instead of seven. See KNOWLEDGE_PIPELINE_TODO.md Β§3. | |
| """ | |
| from __future__ import annotations | |
| from enum import Enum | |
| from typing import Literal | |
| from pydantic import BaseModel, Field | |
| Branch = Literal["glossary", "rule", "formula", "summary"] | |
| ExtractionStatus = Literal["ok", "no_definition_found", "escalated"] | |
| DiffStatus = Literal["new", "duplicate", "conflicting"] | |
| class SubdomainEnum(str, Enum): | |
| """Classification target. Extend deliberately β a new member changes what | |
| the model is allowed to answer, which is a prompt change, not a data one.""" | |
| production = "production" | |
| maintenance = "maintenance" | |
| hauling = "hauling" | |
| loading = "loading" | |
| drilling_blasting = "drilling_blasting" | |
| equipment = "equipment" | |
| safety = "safety" | |
| quality = "quality" | |
| planning = "planning" | |
| cost = "cost" | |
| geology = "geology" | |
| other = "other" | |
| # ββ Stage 1: the chunk (internal view of the seam artifact) βββββββββββββ | |
| class Chunk(BaseModel): | |
| """One unit of a parsed document, as the extraction stages need it. | |
| `text` must stay **verbatim** from the source document. Span validation | |
| locates LLM-quoted spans literally inside this text; if it is ever reflowed | |
| or whitespace-normalised the lookup fails and the field is silently set to | |
| null. The failure presents as a bad model, but the cause would be here. | |
| """ | |
| chunk_id: str | |
| doc_id: str | |
| text: str | |
| page_start: int | |
| page_end: int | |
| ordinal: int = 0 | |
| # Structural context. Both Optional β many documents carry no numbering. | |
| section_no: str | None = None | |
| heading: str | None = None | |
| # Cheap downstream filters / ranking signals | |
| has_formula: bool = False | |
| is_tabular: bool = False | |
| bold_spans: list[str] = Field(default_factory=list) | |
| class ParsedDoc(BaseModel): | |
| """A document's chunks plus the identity needed to version and cache them.""" | |
| doc_id: str | |
| source_ref: str | |
| content_hash: str | |
| n_pages: int | |
| chunks: list[Chunk] | |
| parser_name: str = "unknown" | |
| parser_version: str = "" | |
| used_heading_split: bool = False | |
| # ββ Stage 2: filters ββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class Mention(BaseModel): | |
| """One occurrence of a candidate term inside a chunk.""" | |
| surface: str | |
| chunk_id: str | |
| char_start: int | |
| char_end: int | |
| label: str = "" | |
| score: float = 0.0 | |
| hit_span_cap: bool = False | |
| class RuleCandidate(BaseModel): | |
| """A passage a discourse cue marks as possibly stating a rule of thumb.""" | |
| chunk_id: str | |
| cue: str | |
| char_start: int | |
| char_end: int | |
| snippet: str | |
| class AbbrevPair(BaseModel): | |
| """`PA` β `Physical Availability`, harvested from a legend block. | |
| Legend extraction must run before clustering: without these, an | |
| abbreviation and its expansion cluster as two unrelated terms. | |
| """ | |
| abbrev: str | |
| expansion: str | |
| chunk_id: str | |
| class FilterResult(BaseModel): | |
| doc_id: str | |
| mentions: list[Mention] = Field(default_factory=list) | |
| rule_candidates: list[RuleCandidate] = Field(default_factory=list) | |
| abbrev_pairs: list[AbbrevPair] = Field(default_factory=list) | |
| # ββ Stage 3: clusters βββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| class TermCluster(BaseModel): | |
| """All mentions of one term. **The LLM call unit is the cluster**, not the | |
| chunk and not the mention β that is what cuts expert review burden, and it | |
| is also the only reason conflicting definitions can be detected at all | |
| (they must arrive in the same call to be compared).""" | |
| cluster_id: str | |
| canonical: str | |
| variants: list[str] = Field(default_factory=list) | |
| mentions: list[Mention] = Field(default_factory=list) | |
| mention_count: int = 0 | |
| merge_reasons: list[str] = Field(default_factory=list) | |
| # Ranked best-first. The FULL list is kept, not just the top K β | |
| # escalation consumes the tail. | |
| evidence_chunk_ids: list[str] = Field(default_factory=list) | |
| evidence_scores: list[float] = Field(default_factory=list) | |
| class ClusterResult(BaseModel): | |
| doc_id: str | |
| clusters: list[TermCluster] = Field(default_factory=list) | |
| n_mentions: int = 0 | |
| n_clusters: int = 0 | |
| compression_ratio: float = 0.0 | |
| # ββ Stage 4+: extracted entries βββββββββββββββββββββββββββββββββββββββββ | |
| class Provenance(BaseModel): | |
| """Where a claim came from. `span` is mandatory and must appear verbatim in | |
| the evidence text; a field whose span cannot be located is rejected, never | |
| repaired. A repaired span is an unfalsifiable claim.""" | |
| doc_id: str | |
| span: str | |
| page: int | None = None | |
| section_no: str | None = None | |
| chunk_id: str | None = None | |
| class GlossaryEntry(BaseModel): | |
| term: str | |
| full_name: str | None = None | |
| # The literal wording as the document writes it, un-normalised. The BUMA | |
| # standard heads its section "Physical of Availability (PA)" while the | |
| # legend says "Physical Availability"; the discrepancy is surfaced to the | |
| # expert rather than silently corrected. | |
| source_wording: str | None = None | |
| definition: str | None = None | |
| formula_latex: str | None = None | |
| interpretation: str | None = None | |
| subdomain_tags: list[SubdomainEnum] = Field(default_factory=list) | |
| domain: str | None = None | |
| company: str | None = None | |
| language: str | None = None | |
| mention_count: int = 0 | |
| provenance: Provenance | |
| extraction_status: ExtractionStatus = "ok" | |
| diff_status: DiffStatus | None = None | |
| definition_conflict: bool = False | |
| conflict_variants: list[str] = Field(default_factory=list) | |
| class RuleEntry(BaseModel): | |
| """A rule of thumb / operational convention stated by the document.""" | |
| rule_id: str | |
| statement: str | None = None | |
| condition: str | None = None | |
| consequence: str | None = None | |
| applies_to: str | None = None | |
| subdomain_tags: list[SubdomainEnum] = Field(default_factory=list) | |
| language: str | None = None | |
| provenance: Provenance | |
| extraction_status: ExtractionStatus = "ok" | |
| class FormulaVariable(BaseModel): | |
| symbol: str | |
| meaning: str | None = None | |
| class FormulaEntry(BaseModel): | |
| name: str | None = None | |
| formula_latex: str | None = None | |
| variables: list[FormulaVariable] = Field(default_factory=list) | |
| unit: str | None = None | |
| provenance: Provenance | |
| extraction_status: ExtractionStatus = "ok" | |
| class BriefContext(BaseModel): | |
| """Whole-document summary. The only branch that cannot be span-checked β | |
| a plausible summary is indistinguishable from a correct one, which is why | |
| it belongs on the larger model tier when one is available.""" | |
| title: str | None = None | |
| purpose: str | None = None | |
| scope: str | None = None | |
| key_parameters: list[str] = Field(default_factory=list) | |
| summary_md: str | None = None | |
| provenance: Provenance | |
| class CallUsage(BaseModel): | |
| """Per-call accounting. `cached_tokens` comes from the API and is never | |
| modelled: caching does not engage below 1024 prompt tokens, so assuming it | |
| would understate cost by ~10x on the input side.""" | |
| branch: Branch | |
| deployment: str | |
| tier: str = "nano" | |
| prompt_tokens: int = 0 | |
| cached_tokens: int = 0 | |
| completion_tokens: int = 0 | |
| latency_s: float = 0.0 | |
| retries: int = 0 | |
| structured_output_mode: str = "" | |
| simulated: bool = False | |
| class RejectedField(BaseModel): | |
| """Audit row for a field the span check refused. Kept so a reviewer can see | |
| what the control caught rather than only what it let through.""" | |
| entry_term: str | |
| field: str | |
| offending_value: str | |
| reason: str | |
| branch: Branch | |