"""Tunables for the knowledge-extraction pipeline. Every value here was calibrated on real documents and each one has a reason recorded in KNOWLEDGE_PIPELINE_CALIBRATION.md. Change them deliberately — most were arrived at by a measurement, and two of them (`FUZZY_MIN_LEN`, `CACHE_MIN_TOKENS`) fix bugs that are silent when reintroduced. Label and cue sets live in `config/*.yaml` so they can be tuned without a code change: label phrasing is the main recall lever and the filter is very sensitive to it. """ from __future__ import annotations from functools import lru_cache from pathlib import Path import yaml CONFIG_DIR = Path(__file__).resolve().parent / "config" # ── Term filter ───────────────────────────────────────────────────────── # Variant C beat both the English-default and Indonesian-phrasing label sets: # the other two missed the same class (mining activities and materials). LABELS_VARIANT = "broad" # 0.25, not 0.35: measured recall 0.854 @ 0.25 vs 0.658 @ 0.35 on `broad`. # Precision falls (0.41 vs 0.50) and that is the intended trade — the filter is # deliberately over-inclusive, clustering and ranking absorb the noise, and a # term the filter never proposes can never be recovered downstream. SPAN_SCORE_THRESHOLD = 0.25 # The span NER model truncates past ~384 of its own tokens and *warns rather # than failing*, so a long chunk silently loses its tail. Indonesian technical # prose subword-tokenises at roughly 2.5x, so 220-word windows still tripped the # cap; 130 does not. Chunks are fed as overlapping windows with offsets remapped. WINDOW_WORDS = 130 WINDOW_OVERLAP = 30 SPAN_TOKEN_CAP = 12 # ── Clustering ────────────────────────────────────────────────────────── FUZZY_THRESHOLD = 92 # Below this length only exact matching is allowed: "PA" and "UA" score highly # against each other on token_set_ratio. Over-merging is far worse than # under-merging — an under-merge costs one extra call and one extra review row, # a wrong merge destroys a distinct term and the expert never sees it. FUZZY_MIN_LEN = 5 # ── Evidence ranking ──────────────────────────────────────────────────── EVIDENCE_K = 3 CUE_PROXIMITY_CHARS = 100 EVIDENCE_WEIGHTS: dict[str, float] = { "definitional_cue_near": 5.0, "term_in_heading": 4.0, "in_legend_block": 3.5, "formula_present": 2.0, "bold_or_italic": 1.5, "first_occurrence": 1.0, "tabular_penalty": -3.0, } # ── Chunking ──────────────────────────────────────────────────────────── MAX_CHUNK_TOKENS = 1500 MAX_HEADING_LEN = 90 BOILERPLATE_MIN_FRAC = 0.6 # ── Extraction ────────────────────────────────────────────────────────── TEMPERATURE = 0.0 # OpenAI-family prompt caching does not engage AT ALL below this many prompt # tokens, so a shorter fixed prefix caches nothing and costs ~10x on input. The # measured hit rate at/above it was 54%. CACHE_MIN_TOKENS = 1024 # ── Validation ────────────────────────────────────────────────────────── MAX_ESCALATION_ROUNDS = 2 CONFLICT_OVERLAP_THRESHOLD = 0.4 DUPLICATE_OVERLAP_THRESHOLD = 0.8 @lru_cache(maxsize=4) def load_yaml(name: str) -> dict: with open(CONFIG_DIR / name, encoding="utf-8") as fh: return yaml.safe_load(fh) def labels_for(variant: str = LABELS_VARIANT) -> tuple[list[str], float]: """Returns (labels, threshold) for a label variant.""" cfg = load_yaml("labels.yaml") labels = cfg.get(variant) or cfg.get(LABELS_VARIANT) or [] return list(labels), float(cfg.get("threshold", SPAN_SCORE_THRESHOLD))