RICS / app /extraction /domain_scope.py
StormShadow308's picture
Ship production RAG hardening: citation extraction, full-library retrieval, auth.
865bc90
Raw
History Blame Contribute Delete
7.93 kB
"""Section-domain scoping to prevent cross-section contamination (STEP 6).
Roofing prompts must only see roofing evidence; drainage only drainage, etc.
This module is a pure, deterministic keyword classifier — no embeddings, no
network — so the scoping decision is reproducible and unit-testable.
Usage::
domain = classify_section("Roofing", template_id="E1")
roofing_only = scope_chunks(domain, retrieved_chunks)
The classifier is intentionally conservative: a chunk with no clear domain
signal is treated as ``GENERAL`` and is admissible to any section (so we never
starve a section of context), but a chunk that clearly belongs to a *different*
domain is excluded.
"""
from __future__ import annotations
from app.extraction.citation_validator import normalize_for_match
GENERAL = "general"
# Domain → indicative terms. Order matters for deterministic tie-breaking
# (earlier domains win ties). Terms are matched as normalized substrings, so
# multi-word phrases are supported.
DOMAIN_LEXICON: dict[str, frozenset[str]] = {
"roofing": frozenset({
"roof", "roofing", "slate", "tile", "tiling", "ridge", "hip", "valley",
"verge", "eaves", "soffit", "fascia", "felt", "flashing", "parapet",
"covering", "rafter", "purlin", "truss", "flat roof", "pitched roof",
}),
"chimney": frozenset({
"chimney", "chimney stack", "flaunching", "pot", "cowl", "flue",
"breast", "corbel", "pointing to the stack",
}),
"rainwater": frozenset({
"gutter", "guttering", "downpipe", "rainwater", "hopper", "rwp",
"rainwater goods", "fall pipe",
}),
"drainage": frozenset({
"drain", "drainage", "sewer", "manhole", "inspection chamber", "gully",
"soil pipe", "foul", "surface water", "septic", "soakaway", "below ground",
}),
"walls": frozenset({
"wall", "masonry", "brickwork", "blockwork", "render", "rendering",
"pointing", "cavity", "spalling", "cracking", "lintel", "dpc",
"damp proof course", "external wall", "load bearing",
}),
"dampness": frozenset({
"damp", "dampness", "moisture", "rising damp", "penetrating damp",
"condensation", "mould", "mold", "hygroscopic", "salts", "tide mark",
}),
"timber": frozenset({
"timber", "woodworm", "beetle", "wet rot", "dry rot", "joist", "decay",
"fungal", "infestation", "rot to the",
}),
"windows_doors": frozenset({
"window", "windows", "door", "doors", "glazing", "double glazing",
"frame", "casement", "sash", "joinery", "fenestration", "sill",
}),
"ceilings_floors": frozenset({
"ceiling", "floor", "flooring", "floorboard", "screed", "plaster",
"lath", "cornice", "skirting", "subfloor",
}),
"electrical": frozenset({
"electric", "electrical", "wiring", "consumer unit", "fuse box",
"rcd", "socket", "circuit", "earthing", "eicr", "fixed wiring",
"distribution board",
}),
"heating": frozenset({
"boiler", "heating", "radiator", "central heating", "flue", "gas",
"thermostat", "hot water", "cylinder", "underfloor heating", "combi",
}),
"plumbing": frozenset({
"plumbing", "pipework", "water supply", "stopcock", "waste pipe",
"mains water", "lead pipe", "tank", "overflow", "sanitary",
}),
"insulation_energy": frozenset({
"insulation", "epc", "energy performance", "sap", "u-value",
"thermal", "loft insulation", "cavity insulation", "energy efficiency",
}),
"grounds": frozenset({
"garden", "boundary", "fence", "fencing", "patio", "driveway", "path",
"tree", "hedge", "retaining wall", "outbuilding", "grounds",
}),
"services_other": frozenset({
"ventilation", "extractor", "smoke alarm", "carbon monoxide",
"asbestos", "fire", "security",
}),
}
# Map common RICS section titles / template hints to a canonical domain.
_SECTION_ALIASES: dict[str, str] = {
"roof": "roofing", "roof coverings": "roofing", "main roof": "roofing",
"chimney stacks": "chimney", "chimneys": "chimney",
"rainwater pipes and gutters": "rainwater", "gutters": "rainwater",
"drainage": "drainage",
"main walls": "walls", "external walls": "walls", "walls": "walls",
"dampness": "dampness", "damp": "dampness",
"windows": "windows_doors", "doors": "windows_doors",
"ceilings": "ceilings_floors", "floors": "ceilings_floors",
"electricity": "electrical", "electrical": "electrical",
"heating": "heating", "gas": "heating",
"water": "plumbing", "plumbing": "plumbing",
"insulation": "insulation_energy", "energy efficiency": "insulation_energy",
"grounds": "grounds", "gardens": "grounds", "boundaries": "grounds",
}
def _score_domains(text: str) -> dict[str, int]:
norm = normalize_for_match(text)
if not norm:
return {}
scores: dict[str, int] = {}
for domain, terms in DOMAIN_LEXICON.items():
hits = 0
for term in terms:
# Count occurrences; multi-word terms matched as substrings.
if " " in term:
hits += norm.count(term)
else:
# word-ish boundary check to avoid 'gas' in 'gasket' etc.
hits += _count_word(norm, term)
if hits:
scores[domain] = hits
return scores
def _count_word(haystack: str, word: str) -> int:
count = 0
start = 0
n = len(word)
while True:
idx = haystack.find(word, start)
if idx == -1:
break
before = haystack[idx - 1] if idx > 0 else " "
after = haystack[idx + n] if idx + n < len(haystack) else " "
if not before.isalnum() and not after.isalnum():
count += 1
start = idx + n
return count
def classify_text(text: str) -> str:
"""Return the dominant domain for ``text`` or :data:`GENERAL`."""
scores = _score_domains(text)
if not scores:
return GENERAL
# Highest score wins; ties broken by lexicon insertion order (stable).
best = max(DOMAIN_LEXICON.keys(), key=lambda d: scores.get(d, 0))
return best if scores.get(best, 0) > 0 else GENERAL
def classify_section(section_name: str | None, template_id: str | None = None) -> str:
"""Resolve a section title/template into a canonical domain.
Falls back to keyword classification of the section name, then
:data:`GENERAL`.
"""
name = normalize_for_match(section_name or "")
if name in _SECTION_ALIASES:
return _SECTION_ALIASES[name]
for alias, domain in _SECTION_ALIASES.items():
if alias in name:
return domain
guessed = classify_text(section_name or "")
return guessed
def scope_chunks(
section_domain: str,
chunks: list[object],
*,
strict: bool = True,
) -> list[object]:
"""Filter retrieved chunks to those admissible for ``section_domain``.
A chunk is admissible when its dominant domain equals ``section_domain`` or
is :data:`GENERAL`. Chunks clearly belonging to a *different* domain are
excluded. If filtering would remove everything (e.g. weak classification),
the original list is returned to avoid starving extraction — unless
``strict`` is False, in which case the filtered (possibly empty) list is
always returned.
``chunks`` are SearchResult-like objects exposing ``.text``.
"""
if section_domain == GENERAL or not chunks:
return chunks
kept: list[object] = []
for c in chunks:
text = getattr(c, "text", "") or ""
cdom = classify_text(text)
if cdom == section_domain or cdom == GENERAL:
kept.append(c)
if not kept and strict:
# Never starve the extractor on a misclassification; better to extract
# from broader context than to silently drop the whole section.
return chunks
return kept