Spaces:
Runtime error
Runtime error
| """LLM + regex sanitisation before tenant uploads are indexed for RAG / style learning. | |
| On-disk uploads are unchanged; only text written to the vector index is sanitised. | |
| When the v2 backend is importable, deterministic redaction uses | |
| ``backend.core.pii_scrubber`` (dual-pass regex + PERSON NER + PropTech whitelist) | |
| as a mandatory post-pass after the LLM tier. The LLM prompt teaches descriptive | |
| intent and cites real failure-mode anti-patterns so survey vocabulary survives. | |
| """ | |
| from __future__ import annotations | |
| import logging | |
| import re | |
| from enum import Enum | |
| from typing import TYPE_CHECKING | |
| from app.config import settings | |
| if TYPE_CHECKING: | |
| from langchain_core.documents import Document | |
| logger = logging.getLogger(__name__) | |
| class DocumentSanitisationError(RuntimeError): | |
| """Raised when sanitisation is required but cannot complete safely.""" | |
| class RedactionStrategy(Enum): | |
| """Selects which redaction engine ``sanitise_text_for_rag_sync`` routes to. | |
| ``AI_HYBRID`` runs the existing LLM + regex/NER pipeline (unchanged). | |
| ``DETERMINISTIC_CODE`` runs a self-contained regex + flashtext pass with no | |
| LLM, NER model, or network I/O — suitable for offline / air-gapped redaction. | |
| """ | |
| AI_HYBRID = "ai_hybrid" | |
| DETERMINISTIC_CODE = "deterministic" | |
| RAG_UPLOAD_SANITISATION_SYSTEM_PROMPT = """\ | |
| You are a zero-bleed Data Privacy Compliance Engine for UK PropTech and RICS Home \ | |
| Survey Level 3 documentation. Purge unique identifiers belonging to individuals, \ | |
| corporate entities, transactions, or exact physical locations. Keep 100% intact the \ | |
| technical, architectural, structural, and pathological diagnostics — over-redaction \ | |
| that strips property-composition vocabulary is a critical failure for downstream RAG \ | |
| style learning. | |
| ### THE DESCRIPTIVE INTENT RULE (apply before removing any token) | |
| If a word or phrase describes HOW a property is built, WHERE a defect is physically \ | |
| located, or WHAT structural condition it is in, it is EXEMPT. Only remove a token if \ | |
| it identifies a single unique transaction, client, or specific property plot. | |
| ### WHITELIST — never remove, alter, summarise, or paraphrase | |
| Leave these categories exactly as written (enforcement also runs in a deterministic \ | |
| post-pass with the same domain lexicon): | |
| - Spatial orientations & locations: front, rear, left, right, elevation, side, landing, \ | |
| bedroom, kitchen, loft, eaves, ground floor, upper level, apex, perimeter, boundary. | |
| - Materials & components: timber, purlins, brickwork, lead, flashing, uPVC, slate, felt, \ | |
| Velux, concrete, downpipe, cladding, glazing, etc. | |
| - Pathology & condition: condition rating, defect, cracking, damp, spalled, deflection, \ | |
| moisture, insulation, rot, water ingress, settlement, distortion, etc. | |
| - Regulatory frameworks (generic names only): Building Regulations, Local Authority, \ | |
| FENSA, Gas Safe, RICS, Home Survey Level 3, Environment Agency, Council Tax Band, \ | |
| Party Wall etc. Act 1996, lease, freeholder, covenants, obligations. | |
| - Generic historical/statutory era references (e.g. "alterations between the 1960s and \ | |
| 1980s") — not transaction-specific calendar dates. | |
| Remove only when a regulatory or certification name is paired with a specific licence \ | |
| number, account number, or job reference tied to an identifiable matter. | |
| ### ANTI-PATTERNS — do not repeat these systemic errors | |
| - Do NOT redact "timber", "asbestos", or "combination" inside structural phrases \ | |
| (e.g. "timber purlins", "Asbestos Containing Materials (ACMs)", "combination boiler"). | |
| - Do NOT redact positional descriptions (e.g. "left side of the living room", \ | |
| "rear right bedroom", "front elevation"). | |
| - Do NOT redact standalone certification scheme labels (ELECSA, NICEIC, FENSA) — but \ | |
| DO remove postcode-like or numeric middle segments inside reference strings \ | |
| (e.g. in "Ref. No. 22/SW1A1AA/ELECSA", remove the middle segment only; a downstream \ | |
| deterministic pass handles rigid ID patterns). | |
| ### BLACKLIST — surgically remove the entire span | |
| Delete the sensitive text cleanly. Do not insert placeholders like "[REDACTED]" — \ | |
| remove the span and collapse surrounding whitespace naturally. A downstream \ | |
| deterministic pass may apply further token-level redaction after your output. | |
| - Personal identifiers: client, occupier, vendor, purchaser, solicitor, agent, \ | |
| surveyor, witness, contractor names, initials, signatures. | |
| - Contact vectors: email, phone, fax, identifying URLs, portal credentials. | |
| - Property locations: house/flat numbers, property names, street names, postcodes, \ | |
| correspondence addresses. Remove address fragments; do not attempt to keep or drop \ | |
| town/city names by judgment — remove only explicit pinpointing fragments listed here. | |
| - Registry & file IDs: Land Registry title numbers, UPRNs, EPC certificate numbers, \ | |
| planning application numbers, council tax account numbers, job/file/invoice serials, \ | |
| National Insurance numbers, bank details, tribunal/court/lease reference numbers. | |
| - Financial metrics: valuations, prices, ground rent, service charges, mortgage figures. | |
| - Transaction dates: exact inspection, report, exchange, or completion dates for this job. | |
| - Special category: health, disability, family circumstances, non-property litigation, \ | |
| complaints. | |
| ### OUTPUT | |
| - Inline text parser only. Retain paragraph breaks, capitalization, headers, and \ | |
| section order. | |
| - If a sentence contains no blacklist items, output it verbatim. | |
| - Do not summarise, compress, paraphrase, or invent information. | |
| - No markdown fences, no JSON, no commentary. Output ONLY the sanitised excerpt text.""" | |
| _POSTCODE_RE = re.compile( | |
| r"\b([A-Z]{1,2}\d[A-Z\d]?\s*\d[A-Z]{2})\b", re.IGNORECASE | |
| ) | |
| _ADDRESS_LINE_RE = re.compile( | |
| r"\b(\d{1,4}\s+[A-Za-z][A-Za-z'\-]*(?:\s+[A-Za-z][A-Za-z'\-]*){0,6}\s+" | |
| r"(Road|Rd|Street|St|Avenue|Ave|Lane|Ln|Drive|Dr|Crescent|Close|Place|Way|Gardens|Gdns|Court|Ct|Terrace|Terr))\b", | |
| re.IGNORECASE, | |
| ) | |
| _MONEY_RE = re.compile( | |
| r"(£\s*\d[\d,]*(?:\.\d+)?|\b\d[\d,]*(?:\.\d+)?\s*(?:gbp|pounds)\b)", | |
| re.IGNORECASE, | |
| ) | |
| _DATE_RE = re.compile( | |
| r"\b(?:\d{1,2}[/-]\d{1,2}[/-]\d{2,4}|\d{1,2}\s+" | |
| r"(?:jan|feb|mar|apr|may|jun|jul|aug|sep|sept|oct|nov|dec)[a-z]*\s+\d{2,4})\b", | |
| re.IGNORECASE, | |
| ) | |
| _LONG_NUMBER_RE = re.compile(r"\b\d{6,}\b") | |
| _EMAIL_RE = re.compile(r"\b[\w.+-]+@[\w.-]+\.[A-Za-z]{2,}\b") | |
| _PHONE_RE = re.compile( | |
| r"\b(?:\+44\s?|0)(?:\d[\s-]?){9,12}\b|\b\d{3,4}[\s-]\d{3,4}[\s-]\d{3,4}\b" | |
| ) | |
| _URL_RE = re.compile(r"https?://[^\s<>\"']+", re.IGNORECASE) | |
| _NINO_RE = re.compile( | |
| r"\b[A-CEGHJ-PR-TW-Z]{2}\s?\d{2}\s?\d{2}\s?\d{2}\s?[A-D]\b", re.IGNORECASE | |
| ) | |
| _TITLE_NUMBER_RE = re.compile(r"\b[A-Z]{1,3}\d{5,6}\b") | |
| _REF_ID_RE = re.compile(r"\b[A-Z]{2,}[\-/][A-Z0-9][\-/A-Z0-9]{2,}\b", re.IGNORECASE) | |
| # ── Deterministic strategy — compiled once at module load (zero-AI/zero-network) ─ | |
| # Used only by ``_deterministic_redact``. Kept boundary-anchored so they never | |
| # swallow adjacent survey vocabulary. | |
| _REDACT_PLACEHOLDER = "[REDACTED]" | |
| _DET_RE_POSTCODE = re.compile(r"\b[A-Z]{1,2}[0-9][A-Z0-9]?\s?[0-9][A-Z]{2}\b", re.IGNORECASE) | |
| _DET_RE_EMAIL = re.compile(r"[\w.\-]+@[\w.\-]+\.\w+") | |
| _DET_RE_PHONE_UK = re.compile( | |
| r"(?:(?:\+44\s?|0)(?:7\d{3}|\d{2,4})[\s\-]?\d{3,4}[\s\-]?\d{3,4})" | |
| ) | |
| # Applied in order: most-specific (email) → phone → postcode. | |
| _DET_REDACTORS: tuple[re.Pattern[str], ...] = ( | |
| _DET_RE_EMAIL, | |
| _DET_RE_PHONE_UK, | |
| _DET_RE_POSTCODE, | |
| ) | |
| def should_sanitise_for_rag(tenant_id: str) -> bool: | |
| """Whether uploads for this tenant should be sanitised before vector indexing.""" | |
| if not settings.enable_rag_upload_sanitisation: | |
| return False | |
| if ( | |
| settings.rag_sanitisation_skip_kb_tenant | |
| and tenant_id == settings.knowledge_base_tenant_id | |
| ): | |
| return False | |
| return True | |
| def _v2_scrub(text: str) -> str | None: | |
| """Use the v2 dual-pass scrubber when the backend package is importable.""" | |
| try: | |
| from backend.core import pii_scrubber | |
| return pii_scrubber.scrub(text).text | |
| except ImportError: | |
| return None | |
| def regex_sanitise_text(text: str) -> str: | |
| """Deterministic redaction when the LLM path is unavailable or as a safety net.""" | |
| v2 = _v2_scrub(text) | |
| if v2 is not None: | |
| return v2 | |
| t = (text or "").strip() | |
| if not t: | |
| return "" | |
| t = _EMAIL_RE.sub("", t) | |
| t = _URL_RE.sub("", t) | |
| t = _PHONE_RE.sub("", t) | |
| t = _POSTCODE_RE.sub("", t) | |
| t = _ADDRESS_LINE_RE.sub("", t) | |
| t = _MONEY_RE.sub("", t) | |
| t = _DATE_RE.sub("", t) | |
| t = _NINO_RE.sub("", t) | |
| t = _TITLE_NUMBER_RE.sub("", t) | |
| t = _REF_ID_RE.sub("", t) | |
| t = _LONG_NUMBER_RE.sub("", t) | |
| t = re.sub(r"[ \t]{2,}", " ", t) | |
| t = re.sub(r"\n{3,}", "\n\n", t) | |
| return t.strip() | |
| def _deterministic_post_pass(text: str) -> str: | |
| """Second-pass deterministic scrub after LLM output (whitelist-aware when v2 available).""" | |
| cleaned = regex_sanitise_text(text) | |
| return cleaned.strip() | |
| def _deterministic_redact( | |
| text: str, | |
| db_context: dict[str, str] | None = None, | |
| ) -> str: | |
| """Zero-AI, zero-network text redaction (compiled regex + exact-string wipe). | |
| Runs entirely in-process: no LLM, no NER model, no network I/O. Stages, in | |
| order, mirror the deterministic redaction spec: | |
| 1. **Metadata purge** — not applicable at the text layer. The original PDF | |
| bytes never reach this function (PDFs are extracted to text upstream and | |
| on-disk uploads are left intact by design), so this stage is a documented | |
| no-op here rather than a ``fitz`` metadata wipe. | |
| 2. **Compiled-regex substitution** — UK postcodes, emails, and phone numbers | |
| are replaced with ``[REDACTED]`` using module-level patterns compiled at | |
| import time (never inside this function). | |
| 3. **Context-driven exact-string wipe** — when ``db_context`` is non-empty, | |
| its values (session-known PII such as client name / address) are removed | |
| case-insensitively via :class:`flashtext.KeywordProcessor`. | |
| Args: | |
| text: Extracted document text to sanitise. Must be a non-empty string. | |
| db_context: Optional ``{label: pii_value}`` map of exact sensitive strings | |
| to wipe. ``None`` or empty skips stage 3 gracefully. | |
| Returns: | |
| The sanitised text with all matches replaced by ``[REDACTED]``. | |
| Raises: | |
| ValueError: If ``text`` is not a non-empty string. | |
| """ | |
| if not isinstance(text, str) or not text.strip(): | |
| raise ValueError("_deterministic_redact requires non-empty text input") | |
| # Stage 2 — compiled-regex substitution. | |
| cleaned = text | |
| for pattern in _DET_REDACTORS: | |
| cleaned = pattern.sub(_REDACT_PLACEHOLDER, cleaned) | |
| # Stage 3 — exact-string context wipe (only when context is supplied). | |
| if db_context: | |
| try: | |
| from flashtext import KeywordProcessor | |
| except ImportError: | |
| logger.warning( | |
| "flashtext not installed; skipping deterministic context wipe " | |
| "(add flashtext>=2.7 to enable stage 3)." | |
| ) | |
| else: | |
| kp = KeywordProcessor(case_sensitive=False) | |
| for value in db_context.values(): | |
| if value and value.strip(): | |
| kp.add_keyword(value.strip(), _REDACT_PLACEHOLDER) | |
| if kp.get_all_keywords(): | |
| cleaned = kp.replace_keywords(cleaned) | |
| cleaned = re.sub(r"[ \t]{2,}", " ", cleaned) | |
| cleaned = re.sub(r"\n{3,}", "\n\n", cleaned) | |
| return cleaned.strip() | |
| def _split_text_chunks(text: str, max_chars: int) -> list[str]: | |
| body = (text or "").strip() | |
| if not body: | |
| return [] | |
| if len(body) <= max_chars: | |
| return [body] | |
| parts = re.split(r"(\n\s*\n)", body) | |
| chunks: list[str] = [] | |
| current = "" | |
| for part in parts: | |
| if not part: | |
| continue | |
| candidate = current + part | |
| if len(candidate) <= max_chars: | |
| current = candidate | |
| continue | |
| if current.strip(): | |
| chunks.append(current.strip()) | |
| if len(part) <= max_chars: | |
| current = part | |
| else: | |
| for i in range(0, len(part), max_chars): | |
| segment = part[i : i + max_chars].strip() | |
| if segment: | |
| chunks.append(segment) | |
| current = "" | |
| if current.strip(): | |
| chunks.append(current.strip()) | |
| return chunks or [body[:max_chars]] | |
| def _llm_sanitise_chunk_sync(chunk: str, *, part: int, total: int) -> str: | |
| from app.llm.openai_chat import chat_completions_create | |
| import asyncio | |
| prefix = ( | |
| f"Sanitise this RICS Home Survey Level 3 excerpt (part {part} of {total}). " | |
| f"Apply the Descriptive Intent rule: preserve all structural survey vocabulary; " | |
| f"remove only unique identifiers.\n\n" | |
| ) | |
| user_content = prefix + chunk | |
| async def _run() -> str: | |
| return await chat_completions_create( | |
| messages=[ | |
| {"role": "system", "content": RAG_UPLOAD_SANITISATION_SYSTEM_PROMPT}, | |
| {"role": "user", "content": user_content}, | |
| ], | |
| model=settings.chat_model, | |
| max_tokens=int(settings.rag_sanitisation_max_output_tokens), | |
| temperature=0.0, | |
| phase="rag_upload_sanitise", | |
| ) | |
| try: | |
| return asyncio.run(_run()) | |
| except RuntimeError: | |
| import concurrent.futures | |
| with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool: | |
| return pool.submit(asyncio.run, _run()).result() | |
| def sanitise_text_for_rag_sync( | |
| text: str, | |
| *, | |
| tenant_id: str, | |
| force: bool = False, | |
| strategy: RedactionStrategy = RedactionStrategy.AI_HYBRID, | |
| db_context: dict[str, str] | None = None, | |
| ) -> str: | |
| """Sanitise plain text before it is embedded into the tenant vector index. | |
| ``force=True`` bypasses the per-tenant policy switch. Use this for | |
| ``style_corpus`` uploads — the user's past completed reports MUST be | |
| PII-scrubbed before indexing because the same content gets reused across | |
| every new report (a missed name/postcode would leak across jobs). | |
| Args: | |
| text: Raw extracted document text to sanitise. | |
| tenant_id: Tenant whose sanitisation policy governs the AI path. | |
| force: Bypass the per-tenant policy switch (AI path only). | |
| strategy: Redaction engine to use. ``AI_HYBRID`` (default) preserves the | |
| existing LLM + regex/NER pipeline exactly. ``DETERMINISTIC_CODE`` | |
| runs the zero-AI, zero-network regex + flashtext pass and is honoured | |
| regardless of the per-tenant policy switch (explicit caller request). | |
| db_context: Optional ``{label: pii_value}`` map of exact sensitive strings | |
| forwarded to the deterministic path's context wipe. Ignored by the | |
| AI path. | |
| Returns: | |
| The sanitised text (empty string for empty input). | |
| """ | |
| raw = (text or "").strip() | |
| if not raw: | |
| return "" | |
| # Explicit deterministic request: run in-process, no AI/network, no policy gate. | |
| if strategy is RedactionStrategy.DETERMINISTIC_CODE: | |
| return _deterministic_redact(raw, db_context=db_context) | |
| if not force and not should_sanitise_for_rag(tenant_id): | |
| return raw | |
| max_chars = int(settings.rag_sanitisation_chunk_chars) | |
| chunks = _split_text_chunks(raw, max_chars) | |
| has_key = bool((settings.openai_api_key or "").strip()) | |
| out_parts: list[str] = [] | |
| for i, chunk in enumerate(chunks, start=1): | |
| cleaned = "" | |
| if has_key: | |
| try: | |
| llm_out = ( | |
| _llm_sanitise_chunk_sync(chunk, part=i, total=len(chunks)) or "" | |
| ).strip() | |
| if llm_out: | |
| cleaned = _deterministic_post_pass(llm_out) | |
| except Exception as exc: | |
| logger.warning( | |
| "RAG sanitisation LLM failed tenant=%s part=%d/%d: %s", | |
| tenant_id, | |
| i, | |
| len(chunks), | |
| exc, | |
| ) | |
| if not cleaned: | |
| cleaned = regex_sanitise_text(chunk) | |
| if has_key: | |
| logger.info( | |
| "RAG sanitisation using regex fallback tenant=%s part=%d/%d", | |
| tenant_id, | |
| i, | |
| len(chunks), | |
| ) | |
| if not cleaned and settings.rag_sanitisation_fail_closed: | |
| raise DocumentSanitisationError( | |
| f"Sanitisation produced empty text for tenant={tenant_id} part={i}/{len(chunks)}" | |
| ) | |
| out_parts.append(cleaned) | |
| combined = "\n\n".join(p for p in out_parts if p).strip() | |
| if not combined and settings.rag_sanitisation_fail_closed: | |
| raise DocumentSanitisationError( | |
| f"Sanitisation produced empty document for tenant={tenant_id}" | |
| ) | |
| return combined or regex_sanitise_text(raw) | |
| def sanitise_langchain_documents_sync( | |
| documents: list[Document], | |
| *, | |
| tenant_id: str, | |
| force: bool = False, | |
| strategy: RedactionStrategy = RedactionStrategy.AI_HYBRID, | |
| db_context: dict[str, str] | None = None, | |
| ) -> list[Document]: | |
| """Return documents whose ``page_content`` has been sanitised for RAG indexing. | |
| Pass ``force=True`` for ``style_corpus`` uploads so PII redaction runs | |
| regardless of the tenant's global sanitisation policy. ``strategy`` and | |
| ``db_context`` are forwarded per-document to :func:`sanitise_text_for_rag_sync`; | |
| the ``DETERMINISTIC_CODE`` strategy always runs regardless of tenant policy. | |
| """ | |
| if ( | |
| strategy is RedactionStrategy.AI_HYBRID | |
| and not force | |
| and not should_sanitise_for_rag(tenant_id) | |
| ): | |
| return documents | |
| from langchain_core.documents import Document as LCDocument | |
| out: list[LCDocument] = [] | |
| for doc in documents: | |
| meta = dict(doc.metadata or {}) | |
| cleaned = sanitise_text_for_rag_sync( | |
| doc.page_content or "", | |
| tenant_id=tenant_id, | |
| force=force, | |
| strategy=strategy, | |
| db_context=db_context, | |
| ) | |
| meta["rag_sanitised"] = True | |
| out.append(LCDocument(page_content=cleaned, metadata=meta)) | |
| return out | |