File size: 18,352 Bytes
e561e67
 
 
aad7814
 
 
 
 
e561e67
 
 
 
 
 
aad7814
e561e67
 
 
 
 
 
 
 
 
 
 
 
 
 
aad7814
 
 
 
 
 
 
 
 
 
 
 
e561e67
aad7814
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561e67
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
aad7814
e561e67
 
 
 
 
aad7814
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561e67
 
 
 
 
 
 
 
 
 
 
 
 
 
aad7814
 
 
 
 
 
 
 
 
 
e561e67
 
aad7814
 
 
 
e561e67
 
 
 
 
 
 
 
 
 
aad7814
 
 
e561e67
 
 
 
 
 
aad7814
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561e67
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
aad7814
 
 
e561e67
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
aad7814
 
 
 
 
 
 
 
c893230
 
 
 
 
 
e561e67
aad7814
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561e67
 
 
 
aad7814
 
 
 
 
 
 
e561e67
 
 
 
 
 
 
 
 
aad7814
 
 
 
 
e561e67
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
c893230
aad7814
 
e561e67
c893230
 
 
aad7814
 
 
c893230
aad7814
 
 
 
 
e561e67
 
 
 
 
 
 
c893230
aad7814
 
 
 
 
c893230
e561e67
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
"""LLM + regex sanitisation before tenant uploads are indexed for RAG / style learning.

On-disk uploads are unchanged; only text written to the vector index is sanitised.

When the v2 backend is importable, deterministic redaction uses
``backend.core.pii_scrubber`` (dual-pass regex + PERSON NER + PropTech whitelist)
as a mandatory post-pass after the LLM tier. The LLM prompt teaches descriptive
intent and cites real failure-mode anti-patterns so survey vocabulary survives.
"""

from __future__ import annotations

import logging
import re
from enum import Enum
from typing import TYPE_CHECKING

from app.config import settings

if TYPE_CHECKING:
    from langchain_core.documents import Document

logger = logging.getLogger(__name__)


class DocumentSanitisationError(RuntimeError):
    """Raised when sanitisation is required but cannot complete safely."""


class RedactionStrategy(Enum):
    """Selects which redaction engine ``sanitise_text_for_rag_sync`` routes to.

    ``AI_HYBRID`` runs the existing LLM + regex/NER pipeline (unchanged).
    ``DETERMINISTIC_CODE`` runs a self-contained regex + flashtext pass with no
    LLM, NER model, or network I/O β€” suitable for offline / air-gapped redaction.
    """

    AI_HYBRID = "ai_hybrid"
    DETERMINISTIC_CODE = "deterministic"


RAG_UPLOAD_SANITISATION_SYSTEM_PROMPT = """\
You are a zero-bleed Data Privacy Compliance Engine for UK PropTech and RICS Home \
Survey Level 3 documentation. Purge unique identifiers belonging to individuals, \
corporate entities, transactions, or exact physical locations. Keep 100% intact the \
technical, architectural, structural, and pathological diagnostics β€” over-redaction \
that strips property-composition vocabulary is a critical failure for downstream RAG \
style learning.

### THE DESCRIPTIVE INTENT RULE (apply before removing any token)
If a word or phrase describes HOW a property is built, WHERE a defect is physically \
located, or WHAT structural condition it is in, it is EXEMPT. Only remove a token if \
it identifies a single unique transaction, client, or specific property plot.

### WHITELIST β€” never remove, alter, summarise, or paraphrase
Leave these categories exactly as written (enforcement also runs in a deterministic \
post-pass with the same domain lexicon):

- Spatial orientations & locations: front, rear, left, right, elevation, side, landing, \
bedroom, kitchen, loft, eaves, ground floor, upper level, apex, perimeter, boundary.
- Materials & components: timber, purlins, brickwork, lead, flashing, uPVC, slate, felt, \
Velux, concrete, downpipe, cladding, glazing, etc.
- Pathology & condition: condition rating, defect, cracking, damp, spalled, deflection, \
moisture, insulation, rot, water ingress, settlement, distortion, etc.
- Regulatory frameworks (generic names only): Building Regulations, Local Authority, \
FENSA, Gas Safe, RICS, Home Survey Level 3, Environment Agency, Council Tax Band, \
Party Wall etc. Act 1996, lease, freeholder, covenants, obligations.
- Generic historical/statutory era references (e.g. "alterations between the 1960s and \
1980s") β€” not transaction-specific calendar dates.

Remove only when a regulatory or certification name is paired with a specific licence \
number, account number, or job reference tied to an identifiable matter.

### ANTI-PATTERNS β€” do not repeat these systemic errors
- Do NOT redact "timber", "asbestos", or "combination" inside structural phrases \
(e.g. "timber purlins", "Asbestos Containing Materials (ACMs)", "combination boiler").
- Do NOT redact positional descriptions (e.g. "left side of the living room", \
"rear right bedroom", "front elevation").
- Do NOT redact standalone certification scheme labels (ELECSA, NICEIC, FENSA) β€” but \
DO remove postcode-like or numeric middle segments inside reference strings \
(e.g. in "Ref. No. 22/SW1A1AA/ELECSA", remove the middle segment only; a downstream \
deterministic pass handles rigid ID patterns).

### BLACKLIST β€” surgically remove the entire span
Delete the sensitive text cleanly. Do not insert placeholders like "[REDACTED]" β€” \
remove the span and collapse surrounding whitespace naturally. A downstream \
deterministic pass may apply further token-level redaction after your output.

- Personal identifiers: client, occupier, vendor, purchaser, solicitor, agent, \
surveyor, witness, contractor names, initials, signatures.
- Contact vectors: email, phone, fax, identifying URLs, portal credentials.
- Property locations: house/flat numbers, property names, street names, postcodes, \
correspondence addresses. Remove address fragments; do not attempt to keep or drop \
town/city names by judgment β€” remove only explicit pinpointing fragments listed here.
- Registry & file IDs: Land Registry title numbers, UPRNs, EPC certificate numbers, \
planning application numbers, council tax account numbers, job/file/invoice serials, \
National Insurance numbers, bank details, tribunal/court/lease reference numbers.
- Financial metrics: valuations, prices, ground rent, service charges, mortgage figures.
- Transaction dates: exact inspection, report, exchange, or completion dates for this job.
- Special category: health, disability, family circumstances, non-property litigation, \
complaints.

### OUTPUT
- Inline text parser only. Retain paragraph breaks, capitalization, headers, and \
section order.
- If a sentence contains no blacklist items, output it verbatim.
- Do not summarise, compress, paraphrase, or invent information.
- No markdown fences, no JSON, no commentary. Output ONLY the sanitised excerpt text."""

_POSTCODE_RE = re.compile(
    r"\b([A-Z]{1,2}\d[A-Z\d]?\s*\d[A-Z]{2})\b", re.IGNORECASE
)
_ADDRESS_LINE_RE = re.compile(
    r"\b(\d{1,4}\s+[A-Za-z][A-Za-z'\-]*(?:\s+[A-Za-z][A-Za-z'\-]*){0,6}\s+"
    r"(Road|Rd|Street|St|Avenue|Ave|Lane|Ln|Drive|Dr|Crescent|Close|Place|Way|Gardens|Gdns|Court|Ct|Terrace|Terr))\b",
    re.IGNORECASE,
)
_MONEY_RE = re.compile(
    r"(Β£\s*\d[\d,]*(?:\.\d+)?|\b\d[\d,]*(?:\.\d+)?\s*(?:gbp|pounds)\b)",
    re.IGNORECASE,
)
_DATE_RE = re.compile(
    r"\b(?:\d{1,2}[/-]\d{1,2}[/-]\d{2,4}|\d{1,2}\s+"
    r"(?:jan|feb|mar|apr|may|jun|jul|aug|sep|sept|oct|nov|dec)[a-z]*\s+\d{2,4})\b",
    re.IGNORECASE,
)
_LONG_NUMBER_RE = re.compile(r"\b\d{6,}\b")
_EMAIL_RE = re.compile(r"\b[\w.+-]+@[\w.-]+\.[A-Za-z]{2,}\b")
_PHONE_RE = re.compile(
    r"\b(?:\+44\s?|0)(?:\d[\s-]?){9,12}\b|\b\d{3,4}[\s-]\d{3,4}[\s-]\d{3,4}\b"
)
_URL_RE = re.compile(r"https?://[^\s<>\"']+", re.IGNORECASE)
_NINO_RE = re.compile(
    r"\b[A-CEGHJ-PR-TW-Z]{2}\s?\d{2}\s?\d{2}\s?\d{2}\s?[A-D]\b", re.IGNORECASE
)
_TITLE_NUMBER_RE = re.compile(r"\b[A-Z]{1,3}\d{5,6}\b")
_REF_ID_RE = re.compile(r"\b[A-Z]{2,}[\-/][A-Z0-9][\-/A-Z0-9]{2,}\b", re.IGNORECASE)

# ── Deterministic strategy β€” compiled once at module load (zero-AI/zero-network) ─
# Used only by ``_deterministic_redact``. Kept boundary-anchored so they never
# swallow adjacent survey vocabulary.
_REDACT_PLACEHOLDER = "[REDACTED]"
_DET_RE_POSTCODE = re.compile(r"\b[A-Z]{1,2}[0-9][A-Z0-9]?\s?[0-9][A-Z]{2}\b", re.IGNORECASE)
_DET_RE_EMAIL = re.compile(r"[\w.\-]+@[\w.\-]+\.\w+")
_DET_RE_PHONE_UK = re.compile(
    r"(?:(?:\+44\s?|0)(?:7\d{3}|\d{2,4})[\s\-]?\d{3,4}[\s\-]?\d{3,4})"
)
# Applied in order: most-specific (email) β†’ phone β†’ postcode.
_DET_REDACTORS: tuple[re.Pattern[str], ...] = (
    _DET_RE_EMAIL,
    _DET_RE_PHONE_UK,
    _DET_RE_POSTCODE,
)


def should_sanitise_for_rag(tenant_id: str) -> bool:
    """Whether uploads for this tenant should be sanitised before vector indexing."""
    if not settings.enable_rag_upload_sanitisation:
        return False
    if (
        settings.rag_sanitisation_skip_kb_tenant
        and tenant_id == settings.knowledge_base_tenant_id
    ):
        return False
    return True


def _v2_scrub(text: str) -> str | None:
    """Use the v2 dual-pass scrubber when the backend package is importable."""
    try:
        from backend.core import pii_scrubber

        return pii_scrubber.scrub(text).text
    except ImportError:
        return None


def regex_sanitise_text(text: str) -> str:
    """Deterministic redaction when the LLM path is unavailable or as a safety net."""
    v2 = _v2_scrub(text)
    if v2 is not None:
        return v2

    t = (text or "").strip()
    if not t:
        return ""
    t = _EMAIL_RE.sub("", t)
    t = _URL_RE.sub("", t)
    t = _PHONE_RE.sub("", t)
    t = _POSTCODE_RE.sub("", t)
    t = _ADDRESS_LINE_RE.sub("", t)
    t = _MONEY_RE.sub("", t)
    t = _DATE_RE.sub("", t)
    t = _NINO_RE.sub("", t)
    t = _TITLE_NUMBER_RE.sub("", t)
    t = _REF_ID_RE.sub("", t)
    t = _LONG_NUMBER_RE.sub("", t)
    t = re.sub(r"[ \t]{2,}", " ", t)
    t = re.sub(r"\n{3,}", "\n\n", t)
    return t.strip()


def _deterministic_post_pass(text: str) -> str:
    """Second-pass deterministic scrub after LLM output (whitelist-aware when v2 available)."""
    cleaned = regex_sanitise_text(text)
    return cleaned.strip()


def _deterministic_redact(
    text: str,
    db_context: dict[str, str] | None = None,
) -> str:
    """Zero-AI, zero-network text redaction (compiled regex + exact-string wipe).

    Runs entirely in-process: no LLM, no NER model, no network I/O. Stages, in
    order, mirror the deterministic redaction spec:

    1. **Metadata purge** β€” not applicable at the text layer. The original PDF
       bytes never reach this function (PDFs are extracted to text upstream and
       on-disk uploads are left intact by design), so this stage is a documented
       no-op here rather than a ``fitz`` metadata wipe.
    2. **Compiled-regex substitution** β€” UK postcodes, emails, and phone numbers
       are replaced with ``[REDACTED]`` using module-level patterns compiled at
       import time (never inside this function).
    3. **Context-driven exact-string wipe** β€” when ``db_context`` is non-empty,
       its values (session-known PII such as client name / address) are removed
       case-insensitively via :class:`flashtext.KeywordProcessor`.

    Args:
        text: Extracted document text to sanitise. Must be a non-empty string.
        db_context: Optional ``{label: pii_value}`` map of exact sensitive strings
            to wipe. ``None`` or empty skips stage 3 gracefully.

    Returns:
        The sanitised text with all matches replaced by ``[REDACTED]``.

    Raises:
        ValueError: If ``text`` is not a non-empty string.
    """
    if not isinstance(text, str) or not text.strip():
        raise ValueError("_deterministic_redact requires non-empty text input")

    # Stage 2 β€” compiled-regex substitution.
    cleaned = text
    for pattern in _DET_REDACTORS:
        cleaned = pattern.sub(_REDACT_PLACEHOLDER, cleaned)

    # Stage 3 β€” exact-string context wipe (only when context is supplied).
    if db_context:
        try:
            from flashtext import KeywordProcessor
        except ImportError:
            logger.warning(
                "flashtext not installed; skipping deterministic context wipe "
                "(add flashtext>=2.7 to enable stage 3)."
            )
        else:
            kp = KeywordProcessor(case_sensitive=False)
            for value in db_context.values():
                if value and value.strip():
                    kp.add_keyword(value.strip(), _REDACT_PLACEHOLDER)
            if kp.get_all_keywords():
                cleaned = kp.replace_keywords(cleaned)

    cleaned = re.sub(r"[ \t]{2,}", " ", cleaned)
    cleaned = re.sub(r"\n{3,}", "\n\n", cleaned)
    return cleaned.strip()


def _split_text_chunks(text: str, max_chars: int) -> list[str]:
    body = (text or "").strip()
    if not body:
        return []
    if len(body) <= max_chars:
        return [body]
    parts = re.split(r"(\n\s*\n)", body)
    chunks: list[str] = []
    current = ""
    for part in parts:
        if not part:
            continue
        candidate = current + part
        if len(candidate) <= max_chars:
            current = candidate
            continue
        if current.strip():
            chunks.append(current.strip())
        if len(part) <= max_chars:
            current = part
        else:
            for i in range(0, len(part), max_chars):
                segment = part[i : i + max_chars].strip()
                if segment:
                    chunks.append(segment)
            current = ""
    if current.strip():
        chunks.append(current.strip())
    return chunks or [body[:max_chars]]


def _llm_sanitise_chunk_sync(chunk: str, *, part: int, total: int) -> str:
    from app.llm.openai_chat import chat_completions_create

    import asyncio

    prefix = (
        f"Sanitise this RICS Home Survey Level 3 excerpt (part {part} of {total}). "
        f"Apply the Descriptive Intent rule: preserve all structural survey vocabulary; "
        f"remove only unique identifiers.\n\n"
    )
    user_content = prefix + chunk

    async def _run() -> str:
        return await chat_completions_create(
            messages=[
                {"role": "system", "content": RAG_UPLOAD_SANITISATION_SYSTEM_PROMPT},
                {"role": "user", "content": user_content},
            ],
            model=settings.chat_model,
            max_tokens=int(settings.rag_sanitisation_max_output_tokens),
            temperature=0.0,
            phase="rag_upload_sanitise",
        )

    try:
        return asyncio.run(_run())
    except RuntimeError:
        import concurrent.futures

        with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
            return pool.submit(asyncio.run, _run()).result()


def sanitise_text_for_rag_sync(
    text: str,
    *,
    tenant_id: str,
    force: bool = False,
    strategy: RedactionStrategy = RedactionStrategy.AI_HYBRID,
    db_context: dict[str, str] | None = None,
) -> str:
    """Sanitise plain text before it is embedded into the tenant vector index.

    ``force=True`` bypasses the per-tenant policy switch. Use this for
    ``style_corpus`` uploads β€” the user's past completed reports MUST be
    PII-scrubbed before indexing because the same content gets reused across
    every new report (a missed name/postcode would leak across jobs).

    Args:
        text: Raw extracted document text to sanitise.
        tenant_id: Tenant whose sanitisation policy governs the AI path.
        force: Bypass the per-tenant policy switch (AI path only).
        strategy: Redaction engine to use. ``AI_HYBRID`` (default) preserves the
            existing LLM + regex/NER pipeline exactly. ``DETERMINISTIC_CODE``
            runs the zero-AI, zero-network regex + flashtext pass and is honoured
            regardless of the per-tenant policy switch (explicit caller request).
        db_context: Optional ``{label: pii_value}`` map of exact sensitive strings
            forwarded to the deterministic path's context wipe. Ignored by the
            AI path.

    Returns:
        The sanitised text (empty string for empty input).
    """
    raw = (text or "").strip()
    if not raw:
        return ""

    # Explicit deterministic request: run in-process, no AI/network, no policy gate.
    if strategy is RedactionStrategy.DETERMINISTIC_CODE:
        return _deterministic_redact(raw, db_context=db_context)

    if not force and not should_sanitise_for_rag(tenant_id):
        return raw

    max_chars = int(settings.rag_sanitisation_chunk_chars)
    chunks = _split_text_chunks(raw, max_chars)
    has_key = bool((settings.openai_api_key or "").strip())
    out_parts: list[str] = []

    for i, chunk in enumerate(chunks, start=1):
        cleaned = ""
        if has_key:
            try:
                llm_out = (
                    _llm_sanitise_chunk_sync(chunk, part=i, total=len(chunks)) or ""
                ).strip()
                if llm_out:
                    cleaned = _deterministic_post_pass(llm_out)
            except Exception as exc:
                logger.warning(
                    "RAG sanitisation LLM failed tenant=%s part=%d/%d: %s",
                    tenant_id,
                    i,
                    len(chunks),
                    exc,
                )
        if not cleaned:
            cleaned = regex_sanitise_text(chunk)
            if has_key:
                logger.info(
                    "RAG sanitisation using regex fallback tenant=%s part=%d/%d",
                    tenant_id,
                    i,
                    len(chunks),
                )
        if not cleaned and settings.rag_sanitisation_fail_closed:
            raise DocumentSanitisationError(
                f"Sanitisation produced empty text for tenant={tenant_id} part={i}/{len(chunks)}"
            )
        out_parts.append(cleaned)

    combined = "\n\n".join(p for p in out_parts if p).strip()
    if not combined and settings.rag_sanitisation_fail_closed:
        raise DocumentSanitisationError(
            f"Sanitisation produced empty document for tenant={tenant_id}"
        )
    return combined or regex_sanitise_text(raw)


def sanitise_langchain_documents_sync(
    documents: list[Document],
    *,
    tenant_id: str,
    force: bool = False,
    strategy: RedactionStrategy = RedactionStrategy.AI_HYBRID,
    db_context: dict[str, str] | None = None,
) -> list[Document]:
    """Return documents whose ``page_content`` has been sanitised for RAG indexing.

    Pass ``force=True`` for ``style_corpus`` uploads so PII redaction runs
    regardless of the tenant's global sanitisation policy. ``strategy`` and
    ``db_context`` are forwarded per-document to :func:`sanitise_text_for_rag_sync`;
    the ``DETERMINISTIC_CODE`` strategy always runs regardless of tenant policy.
    """
    if (
        strategy is RedactionStrategy.AI_HYBRID
        and not force
        and not should_sanitise_for_rag(tenant_id)
    ):
        return documents

    from langchain_core.documents import Document as LCDocument

    out: list[LCDocument] = []
    for doc in documents:
        meta = dict(doc.metadata or {})
        cleaned = sanitise_text_for_rag_sync(
            doc.page_content or "",
            tenant_id=tenant_id,
            force=force,
            strategy=strategy,
            db_context=db_context,
        )
        meta["rag_sanitised"] = True
        out.append(LCDocument(page_content=cleaned, metadata=meta))
    return out