File size: 20,403 Bytes
6143b69
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
"""Report Post-Processing & Quality Normalisation Pipeline.

A pure, deterministic transformation pass that runs as the FINAL stage after the
RAG report generator produces a section's draft text. It removes PDF-extraction
artifacts, repairs grammar broken by redaction tokens, deduplicates stitched
content and normalises whitespace/punctuation.

HARD CONSTRAINTS (enforced by design):
    * Zero ML models, zero LLM calls, zero network I/O.
    * Never paraphrases or rewrites substantive survey findings β€” only removes
      artifacts, repairs grammar around redaction tokens, and deduplicates.
    * Stage order is fixed; later stages depend on earlier ones.
    * Does not touch the RAG pipeline, redaction pipeline, or template ingestion.

The single public entry point is :func:`polish_report`.
"""

from __future__ import annotations

import re
from functools import lru_cache

# ── Project defaults ─────────────────────────────────────────────────────────
# The redaction layer emits both base tokens ([REDACTED_REF]) and referential
# tokens ([REDACTED_NAME_1]); some draft text contains double brackets
# ([[REDACTED_NAME_2]]). This default covers all three. Override per-call/config.
DEFAULT_PLACEHOLDER_PATTERN = r"\[{1,2}REDACTED_[A-Z]+(?:_\d+)?\]{1,2}"
DEFAULT_SECTION_SEPARATOR = "\n\n"
DEFAULT_DEDUP_THRESHOLD = 0.85

# Short, meaningful rating boilerplate (e.g. a one-line condition score) must
# never be deduplicated β€” only substantial repeated content is collapsed.
_MIN_DEDUP_SENTENCE_CHARS = 40
_MIN_DEDUP_PARAGRAPH_CHARS = 80

# Private-use sentinel for placeholder grammar repair (never appears in input).
_PH_SENTINEL = "\ue000"

# Visible, honest placeholder for an unresolved redaction token. Never emit the
# bare word "specified" β€” an operator must be able to see what needs confirming.
NEUTRAL_PLACEHOLDER = "[SURVEYOR TO CONFIRM]"

# ── Stage 1: PDF extraction artifacts ────────────────────────────────────────
_RE_PAGE_LABEL = re.compile(r"(?i)\bpage\s*\d{1,4}\b")
_RE_ORPHAN_PAGE_NUMBER = re.compile(r"(?m)^[ \t]*\d{1,4}[ \t]*$")
_RE_DEFAULT_DOC_HEADER = re.compile(
    r"(?i)RICS\s+Home\s+Survey\s*[-\u2013\u2014]?\s*Level\s+\d+",
)
_RE_TOC_LINE = re.compile(r"(?im)^[ \t]*ToC\s*[:\-].*$")

# ── Stage 2: repeated section headers ────────────────────────────────────────
_RE_SECTION_HEADER = re.compile(r"^([A-Z][A-Z0-9\s\-/&]{3,}:)\s*$")
_HEADER_DEDUP_WINDOW_WORDS = 500

# ── Stage 4: sentence splitting (newline-agnostic) ───────────────────────────
_RE_SENTENCE_SPLIT = re.compile(r"(?<=[.!?])\s+")
_RE_WS = re.compile(r"\s+")

# ── Stage 5: ALL-CAPS internal directives ────────────────────────────────────
_DIRECTIVE_CUES = frozenset(
    {"SEE", "REFER", "OUTLINED", "DESCRIBED", "ABOVE", "BELOW", "STATED", "NOTED"}
)
_RE_WORD_TOKEN = re.compile(r"[A-Za-z']+")

# ── Stage 6: whitespace / punctuation ────────────────────────────────────────
_RE_MULTI_NEWLINE = re.compile(r"\n{3,}")
_RE_TRAILING_SPACE = re.compile(r"[ \t]+$", re.MULTILINE)
_RE_DOUBLE_SPACE = re.compile(r"[ \t]{2,}")
_RE_HYPHEN_LINEBREAK = re.compile(r"([A-Za-z])-\n[ \t]*([a-z])")
# Rejoin a mid-sentence line break (prev char is alnum/comma/semicolon, next is
# lowercase/digit). Crosses blank-line gaps left by stripped page-break artifacts,
# but never merges across sentence-ending punctuation.
_RE_SOFT_WRAP = re.compile(r"(?<=[A-Za-z,;0-9])[ \t]*\n[ \t\n]*(?=[a-z0-9(])")
_RE_SPACE_BEFORE_PUNCT = re.compile(r"[ \t]+([,.;:!?])")
_RE_SPACE_AFTER_SENTENCE = re.compile(r"([a-z0-9])([.!?])([A-Z])")
_UNICODE_TRANSLATION = {
    ord("\u201c"): '"',
    ord("\u201d"): '"',
    ord("\u2018"): "'",
    ord("\u2019"): "'",
    ord("\u2013"): "-",
    ord("\u2014"): "-",
    ord("\u00a0"): " ",
    ord("\u2022"): "-",
}

# ── Stage 7: paragraph flow ──────────────────────────────────────────────────
_LEADING_CONJUNCTIONS = frozenset({"and", "but", "or", "nor", "so", "yet"})
_RE_LEADING_CONJUNCTION = re.compile(r"^(and|but|or|nor|so|yet)\b[ ,]*", re.IGNORECASE)
_RE_INLINE_HEADER = re.compile(r"[ \t]*\b([A-Z][A-Z0-9 \-/&]{3,}:)[ \t]*")


def _stage_1_remove_extraction_artifacts(
    text: str,
    extra_header_patterns: list[re.Pattern[str]] | None = None,
) -> str:
    """Remove orphaned page numbers and repeated document headers from PDF extraction."""
    text = _RE_PAGE_LABEL.sub(" ", text)
    text = _RE_DEFAULT_DOC_HEADER.sub(" ", text)
    for pattern in extra_header_patterns or ():
        text = pattern.sub(" ", text)
    text = _RE_TOC_LINE.sub("", text)
    text = _RE_ORPHAN_PAGE_NUMBER.sub("", text)
    return text


def _stage_2_deduplicate_section_headers(text: str) -> str:
    """Keep only the first occurrence of each ALL-CAPS section header within a 500-word window."""
    last_seen: dict[str, int] = {}
    kept_lines: list[str] = []
    word_count = 0
    for line in text.split("\n"):
        match = _RE_SECTION_HEADER.match(line.strip())
        if match:
            key = _RE_WS.sub(" ", match.group(1).strip().upper())
            previous = last_seen.get(key)
            if (
                previous is not None
                and (word_count - previous) <= _HEADER_DEDUP_WINDOW_WORDS
            ):
                last_seen[key] = word_count
                continue
            last_seen[key] = word_count
        word_count += len(line.split())
        kept_lines.append(line)
    return "\n".join(kept_lines)


@lru_cache(maxsize=16)
def _placeholder_regexes(
    placeholder_pattern: str,
) -> tuple[re.Pattern[str], re.Pattern[str]]:
    """Compile (once per unique pattern) the standalone-line and token-run regexes."""
    token = rf"(?:{placeholder_pattern})"
    standalone = re.compile(
        rf"(?m)^[ \t]*{token}(?:[ \t]*{token})*[ \t]*[.;:,\-]?[ \t]*$\n?"
    )
    run = re.compile(rf"{token}(?:\s*{token})*")
    return standalone, run


# Sentinel-stage regexes (independent of the configurable token pattern).
_RE_PH_POSSESSIVE = re.compile(rf"{_PH_SENTINEL}'s\b")
_RE_PH_SENTENCE_INITIAL = re.compile(
    rf"(\A|[.!?][\"')\]]?\s+|\n[ \t]*){_PH_SENTINEL}[ \t]*(\w)"
)
_RE_PLACEHOLDER_DUP = re.compile(
    r"\[SURVEYOR TO CONFIRM\](?:\s+\[SURVEYOR TO CONFIRM\])+"
)
_RE_ARTICLE_DUP = re.compile(r"\b(the|a|an)\s+(the|a|an)\b", re.IGNORECASE)

# ── Dangling redaction-date artifacts ────────────────────────────────────────
# Fixed safety/regulatory boilerplate embeds a date/reference that PII redaction
# sometimes over-masks, leaving "...work undertaken after [SURVEYOR TO CONFIRM]
# should..." or a trailing "...installed in [SURVEYOR TO CONFIRM].". Dropping the
# temporal/locative-date clause yields clean professional prose without inventing
# a value or losing a finding. A placeholder standing for a material, location or
# other finding is deliberately left visible for the surveyor to confirm.
_PH_LITERAL = re.escape(NEUTRAL_PLACEHOLDER)
_RE_PH_DATED_BEFORE_MODAL = re.compile(
    r"\b(?:undertaken|carried out|completed|performed|installed|fitted|constructed|built)\s+"
    r"(?:after|before|since|from|in|on|around|circa|prior to|up to)\s+"
    rf"{_PH_LITERAL}\s+"
    r"(?=(?:should|must|shall|will|would|may|might|can|are|is|was|were|be)\b)",
    re.IGNORECASE,
)
# Trailing rule is restricted to unambiguously TEMPORAL prepositions β€” "of", "in",
# "on", "from" can introduce a redacted name or location, which must stay visible.
_RE_PH_TRAILING_DATE = re.compile(
    r"\s*,?\s*(?:after|before|since|dated|around|circa|prior to|up to)\s+"
    rf"{_PH_LITERAL}(?=\s*[.;:,)])",
    re.IGNORECASE,
)


def neutralise_dangling_placeholders(text: str) -> str:
    """Drop a temporal/date clause whose only value is an unresolved placeholder.

    Targets the recurring redaction-date artifact in fixed safety/regulatory
    boilerplate. Never removes a placeholder that stands for a material, location
    or finding β€” that stays visible for surveyor confirmation.
    """
    if NEUTRAL_PLACEHOLDER not in text:
        return text
    text = _RE_PH_DATED_BEFORE_MODAL.sub("", text)
    text = _RE_PH_TRAILING_DATE.sub("", text)
    return text


def _stage_3_normalise_placeholder_grammar(text: str, placeholder_pattern: str) -> str:
    """Repair grammatical breakage caused by inline redaction tokens.

    Tokens are replaced with neutral language; the redacted value is never
    revealed or approximated.
    """
    standalone_re, run_re = _placeholder_regexes(placeholder_pattern)

    # 1. Standalone token(s) on their own line β†’ drop the entire line.
    text = standalone_re.sub("", text)
    # 2. Collapse any run of (adjacent) tokens to a single sentinel.
    text = run_re.sub(_PH_SENTINEL, text)
    # 3. Drop a possessive clitic clinging to a sentinel ([REDACTED]'s β†’ sentinel).
    text = _RE_PH_POSSESSIVE.sub(_PH_SENTINEL, text)
    # 4. Sentence-initial sentinel β†’ drop token, capitalise the following word.
    text = _RE_PH_SENTENCE_INITIAL.sub(lambda m: m.group(1) + m.group(2).upper(), text)
    # 5. Remaining mid-sentence sentinels β†’ explicit surveyor placeholder
    #    (never the bare word "specified", which reads as real content).
    text = text.replace(_PH_SENTINEL, NEUTRAL_PLACEHOLDER)
    # 6. Tidy artefacts of the substitution.
    text = _RE_PLACEHOLDER_DUP.sub(NEUTRAL_PLACEHOLDER, text)
    text = _RE_ARTICLE_DUP.sub(r"\1", text)
    text = neutralise_dangling_placeholders(text)
    return text


def _normalise_for_compare(text: str) -> str:
    return _RE_WS.sub(" ", text.lower()).strip()


def _trigrams(text: str) -> set[str]:
    return {text[i : i + 3] for i in range(len(text) - 2)}


def _jaccard_similarity(a: str, b: str) -> float:
    """Character-trigram Jaccard similarity β€” fast, zero dependencies."""
    t_a, t_b = _trigrams(a.lower()), _trigrams(b.lower())
    if not t_a or not t_b:
        return 0.0
    return len(t_a & t_b) / len(t_a | t_b)


def _stage_4_deduplicate_content(
    text: str, similarity_threshold: float = DEFAULT_DEDUP_THRESHOLD
) -> str:
    """Remove duplicate sentences (exact) and near-duplicate paragraphs (Jaccard).

    Short boilerplate below ``_MIN_DEDUP_SENTENCE_CHARS`` (e.g. condition-rating
    tags) is preserved verbatim. Pure string comparison only β€” no ML.
    """
    paragraphs = re.split(r"\n{2,}", text)
    seen_sentences: set[str] = set()
    kept_paragraphs: list[str] = []
    kept_norms: list[str] = []

    for paragraph in paragraphs:
        sentences = _RE_SENTENCE_SPLIT.split(paragraph.strip())
        deduped: list[str] = []
        prev_norm: str | None = None
        for sentence in sentences:
            norm = _normalise_for_compare(sentence)
            # Collapse immediately-adjacent identical sentences (stitching noise),
            # regardless of length β€” non-adjacent short repeats are preserved.
            if norm and norm == prev_norm:
                continue
            if len(norm) >= _MIN_DEDUP_SENTENCE_CHARS:
                if norm in seen_sentences:
                    prev_norm = norm
                    continue
                seen_sentences.add(norm)
            if sentence.strip():
                deduped.append(sentence.strip())
            prev_norm = norm

        rebuilt = " ".join(deduped)
        if not rebuilt.strip():
            continue

        norm_paragraph = _normalise_for_compare(rebuilt)
        if len(norm_paragraph) >= _MIN_DEDUP_PARAGRAPH_CHARS and any(
            _jaccard_similarity(norm_paragraph, prior) >= similarity_threshold
            for prior in kept_norms
        ):
            continue

        kept_paragraphs.append(rebuilt)
        kept_norms.append(norm_paragraph)

    return "\n\n".join(kept_paragraphs)


def _is_caps_directive(sentence: str) -> bool:
    """True when a sentence is an ALL-CAPS internal directive (not client-facing)."""
    tokens = _RE_WORD_TOKEN.findall(sentence)
    alpha = [t for t in tokens if any(c.isalpha() for c in t)]
    if len(alpha) < 2:
        return False
    upper = [t for t in alpha if len(t) >= 2 and t.upper() == t]
    if len(upper) / len(alpha) <= 0.6:
        return False
    return bool({t.upper() for t in tokens} & _DIRECTIVE_CUES)


def _stage_5_scrub_internal_directives(text: str) -> str:
    """Remove ALL-CAPS procedural directives (e.g. SEE THE LIMITATIONS ... ABOVE)."""
    paragraphs = re.split(r"\n{2,}", text)
    out: list[str] = []
    for paragraph in paragraphs:
        sentences = _RE_SENTENCE_SPLIT.split(paragraph)
        kept = [s for s in sentences if not _is_caps_directive(s)]
        rebuilt = " ".join(s.strip() for s in kept if s.strip())
        if rebuilt.strip():
            out.append(rebuilt)
    return "\n\n".join(out)


def _stage_6_normalise_whitespace_and_punctuation(text: str) -> str:
    """Normalise unicode, rejoin soft-wrapped lines, and tidy spacing/newlines."""
    text = text.translate(_UNICODE_TRANSLATION)
    text = _RE_HYPHEN_LINEBREAK.sub(r"\1\2", text)  # de-hyphenate broken words
    text = _RE_SOFT_WRAP.sub(" ", text)  # rejoin mid-sentence line wraps
    text = _RE_TRAILING_SPACE.sub("", text)
    text = _RE_DOUBLE_SPACE.sub(" ", text)
    text = _RE_SPACE_BEFORE_PUNCT.sub(r"\1", text)
    text = _RE_SPACE_AFTER_SENTENCE.sub(r"\1\2 \3", text)
    text = _RE_MULTI_NEWLINE.sub("\n\n", text)
    return text.strip()


def _repair_paragraph(paragraph: str) -> str:
    stripped = paragraph.strip()
    if not stripped:
        return ""
    # Drop a coordinating conjunction orphaned from a stripped predecessor sentence.
    first_word = stripped.split(maxsplit=1)[0].lower().strip(",")
    if first_word in _LEADING_CONJUNCTIONS and not stripped[0].isupper():
        stripped = _RE_LEADING_CONJUNCTION.sub("", stripped, count=1).lstrip()
    if not stripped:
        return ""
    # Capitalise the first alphabetic character.
    if stripped[0].islower():
        stripped = stripped[0].upper() + stripped[1:]
    # Ensure terminal punctuation (headers ending ':' are left alone).
    if stripped[-1] not in ".!?:":
        stripped = f"{stripped}."
    return stripped


def _stage_7_repair_paragraph_flow(text: str) -> str:
    """Capitalise paragraph starts, drop orphan leading conjunctions, ensure terminal stops."""
    # Lift inline ALL-CAPS headers onto their own line for readability.
    text = _RE_INLINE_HEADER.sub(lambda m: f"\n\n{m.group(1)}\n", text)
    text = _RE_MULTI_NEWLINE.sub("\n\n", text)
    paragraphs = re.split(r"\n{2,}", text)
    repaired = [_repair_paragraph(p) for p in paragraphs]
    return "\n\n".join(p for p in repaired if p)


def _resolve_defaults(
    placeholder_pattern: str | None,
    dedup_similarity_threshold: float | None,
    extra_header_patterns: list[re.Pattern[str]] | list[str] | None,
    debug: bool | None,
) -> tuple[str, float, list[re.Pattern[str]], bool]:
    """Pull unset arguments from project config, with hard-coded fallbacks."""
    cfg_pattern = DEFAULT_PLACEHOLDER_PATTERN
    cfg_threshold = DEFAULT_DEDUP_THRESHOLD
    cfg_extra: list[str] = []
    cfg_debug = False
    try:  # config is optional so the module stays standalone-importable/testable.
        from backend.config import settings

        cfg_pattern = getattr(settings, "postprocess_placeholder_pattern", cfg_pattern)
        cfg_threshold = getattr(settings, "dedup_similarity_threshold", cfg_threshold)
        cfg_extra = list(getattr(settings, "extra_header_patterns", []) or [])
        cfg_debug = bool(getattr(settings, "postprocess_debug", False))
    except Exception:  # noqa: BLE001 β€” never let config import break a pure transform.
        pass

    pattern = placeholder_pattern if placeholder_pattern is not None else cfg_pattern
    threshold = (
        dedup_similarity_threshold
        if dedup_similarity_threshold is not None
        else cfg_threshold
    )

    raw_extra = (
        extra_header_patterns if extra_header_patterns is not None else cfg_extra
    )
    compiled_extra: list[re.Pattern[str]] = []
    for item in raw_extra:
        compiled_extra.append(
            item if isinstance(item, re.Pattern) else re.compile(item, re.IGNORECASE)
        )

    return (
        pattern,
        threshold,
        compiled_extra,
        (debug if debug is not None else cfg_debug),
    )


def polish_report(
    raw_report: str,
    placeholder_pattern: str | None = None,
    section_separator: str = DEFAULT_SECTION_SEPARATOR,
    extra_header_patterns: list[re.Pattern[str]] | list[str] | None = None,
    dedup_similarity_threshold: float | None = None,
    debug: bool | None = None,
) -> str:
    """Apply a sequential, non-destructive post-processing pipeline to a draft report.

    Stages (in strict order):
      1. PDF extraction artifact removal
      2. Repeated section header suppression
      3. Placeholder grammar normalisation
      4. Duplicate sentence / paragraph deduplication
      5. ALL-CAPS directive scrubbing
      6. Whitespace and punctuation normalisation
      7. Paragraph-level flow repair

    Args:
        raw_report: The full draft report string from the RAG pipeline.
        placeholder_pattern: Regex matching redacted tokens. Defaults to the
            project pattern (config ``postprocess_placeholder_pattern``), which
            covers ``[REDACTED_NAME_1]``, ``[REDACTED_REF]`` and ``[[...]]``.
        section_separator: Delimiter used to join the polished section blocks.
        extra_header_patterns: Additional firm-specific header patterns to strip
            (strings or compiled patterns). Falls back to config
            ``extra_header_patterns``; keeps the pipeline firm-agnostic.
        dedup_similarity_threshold: 0.0-1.0 character-trigram Jaccard threshold
            for near-duplicate paragraph removal (config ``dedup_similarity_threshold``).
        debug: When true, print a before/after preview to stdout.

    Returns:
        A clean, publication-ready report string.

    Raises:
        ValueError: If ``raw_report`` is empty or None.
    """
    if not raw_report or not raw_report.strip():
        raise ValueError("raw_report cannot be empty.")

    pattern, threshold, compiled_extra, debug_on = _resolve_defaults(
        placeholder_pattern, dedup_similarity_threshold, extra_header_patterns, debug
    )

    text = raw_report
    text = _stage_1_remove_extraction_artifacts(text, compiled_extra)
    text = _stage_2_deduplicate_section_headers(text)
    text = _stage_3_normalise_placeholder_grammar(text, pattern)
    text = _stage_4_deduplicate_content(text, threshold)
    text = _stage_5_scrub_internal_directives(text)
    text = _stage_6_normalise_whitespace_and_punctuation(text)
    text = _stage_7_repair_paragraph_flow(text)

    result = section_separator.join(
        block for block in re.split(r"\n{2,}", text) if block.strip()
    ).strip()

    if debug_on:
        print("=" * 72)
        print("polish_report β€” BEFORE:\n")
        print(raw_report[:1200])
        print("\n" + "-" * 72)
        print("polish_report β€” AFTER:\n")
        print(result[:1200])
        print("=" * 72)

    return result


if __name__ == "__main__":  # pragma: no cover β€” manual before/after demo
    _SAMPLE = (
        "[REDACTED_NAME_1] repairs tend to be expensive due to the associated "
        "scaffolding costs. SEE THE LIMITATIONS OF OUR INSPECTION ABOVE. MAIN ROOF:\n"
        "The main roof structure is of hipped pitch and [[REDACTED_NAME_2]] "
        "construction, partially extended to form\ndormers. The pitched roof "
        "covering appears to be of plain concrete tiles. However, some areas were "
        "concealed from view due to the loft\n2\nRICS Home Survey - Level 3\n26\n"
        "conversion. The element rating is 2. The element rating is 2."
    )
    polish_report(_SAMPLE, debug=True)