"""OCR / layout normalization for extracted PDF text (STEP 1). PDF text extraction routinely corrupts the document in ways that poison downstream retrieval and generation: * sentences are broken by hard line wraps, * words are split with end-of-line hyphens ("condi-\ntion"), * the same running header / footer / page number repeats on every page, * whitespace and newlines are inconsistent. These functions repair that corruption deterministically before chunking, so chunks contain whole sentences and no boilerplate noise. Pure string ops — no dependencies, fully unit-testable. The table sentinel (``[TABLE]…[/TABLE]``) emitted by the parser is treated as opaque: normalization never reflows text inside a table block. """ from __future__ import annotations import re TABLE_OPEN = "[TABLE]" TABLE_CLOSE = "[/TABLE]" _TABLE_BLOCK_RE = re.compile( re.escape(TABLE_OPEN) + r".*?" + re.escape(TABLE_CLOSE), re.DOTALL, ) # end-of-line hyphenation: "condi-\ntion" -> "condition" _HYPHEN_WRAP_RE = re.compile(r"(\w)-\n[ \t]*(\w)") # bare page-number / "Page x of y" / "x / y" lines _PAGE_NUM_RE = re.compile( r"^\s*(?:page\s+)?\d+\s*(?:of|/)\s*\d+\s*$|^\s*page\s+\d+\s*$|^\s*\d{1,4}\s*$", re.IGNORECASE, ) def _protect_tables(text: str) -> tuple[str, list[str]]: """Replace table blocks with placeholders so reflow never touches them.""" blocks: list[str] = [] def _stash(m: re.Match[str]) -> str: blocks.append(m.group(0)) return f"\x00TBL{len(blocks) - 1}\x00" return _TABLE_BLOCK_RE.sub(_stash, text), blocks def _restore_tables(text: str, blocks: list[str]) -> str: for i, block in enumerate(blocks): text = text.replace(f"\x00TBL{i}\x00", block) return text def normalize_text(text: str) -> str: """Repair a single page/section of extracted text. Steps: normalize unicode whitespace, de-hyphenate wrapped words, unwrap hard-wrapped sentences within a paragraph (single newline -> space) while preserving paragraph breaks (blank lines), and collapse excess whitespace. Table blocks are preserved verbatim. """ if not text: return "" protected, blocks = _protect_tables(text) # Normalize unicode whitespace / non-breaking spaces. protected = protected.replace("\u00a0", " ").replace("\r\n", "\n").replace("\r", "\n") # De-hyphenate words split across a line break. protected = _HYPHEN_WRAP_RE.sub(r"\1\2", protected) # Unwrap: within each blank-line-delimited paragraph, join hard-wrapped # lines into a single line. This restores sentence continuity that PDF # extraction destroys by emitting one newline per visual line. paragraphs = re.split(r"\n[ \t]*\n", protected) rebuilt: list[str] = [] for para in paragraphs: if "\x00TBL" in para: rebuilt.append(para.strip()) continue lines = [ln.strip() for ln in para.split("\n") if ln.strip()] if not lines: continue rebuilt.append(" ".join(lines)) out = "\n\n".join(rebuilt) # Collapse runs of spaces and excessive blank lines. out = re.sub(r"[ \t]{2,}", " ", out) out = re.sub(r"\n{3,}", "\n\n", out) return _restore_tables(out.strip(), blocks) def _candidate_boundary_lines(page_text: str, edge: int = 3) -> set[str]: """First/last ``edge`` non-empty lines of a page (header/footer candidates).""" lines = [ln.strip() for ln in page_text.split("\n") if ln.strip()] if not lines: return set() return set(lines[:edge]) | set(lines[-edge:]) def strip_running_headers_footers(pages: list[str], *, edge: int = 3) -> list[str]: """Remove repeated running headers/footers and page numbers across pages. A short line appearing in the top/bottom ``edge`` lines of a majority of pages is treated as boilerplate and removed from every page. Bare page numbers are always removed. Single-page documents are returned unchanged (no cross-page signal to safely act on). """ if len(pages) < 3: # Still strip bare page numbers even when we can't detect repetition. return [_drop_page_numbers(p) for p in pages] freq: dict[str, int] = {} for p in pages: for line in _candidate_boundary_lines(p, edge): if len(line) <= 120: freq[line] = freq.get(line, 0) + 1 threshold = max(2, int(len(pages) * 0.5)) boilerplate = {ln for ln, n in freq.items() if n >= threshold} cleaned: list[str] = [] for p in pages: kept = [] for line in p.split("\n"): s = line.strip() if s and s in boilerplate: continue kept.append(line) cleaned.append(_drop_page_numbers("\n".join(kept))) return cleaned def _drop_page_numbers(text: str) -> str: return "\n".join( ln for ln in text.split("\n") if not _PAGE_NUM_RE.match(ln.strip()) ) def normalize_pages(pages: list[str]) -> list[str]: """Full document-level normalization: strip boilerplate, then reflow each page.""" deboiled = strip_running_headers_footers(pages) return [normalize_text(p) for p in deboiled]