RICS / app /ingest /ocr_normalize.py
StormShadow308's picture
Ship production RAG hardening: citation extraction, full-library retrieval, auth.
865bc90
Raw
History Blame Contribute Delete
5.17 kB
"""OCR / layout normalization for extracted PDF text (STEP 1).
PDF text extraction routinely corrupts the document in ways that poison
downstream retrieval and generation:
* sentences are broken by hard line wraps,
* words are split with end-of-line hyphens ("condi-\ntion"),
* the same running header / footer / page number repeats on every page,
* whitespace and newlines are inconsistent.
These functions repair that corruption deterministically before chunking, so
chunks contain whole sentences and no boilerplate noise. Pure string ops — no
dependencies, fully unit-testable.
The table sentinel (``[TABLE]…[/TABLE]``) emitted by the parser is treated as
opaque: normalization never reflows text inside a table block.
"""
from __future__ import annotations
import re
TABLE_OPEN = "[TABLE]"
TABLE_CLOSE = "[/TABLE]"
_TABLE_BLOCK_RE = re.compile(
re.escape(TABLE_OPEN) + r".*?" + re.escape(TABLE_CLOSE),
re.DOTALL,
)
# end-of-line hyphenation: "condi-\ntion" -> "condition"
_HYPHEN_WRAP_RE = re.compile(r"(\w)-\n[ \t]*(\w)")
# bare page-number / "Page x of y" / "x / y" lines
_PAGE_NUM_RE = re.compile(
r"^\s*(?:page\s+)?\d+\s*(?:of|/)\s*\d+\s*$|^\s*page\s+\d+\s*$|^\s*\d{1,4}\s*$",
re.IGNORECASE,
)
def _protect_tables(text: str) -> tuple[str, list[str]]:
"""Replace table blocks with placeholders so reflow never touches them."""
blocks: list[str] = []
def _stash(m: re.Match[str]) -> str:
blocks.append(m.group(0))
return f"\x00TBL{len(blocks) - 1}\x00"
return _TABLE_BLOCK_RE.sub(_stash, text), blocks
def _restore_tables(text: str, blocks: list[str]) -> str:
for i, block in enumerate(blocks):
text = text.replace(f"\x00TBL{i}\x00", block)
return text
def normalize_text(text: str) -> str:
"""Repair a single page/section of extracted text.
Steps: normalize unicode whitespace, de-hyphenate wrapped words, unwrap
hard-wrapped sentences within a paragraph (single newline -> space) while
preserving paragraph breaks (blank lines), and collapse excess whitespace.
Table blocks are preserved verbatim.
"""
if not text:
return ""
protected, blocks = _protect_tables(text)
# Normalize unicode whitespace / non-breaking spaces.
protected = protected.replace("\u00a0", " ").replace("\r\n", "\n").replace("\r", "\n")
# De-hyphenate words split across a line break.
protected = _HYPHEN_WRAP_RE.sub(r"\1\2", protected)
# Unwrap: within each blank-line-delimited paragraph, join hard-wrapped
# lines into a single line. This restores sentence continuity that PDF
# extraction destroys by emitting one newline per visual line.
paragraphs = re.split(r"\n[ \t]*\n", protected)
rebuilt: list[str] = []
for para in paragraphs:
if "\x00TBL" in para:
rebuilt.append(para.strip())
continue
lines = [ln.strip() for ln in para.split("\n") if ln.strip()]
if not lines:
continue
rebuilt.append(" ".join(lines))
out = "\n\n".join(rebuilt)
# Collapse runs of spaces and excessive blank lines.
out = re.sub(r"[ \t]{2,}", " ", out)
out = re.sub(r"\n{3,}", "\n\n", out)
return _restore_tables(out.strip(), blocks)
def _candidate_boundary_lines(page_text: str, edge: int = 3) -> set[str]:
"""First/last ``edge`` non-empty lines of a page (header/footer candidates)."""
lines = [ln.strip() for ln in page_text.split("\n") if ln.strip()]
if not lines:
return set()
return set(lines[:edge]) | set(lines[-edge:])
def strip_running_headers_footers(pages: list[str], *, edge: int = 3) -> list[str]:
"""Remove repeated running headers/footers and page numbers across pages.
A short line appearing in the top/bottom ``edge`` lines of a majority of
pages is treated as boilerplate and removed from every page. Bare page
numbers are always removed. Single-page documents are returned unchanged
(no cross-page signal to safely act on).
"""
if len(pages) < 3:
# Still strip bare page numbers even when we can't detect repetition.
return [_drop_page_numbers(p) for p in pages]
freq: dict[str, int] = {}
for p in pages:
for line in _candidate_boundary_lines(p, edge):
if len(line) <= 120:
freq[line] = freq.get(line, 0) + 1
threshold = max(2, int(len(pages) * 0.5))
boilerplate = {ln for ln, n in freq.items() if n >= threshold}
cleaned: list[str] = []
for p in pages:
kept = []
for line in p.split("\n"):
s = line.strip()
if s and s in boilerplate:
continue
kept.append(line)
cleaned.append(_drop_page_numbers("\n".join(kept)))
return cleaned
def _drop_page_numbers(text: str) -> str:
return "\n".join(
ln for ln in text.split("\n") if not _PAGE_NUM_RE.match(ln.strip())
)
def normalize_pages(pages: list[str]) -> list[str]:
"""Full document-level normalization: strip boilerplate, then reflow each page."""
deboiled = strip_running_headers_footers(pages)
return [normalize_text(p) for p in deboiled]