Spaces:
Runtime error
Runtime error
| """OCR / layout normalization for extracted PDF text (STEP 1). | |
| PDF text extraction routinely corrupts the document in ways that poison | |
| downstream retrieval and generation: | |
| * sentences are broken by hard line wraps, | |
| * words are split with end-of-line hyphens ("condi-\ntion"), | |
| * the same running header / footer / page number repeats on every page, | |
| * whitespace and newlines are inconsistent. | |
| These functions repair that corruption deterministically before chunking, so | |
| chunks contain whole sentences and no boilerplate noise. Pure string ops — no | |
| dependencies, fully unit-testable. | |
| The table sentinel (``[TABLE]…[/TABLE]``) emitted by the parser is treated as | |
| opaque: normalization never reflows text inside a table block. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| TABLE_OPEN = "[TABLE]" | |
| TABLE_CLOSE = "[/TABLE]" | |
| _TABLE_BLOCK_RE = re.compile( | |
| re.escape(TABLE_OPEN) + r".*?" + re.escape(TABLE_CLOSE), | |
| re.DOTALL, | |
| ) | |
| # end-of-line hyphenation: "condi-\ntion" -> "condition" | |
| _HYPHEN_WRAP_RE = re.compile(r"(\w)-\n[ \t]*(\w)") | |
| # bare page-number / "Page x of y" / "x / y" lines | |
| _PAGE_NUM_RE = re.compile( | |
| r"^\s*(?:page\s+)?\d+\s*(?:of|/)\s*\d+\s*$|^\s*page\s+\d+\s*$|^\s*\d{1,4}\s*$", | |
| re.IGNORECASE, | |
| ) | |
| def _protect_tables(text: str) -> tuple[str, list[str]]: | |
| """Replace table blocks with placeholders so reflow never touches them.""" | |
| blocks: list[str] = [] | |
| def _stash(m: re.Match[str]) -> str: | |
| blocks.append(m.group(0)) | |
| return f"\x00TBL{len(blocks) - 1}\x00" | |
| return _TABLE_BLOCK_RE.sub(_stash, text), blocks | |
| def _restore_tables(text: str, blocks: list[str]) -> str: | |
| for i, block in enumerate(blocks): | |
| text = text.replace(f"\x00TBL{i}\x00", block) | |
| return text | |
| def normalize_text(text: str) -> str: | |
| """Repair a single page/section of extracted text. | |
| Steps: normalize unicode whitespace, de-hyphenate wrapped words, unwrap | |
| hard-wrapped sentences within a paragraph (single newline -> space) while | |
| preserving paragraph breaks (blank lines), and collapse excess whitespace. | |
| Table blocks are preserved verbatim. | |
| """ | |
| if not text: | |
| return "" | |
| protected, blocks = _protect_tables(text) | |
| # Normalize unicode whitespace / non-breaking spaces. | |
| protected = protected.replace("\u00a0", " ").replace("\r\n", "\n").replace("\r", "\n") | |
| # De-hyphenate words split across a line break. | |
| protected = _HYPHEN_WRAP_RE.sub(r"\1\2", protected) | |
| # Unwrap: within each blank-line-delimited paragraph, join hard-wrapped | |
| # lines into a single line. This restores sentence continuity that PDF | |
| # extraction destroys by emitting one newline per visual line. | |
| paragraphs = re.split(r"\n[ \t]*\n", protected) | |
| rebuilt: list[str] = [] | |
| for para in paragraphs: | |
| if "\x00TBL" in para: | |
| rebuilt.append(para.strip()) | |
| continue | |
| lines = [ln.strip() for ln in para.split("\n") if ln.strip()] | |
| if not lines: | |
| continue | |
| rebuilt.append(" ".join(lines)) | |
| out = "\n\n".join(rebuilt) | |
| # Collapse runs of spaces and excessive blank lines. | |
| out = re.sub(r"[ \t]{2,}", " ", out) | |
| out = re.sub(r"\n{3,}", "\n\n", out) | |
| return _restore_tables(out.strip(), blocks) | |
| def _candidate_boundary_lines(page_text: str, edge: int = 3) -> set[str]: | |
| """First/last ``edge`` non-empty lines of a page (header/footer candidates).""" | |
| lines = [ln.strip() for ln in page_text.split("\n") if ln.strip()] | |
| if not lines: | |
| return set() | |
| return set(lines[:edge]) | set(lines[-edge:]) | |
| def strip_running_headers_footers(pages: list[str], *, edge: int = 3) -> list[str]: | |
| """Remove repeated running headers/footers and page numbers across pages. | |
| A short line appearing in the top/bottom ``edge`` lines of a majority of | |
| pages is treated as boilerplate and removed from every page. Bare page | |
| numbers are always removed. Single-page documents are returned unchanged | |
| (no cross-page signal to safely act on). | |
| """ | |
| if len(pages) < 3: | |
| # Still strip bare page numbers even when we can't detect repetition. | |
| return [_drop_page_numbers(p) for p in pages] | |
| freq: dict[str, int] = {} | |
| for p in pages: | |
| for line in _candidate_boundary_lines(p, edge): | |
| if len(line) <= 120: | |
| freq[line] = freq.get(line, 0) + 1 | |
| threshold = max(2, int(len(pages) * 0.5)) | |
| boilerplate = {ln for ln, n in freq.items() if n >= threshold} | |
| cleaned: list[str] = [] | |
| for p in pages: | |
| kept = [] | |
| for line in p.split("\n"): | |
| s = line.strip() | |
| if s and s in boilerplate: | |
| continue | |
| kept.append(line) | |
| cleaned.append(_drop_page_numbers("\n".join(kept))) | |
| return cleaned | |
| def _drop_page_numbers(text: str) -> str: | |
| return "\n".join( | |
| ln for ln in text.split("\n") if not _PAGE_NUM_RE.match(ln.strip()) | |
| ) | |
| def normalize_pages(pages: list[str]) -> list[str]: | |
| """Full document-level normalization: strip boilerplate, then reflow each page.""" | |
| deboiled = strip_running_headers_footers(pages) | |
| return [normalize_text(p) for p in deboiled] | |