PDF-Extractor / pipeline /postprocess.py
Archie0099's picture
Fix edge-case bugs found in a thorough code review
74fe989 verified
Raw
History Blame Contribute Delete
10.2 kB
"""Cross-page cleanup: strip running headers / footers and page numbers.
Operates on the FULL list of per-page texts after extraction. A line is treated
as a running header/footer only if a normalized version of it recurs in the top
OR bottom band of a strong majority of pages — and it is removed ONLY from that
band, never from a page body.
Safety properties (each guards a real failure mode found in review):
* The top and bottom bands are DISJOINT and capped at half the page's
non-blank lines, so on short/sparse pages a body line can never be counted
as both header and footer (which previously emptied invoice/form pages).
* Digit normalization (so "Page 1"/"Page 2"/bare "12" match) is applied ONLY
to page-number-like lines; every other line is compared verbatim, so a
numbered heading ("Section 3 …") or a numeric data row ("Total: 1,234.56")
is never collapsed and stripped.
* A line must recur on a strong majority of pages (default 70%), so a real
heading that merely repeats on a couple of pages of a short doc is kept.
* A page is never reduced to nothing: if every candidate line would be
stripped, the page is left untouched.
* No-ops entirely on documents with fewer than ``min_pages`` pages.
"""
import math
import re
from collections import Counter
_DIGITS = re.compile(r"\d+")
_WS = re.compile(r"\s+")
# Words that commonly accompany a page number ("Page 1 of 10", "pg. 3").
_PAGEWORDS = re.compile(r"\b(?:page|pg|p|of|no)\b")
def _key(line: str) -> str:
"""Normalized match key.
Page-number-like lines (only page-words + digits + punctuation remain) get
their digits mapped to ``#`` so "Page 1"/"Page 2"/bare "12" all match. Every
other line is compared verbatim — so numbered headings and numeric data rows
are NOT collapsed together and erased.
The digit→``#`` collapse is gated to lines that are *plausibly a page
number*: at most ONE digit group, OR an explicit page-word ("Page", "of",
…). A bare MULTI-group value — an ISO date ``2026-06-01`` (``#-#-#``), a
currency amount ``1,234.56`` (``#,#.#``), a numeric code — is genuinely
distinct per page; collapsing its skeleton would let the recurrence test
delete real data from every page (e.g. a per-page date header or an invoice
amount footer). Those are compared verbatim so they are never stripped.
"""
base = _WS.sub(" ", line.strip().lower()).strip()
probe = re.sub(r"[^\w\s]", "", _PAGEWORDS.sub(" ", base))
if _DIGITS.sub("", probe).strip() == "": # nothing but page-words + digits
if len(_DIGITS.findall(base)) <= 1 or _PAGEWORDS.search(base):
return _WS.sub(" ", _DIGITS.sub("#", base)).strip()
return base
def _bare_int(line: str):
"""Return the integer value if ``line`` is a BARE page-number-like integer.
"Bare" = a single digit group, NO page-word, and nothing but that integer
plus surrounding punctuation/whitespace — so "5", "4521", "- 12 -" qualify,
but "Page 5" (page-word), "Section 3 intro" (letters) and "1,234.56"
(multi-group) do not. Returns ``None`` otherwise. These bare lines are the
only ones whose page-number collapse can erase real per-page DATA, so they
get the sequentiality guard below; page-word page numbers are always trusted.
"""
base = _WS.sub(" ", line.strip().lower()).strip()
if _PAGEWORDS.search(base):
return None
groups = _DIGITS.findall(base)
if len(groups) != 1:
return None
probe = re.sub(r"[^\w\s]", "", base)
if _DIGITS.sub("", probe).strip() != "": # has letters -> not a bare integer
return None
try:
return int(groups[0])
except ValueError:
return None
def _looks_like_page_numbers(values) -> bool:
"""Do these per-page integer values look like running page numbers?
``values`` is ``None`` for a non-bare-integer key (a page-word page number or
a verbatim text header) — those are trusted and kept. For a BARE-integer band
the values are treated as page numbers ONLY when they are non-decreasing
across pages and span a small range (~ the page count). Arbitrary per-page
data (IDs, measurements, amounts) is neither, so it is preserved rather than
stripped from every page.
"""
if not values: # None (not a bare-int key) or empty -> keep as-is
return True
if len(values) < 2: # too little signal -> fall back to collapsing
return True
non_decreasing = all(values[i] <= values[i + 1] for i in range(len(values) - 1))
return non_decreasing and (max(values) - min(values)) <= 2 * len(values)
def realign_line_confidences(cleaned_text, orig_lines):
"""Map a stripped page's surviving lines back to their confidences.
Stripping only REMOVES whole lines (it never edits or reorders the survivors),
so a two-pointer walk over the original per-line list recovers each surviving
line's confidence. Returns a new ``[{text, confidence}]`` list aligned to
``cleaned_text`` (lines that can't be matched get ``confidence=None``), or
``None`` if there were no original lines.
"""
if not orig_lines:
return None
cleaned = (cleaned_text or "").split("\n")
out, oi = [], 0
for cl in cleaned:
while oi < len(orig_lines) and (orig_lines[oi].get("text", "") != cl):
oi += 1
if oi < len(orig_lines):
out.append(orig_lines[oi])
oi += 1
else:
out.append({"text": cl, "confidence": None})
return out
def _bands(nonblank, band):
"""Disjoint (top, bottom) index lists, capped at half the non-blank lines.
Capping at ``len // 2`` guarantees the two bands never share an index, so
the page interior is never a header/footer candidate.
"""
b = min(band, len(nonblank) // 2)
if b <= 0:
return [], []
return nonblank[:b], nonblank[len(nonblank) - b:]
def strip_running_headers_footers(
page_texts,
*,
band: int = 3,
min_frac: float = 0.7,
min_pages: int = 3,
return_kept: bool = False,
):
"""Return cleaned copies of ``page_texts`` with running headers/footers gone.
Parameters
----------
page_texts : list[str]
One extracted-text string per page, in page order.
band : int
Max non-blank lines at the top and bottom of each page to consider as
header/footer candidates (capped per page so top/bottom never overlap).
min_frac : float
A line must recur in its band on at least ``ceil(min_frac * n_pages)``
pages (min 2) to count as a running header/footer.
min_pages : int
Documents shorter than this are returned unchanged.
return_kept : bool
If True, return ``(cleaned_texts, kept_indices)`` where
``kept_indices[p]`` is the list of ORIGINAL line indices (into
``page_texts[p].split("\\n")``) that survived on page ``p``. This lets a
caller realign per-line metadata (OCR confidences) to the EXACT
surviving line — re-deriving the mapping by text equality misbinds
identical-text duplicate lines. Default False keeps the legacy
list-of-strings return.
"""
n = len(page_texts)
def _all_indices():
return [list(range(len((t or "").split("\n")))) for t in page_texts]
if n < min_pages:
cleaned = list(page_texts)
return (cleaned, _all_indices()) if return_kept else cleaned
pages_lines = []
for txt in page_texts:
lines = (txt or "").split("\n")
nonblank = [i for i, ln in enumerate(lines) if ln.strip()]
pages_lines.append((lines, nonblank))
top_counter, bot_counter = Counter(), Counter()
top_vals, bot_vals = {}, {} # bare-integer key -> per-page int values (page order)
def _tally(indices, lines, counter, vals):
seen = {}
for i in indices:
k = _key(lines[i])
if k and k not in seen:
seen[k] = _bare_int(lines[i])
for k, v in seen.items():
counter[k] += 1
if v is not None:
vals.setdefault(k, []).append(v)
for lines, nonblank in pages_lines:
top, bot = _bands(nonblank, band)
_tally(top, lines, top_counter, top_vals)
_tally(bot, lines, bot_counter, bot_vals)
thresh = max(2, math.ceil(min_frac * n))
running_top = {k for k, c in top_counter.items() if c >= thresh}
running_bot = {k for k, c in bot_counter.items() if c >= thresh}
# Value-aware page-number guard: a running key formed from BARE integers is
# only treated as a running page number when its per-page values look like
# page numbers (see _looks_like_page_numbers). This keeps genuine page
# numbers (1,2,3,…) stripped while PRESERVING distinct per-page numeric DATA
# (IDs, measurements, amounts) that merely recurs in a top/bottom band.
running_top = {k for k in running_top if _looks_like_page_numbers(top_vals.get(k))}
running_bot = {k for k in running_bot if _looks_like_page_numbers(bot_vals.get(k))}
if not running_top and not running_bot:
cleaned = list(page_texts)
return (cleaned, _all_indices()) if return_kept else cleaned
out, kept_all = [], []
for lines, nonblank in pages_lines:
top, bot = _bands(nonblank, band)
strip_idx = set()
for i in top:
if _key(lines[i]) in running_top:
strip_idx.add(i)
for i in bot:
if _key(lines[i]) in running_bot:
strip_idx.add(i)
# Never reduce a page to nothing — if everything would go, keep it as-is.
if nonblank and set(nonblank).issubset(strip_idx):
out.append("\n".join(lines).strip("\n"))
kept_all.append(list(range(len(lines)))) # nothing actually removed
continue
kept_idx = [i for i in range(len(lines)) if i not in strip_idx]
kept = [lines[i] for i in kept_idx]
cleaned = re.sub(r"\n{3,}", "\n\n", "\n".join(kept).strip("\n"))
out.append(cleaned)
kept_all.append(kept_idx)
return (out, kept_all) if return_kept else out