"""Regression tests for OCR normalization, table preservation, and chunking. Covers the spec's remaining failure modes: - OCR contamination (running headers/footers, page numbers, hyphenation) - malformed sentences (hard-wrap unwrap) - missing tables / chunk-boundary corruption (tables kept intact, never split) All deterministic and dependency-light (no live PDF / network needed). """ from __future__ import annotations from langchain_core.documents import Document from app.chunking.splitter import split_documents from app.ingest.ocr_normalize import ( TABLE_CLOSE, TABLE_OPEN, normalize_pages, normalize_text, strip_running_headers_footers, ) from app.ingest.parser_pdf import _table_to_markdown # ── OCR normalization ─────────────────────────────────────────────────────── def test_dehyphenation_repairs_wrapped_words(): out = normalize_text("The roof is in poor condi-\ntion at the ridge.") assert "condition" in out assert "condi-" not in out def test_unwrap_restores_sentence_continuity(): raw = "The main roof covering\nis natural slate and\nremains in sound condition." out = normalize_text(raw) assert out == "The main roof covering is natural slate and remains in sound condition." def test_paragraph_breaks_preserved(): raw = "First paragraph line one\nline two.\n\nSecond paragraph here." out = normalize_text(raw) assert out == "First paragraph line one line two.\n\nSecond paragraph here." def test_table_block_preserved_verbatim_through_normalization(): table = f"{TABLE_OPEN}\n| A | B |\n| --- | --- |\n| 1 | 2 |\n{TABLE_CLOSE}" raw = f"Some intro text that is\nhard wrapped.\n\n{table}\n\nTrailing note." out = normalize_text(raw) assert table in out # untouched assert "intro text that is hard wrapped." in out # prose still reflowed def test_running_headers_and_page_numbers_stripped(): pages = [ "ACME Surveyors Ltd\nRoof section content for page one.\nPage 1 of 3", "ACME Surveyors Ltd\nDrainage content for page two.\nPage 2 of 3", "ACME Surveyors Ltd\nElectrical content for page three.\nPage 3 of 3", ] cleaned = strip_running_headers_footers(pages) joined = "\n".join(cleaned) assert "ACME Surveyors Ltd" not in joined # repeated header removed assert "Page 1 of 3" not in joined # page numbers removed assert "Roof section content" in joined # real content kept def test_normalize_pages_single_page_keeps_content(): pages = ["Only one page here.\nWith a wrapped line."] out = normalize_pages(pages) assert out == ["Only one page here. With a wrapped line."] # ── Table rendering ───────────────────────────────────────────────────────── def test_table_to_markdown_handles_none_and_pipes(): rows = [["Element", "Rating"], ["Roof | main", None], ["Drains", "2"]] md = _table_to_markdown(rows) lines = md.splitlines() assert lines[0] == "| Element | Rating |" assert lines[1] == "| --- | --- |" assert r"Roof \| main" in md # pipe escaped assert "| Drains | 2 |" in lines[-1] def test_table_to_markdown_pads_ragged_rows(): md = _table_to_markdown([["A", "B", "C"], ["1"]]) # ragged body row padded to header width assert "| 1 | | |" in md # ── Table-aware chunking ──────────────────────────────────────────────────── def _table(rows: int) -> str: body = "\n".join(f"| Element{i} | Rating{i} | Cost{i} |" for i in range(rows)) return f"{TABLE_OPEN}\n| Element | Rating | Cost |\n| --- | --- | --- |\n{body}\n{TABLE_CLOSE}" def test_table_kept_as_single_chunk(): table = _table(rows=3) doc = Document(page_content=f"Intro prose.\n\n{table}\n\nOutro prose.", metadata={"doc": "d1"}) chunks = split_documents([doc], chunk_size=200, chunk_overlap=0) table_chunks = [c for c in chunks if c.metadata.get("section_type") == "table"] assert len(table_chunks) == 1 assert table_chunks[0].page_content.startswith(TABLE_OPEN) assert table_chunks[0].page_content.rstrip().endswith(TABLE_CLOSE) def test_large_table_not_split_across_chunks(): # A table far larger than chunk_size must still emit exactly one chunk. table = _table(rows=60) doc = Document(page_content=f"Heading.\n\n{table}", metadata={}) chunks = split_documents([doc], chunk_size=100, chunk_overlap=0) table_chunks = [c for c in chunks if c.metadata.get("section_type") == "table"] assert len(table_chunks) == 1 assert table_chunks[0].page_content.count(TABLE_OPEN) == 1 assert table_chunks[0].page_content.count(TABLE_CLOSE) == 1 def test_prose_without_table_unaffected(): doc = Document(page_content="A simple paragraph of prose with no tables.", metadata={}) chunks = split_documents([doc], chunk_size=200, chunk_overlap=0) assert len(chunks) == 1 assert chunks[0].metadata.get("section_type") != "table" def test_table_and_prose_order_preserved(): table = _table(rows=2) doc = Document(page_content=f"Before text.\n\n{table}\n\nAfter text.", metadata={}) chunks = split_documents([doc], chunk_size=300, chunk_overlap=0) kinds = ["table" if c.metadata.get("section_type") == "table" else "prose" for c in chunks] assert "table" in kinds # the table chunk is between prose chunks t_idx = kinds.index("table") assert any(k == "prose" for k in kinds[:t_idx]) assert any(k == "prose" for k in kinds[t_idx + 1:])