Spaces:
Sleeping
Sleeping
File size: 6,881 Bytes
e86dfae | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 | """
Unit tests for layout-aware PDF extraction.
Two capabilities are covered:
* tables survive extraction as Markdown instead of being flattened into an
unreadable run of cell values
* pages with no extractable text are routed to OCR, within a budget, and
degrade quietly when Tesseract is absent
"""
import os
from unittest.mock import patch
import pytest
from app.services.pdf_layout import PageExtraction, extract_page, table_to_markdown
pytestmark = pytest.mark.unit
# ── Table rendering ────────────────────────────────────────────────────────────
def test_table_renders_as_markdown_with_header_separator():
md = table_to_markdown([["Height", "Bolt"], ["4 ft", "36 in"], ["6 ft", "48 in"]])
lines = md.splitlines()
assert lines[0] == "| Height | Bolt |"
assert lines[1] == "| --- | --- |"
assert lines[2] == "| 4 ft | 36 in |"
def test_row_column_relationship_is_preserved():
"""The whole point: a value must stay associated with its row and column."""
md = table_to_markdown(
[
["Mining Height", "Bolt Length"],
["under 4 ft", "36 in"],
["6 to 8 ft", "60 in"],
]
)
# The 60 in row must be readable as belonging to 6 to 8 ft.
assert "| 6 to 8 ft | 60 in |" in md
def test_none_cells_become_empty_not_the_string_none():
md = table_to_markdown([["A", "B"], ["x", None]])
assert "None" not in md
assert "| x | |" in md
def test_ragged_rows_are_padded():
md = table_to_markdown([["A", "B", "C"], ["1"], ["1", "2"]])
for line in md.splitlines():
assert line.count("|") == 4
def test_embedded_pipes_cannot_break_the_table():
md = table_to_markdown([["Header"], ["a | b"]])
assert r"a \| b" in md
def test_embedded_newlines_are_flattened():
md = table_to_markdown([["Header"], ["line1\nline2"]])
assert "line1 line2" in md
assert len(md.splitlines()) == 3
def test_fully_empty_rows_are_dropped():
md = table_to_markdown([["A", "B"], ["", ""], ["1", "2"]])
assert len(md.splitlines()) == 3 # header, separator, one data row
def test_empty_table_returns_empty_string():
assert table_to_markdown([]) == ""
assert table_to_markdown([["", ""]]) == ""
# ── Page extraction / OCR routing ──────────────────────────────────────────────
class FakePage:
def __init__(self, text="", tables=None):
self._text = text
self._tables = tables or []
def extract_text(self):
return self._text
def extract_tables(self):
return self._tables
def test_text_page_is_not_sent_to_ocr():
page = FakePage(text="x" * 500)
with patch("app.services.pdf_layout.ocr_page") as ocr:
result = extract_page(page, "f.pdf", 1, ocr_budget=[10])
ocr.assert_not_called()
assert not result.used_ocr
def test_scanned_page_is_sent_to_ocr():
"""A page yielding nothing is the signature of a scan."""
page = FakePage(text="")
with patch("app.services.pdf_layout.ocr_page", return_value="RECOVERED TEXT"):
result = extract_page(page, "f.pdf", 3, ocr_budget=[10])
assert result.used_ocr
assert "RECOVERED TEXT" in result.text
def test_ocr_budget_is_consumed_and_enforced():
page = FakePage(text="")
budget = [2]
with patch("app.services.pdf_layout.ocr_page", return_value="TEXT"):
assert extract_page(page, "f.pdf", 1, budget).used_ocr
assert extract_page(page, "f.pdf", 2, budget).used_ocr
# Budget exhausted — third scanned page is not OCR'd.
assert not extract_page(page, "f.pdf", 3, budget).used_ocr
assert budget == [0]
def test_ocr_failure_degrades_quietly():
"""No Tesseract binary must not raise — the document still ingests."""
page = FakePage(text="")
with patch("app.services.pdf_layout.ocr_page", return_value=""):
result = extract_page(page, "f.pdf", 1, ocr_budget=[5])
assert result.used_ocr is False
assert result.text == ""
def test_ocr_supplements_rather_than_replaces_sparse_text():
page = FakePage(text="Figure 4.")
with patch("app.services.pdf_layout.ocr_page", return_value="scanned body"):
result = extract_page(page, "f.pdf", 1, ocr_budget=[5])
assert "Figure 4." in result.text
assert "scanned body" in result.text
def test_tables_are_appended_and_labelled_with_page():
page = FakePage(text="Prose here.", tables=[[["A", "B"], ["1", "2"]]])
result = extract_page(page, "f.pdf", 7, ocr_budget=[5])
assert "Prose here." in result.text
assert "[Table 1 on page 7]" in result.text
assert "| A | B |" in result.text
assert result.tables_found == 1
def test_table_extraction_error_does_not_fail_the_page():
class Exploding(FakePage):
def extract_tables(self):
raise RuntimeError("malformed table object")
result = extract_page(Exploding(text="Prose"), "f.pdf", 1, ocr_budget=[5])
assert result.text == "Prose"
assert result.tables_found == 0
def test_table_can_rescue_a_page_from_being_treated_as_scanned():
"""A page that is only a table has little prose but is not a scan."""
big_table = [["Col"] * 4] + [[f"v{i}{j}" for j in range(4)] for i in range(12)]
page = FakePage(text="", tables=[big_table])
with patch("app.services.pdf_layout.ocr_page") as ocr:
result = extract_page(page, "f.pdf", 1, ocr_budget=[5])
ocr.assert_not_called()
assert result.tables_found == 1
# ── End-to-end against a real PDF ──────────────────────────────────────────────
FIXTURE = os.path.join(
os.path.dirname(__file__), "..", "fixtures", "roof_control_table.pdf"
)
@pytest.mark.skipif(not os.path.exists(FIXTURE), reason="fixture PDF not built")
def test_real_pdf_table_survives_extraction():
pytest.importorskip("pdfplumber")
from app.services.extractors import PDFExtractor
doc = PDFExtractor().extract(FIXTURE)
assert "| Mining Height | Bolt Length | Spacing | Support Type |" in doc.full_text
assert "| 6 to 8 ft | 60 in | 3.5 ft | Mechanical |" in doc.full_text
assert doc.metadata["tables_extracted"] == 1
@pytest.mark.skipif(not os.path.exists(FIXTURE), reason="fixture PDF not built")
def test_pypdf_fallback_loses_the_table_structure():
"""
Documents the regression this replaces: the old path cannot answer
"what bolt length for 6 to 8 ft?" because the association is gone.
"""
from app.services.extractors import PDFExtractor
text = PDFExtractor()._extract_with_pypdf(FIXTURE).full_text
assert "| 6 to 8 ft | 60 in |" not in text
|