"""Unit tests for the pure, fast layers: parsing, validation, status mapping. No model inference here, so this file runs in seconds. """ from __future__ import annotations import io import pytest from docxextract import parsing from docxextract.schema import (ExtractRequest, FieldStatus, UnsupportedMediaError, ValidationError) # ---------------------------------------------------------------- media sniff def test_sniff_pdf(): assert parsing.sniff(b"%PDF-1.7\n...") == "pdf" @pytest.mark.parametrize("data,expected", [ (b"\x89PNG\r\n\x1a\n", "png"), (b"\xff\xd8\xff\xe0", "jpeg"), (b"GIF89a...", "gif"), (b"BM....", "bmp"), (b"not a document", "unknown"), ]) def test_sniff_images(data, expected): assert parsing.sniff(data) == expected def test_rejects_unknown_type(settings): with pytest.raises(UnsupportedMediaError): parsing.parse(b"just some text", settings) def test_rejects_empty(settings): with pytest.raises(Exception): parsing.parse(b"", settings) # ---------------------------------------------------------------- normalisation def test_boxes_are_clamped_to_1000(settings): """LayoutLM indexes learned position embeddings; out-of-range raises.""" words, boxes = parsing._normalize( ["a", "b"], [[-50, -50, 5000, 5000], [0, 0, 100, 100]], 100, 100) assert all(0 <= v <= 1000 for b in boxes for v in b) def test_boxes_ordered_lowercase_first(): _words, boxes = parsing._normalize(["a"], [[300, 400, 100, 200]], 1000, 1000) x0, y0, x1, y1 = boxes[0] assert x0 <= x1 and y0 <= y1 def test_zero_size_page_does_not_divide_by_zero(): words, boxes = parsing._normalize(["a"], [[10, 10, 20, 20]], 0, 0) assert len(words) == 1 and len(boxes) == 1 # ---------------------------------------------------------------- parsing def test_parses_pdf_text_layer(corpus_dir, settings, ground_truth): from pathlib import Path data = Path(ground_truth[0]["path"]).read_bytes() doc = parsing.parse(data, settings) assert doc.word_count > 50 assert doc.source_type.value == "pdf_text_layer" assert doc.page_count >= 1 def test_parse_cache_hits_on_repeat(corpus_dir, settings, ground_truth): from pathlib import Path data = Path(ground_truth[0]["path"]).read_bytes() first = parsing.parse(data, settings) before = parsing.cache_stats()["hits"] second = parsing.parse(data, settings) assert second.document_id == first.document_id assert parsing.cache_stats()["hits"] > before def test_blank_pdf_reports_no_words(tmp_path, settings): from reportlab.lib.pagesizes import A4 from reportlab.pdfgen import canvas pytest.importorskip("pytesseract") if not parsing.tesseract_available(): pytest.skip("Tesseract binary not on PATH") path = tmp_path / "blank.pdf" c = canvas.Canvas(str(path), pagesize=A4) c.showPage() c.save() doc = parsing.parse(path.read_bytes(), settings) # Either genuinely empty, or OCR of a blank page yields nothing usable. assert doc.word_count < 5 def test_tesseract_probe_does_not_raise(): """The availability probe must never raise, whatever the environment.""" assert isinstance(parsing.tesseract_available(), bool) # ---------------------------------------------------------------- validation def test_extract_request_rejects_empty_keys(): with pytest.raises(Exception): ExtractRequest(keys=["", " "]) def test_extract_request_deduplicates_preserving_order(): req = ExtractRequest(keys=["b", "a", "b", "c"]) assert req.keys == ["b", "a", "c"] def test_extract_request_rejects_overlong_key(): with pytest.raises(Exception): ExtractRequest(keys=["x" * 500]) # ---------------------------------------------------------------- status mapping def test_status_below_threshold_is_low_confidence(settings): from docxextract.engine import Span from docxextract.service import ExtractionService service = ExtractionService(settings) span = Span(key="k", answer="$10.00", confidence=0.2, page=1, start=0, end=0) field = service._status_for(span, threshold=0.5, page_has_words=True) assert field.status is FieldStatus.LOW_CONFIDENCE # The value is still returned so a human reviewer can judge it. assert field.value == "$10.00" def test_status_empty_answer_is_not_found(settings): from docxextract.engine import Span from docxextract.service import ExtractionService service = ExtractionService(settings) span = Span(key="k", answer="", confidence=0.99, page=1, start=-1, end=-1) field = service._status_for(span, threshold=0.5, page_has_words=True) assert field.status is FieldStatus.NOT_FOUND assert field.value is None def test_status_high_confidence_is_extracted(settings): from docxextract.engine import Span from docxextract.service import ExtractionService service = ExtractionService(settings) span = Span(key="k", answer="INV-1", confidence=0.95, page=1, start=0, end=0) field = service._status_for(span, threshold=0.5, page_has_words=True) assert field.status is FieldStatus.EXTRACTED assert field.is_usable def test_service_rejects_no_keys(settings, sample_pdf): from docxextract.service import ExtractionService service = ExtractionService(settings) with pytest.raises(ValidationError): service.extract(sample_pdf, []) def test_service_rejects_blank_document(settings, tmp_path): from reportlab.lib.pagesizes import A4 from reportlab.pdfgen import canvas from docxextract.service import ExtractionService pytest.importorskip("pytesseract") if not parsing.tesseract_available(): pytest.skip("Tesseract binary not on PATH") path = tmp_path / "empty.pdf" c = canvas.Canvas(str(path), pagesize=A4) c.showPage() c.save() service = ExtractionService(settings) with pytest.raises(ValidationError): service.extract(path.read_bytes(), ["Total"])