Spaces:
Running
Running
Download docqa/tests/test_unit.py from validops-east-3/instance-2: direct link, hf CLI and curl.
- Browser
- Download file 6.04 kB
-
https://huggingface.co/spaces/validops-east-3/instance-2/resolve/main/docqa/tests/test_unit.py
- Command line
-
hf download hf://spaces/validops-east-3/instance-2/docqa/tests/test_unit.py
-
curl -L -o test_unit.py https://huggingface.co/spaces/validops-east-3/instance-2/resolve/main/docqa/tests/test_unit.py
6.04 kB
| """Unit tests for the pure, fast layers: parsing, validation, status mapping. | |
| No model inference here, so this file runs in seconds. | |
| """ | |
| from __future__ import annotations | |
| import io | |
| import pytest | |
| from docxextract import parsing | |
| from docxextract.schema import (ExtractRequest, FieldStatus, UnsupportedMediaError, | |
| ValidationError) | |
| # ---------------------------------------------------------------- media sniff | |
| def test_sniff_pdf(): | |
| assert parsing.sniff(b"%PDF-1.7\n...") == "pdf" | |
| def test_sniff_images(data, expected): | |
| assert parsing.sniff(data) == expected | |
| def test_rejects_unknown_type(settings): | |
| with pytest.raises(UnsupportedMediaError): | |
| parsing.parse(b"just some text", settings) | |
| def test_rejects_empty(settings): | |
| with pytest.raises(Exception): | |
| parsing.parse(b"", settings) | |
| # ---------------------------------------------------------------- normalisation | |
| def test_boxes_are_clamped_to_1000(settings): | |
| """LayoutLM indexes learned position embeddings; out-of-range raises.""" | |
| words, boxes = parsing._normalize( | |
| ["a", "b"], [[-50, -50, 5000, 5000], [0, 0, 100, 100]], 100, 100) | |
| assert all(0 <= v <= 1000 for b in boxes for v in b) | |
| def test_boxes_ordered_lowercase_first(): | |
| _words, boxes = parsing._normalize(["a"], [[300, 400, 100, 200]], 1000, 1000) | |
| x0, y0, x1, y1 = boxes[0] | |
| assert x0 <= x1 and y0 <= y1 | |
| def test_zero_size_page_does_not_divide_by_zero(): | |
| words, boxes = parsing._normalize(["a"], [[10, 10, 20, 20]], 0, 0) | |
| assert len(words) == 1 and len(boxes) == 1 | |
| # ---------------------------------------------------------------- parsing | |
| def test_parses_pdf_text_layer(corpus_dir, settings, ground_truth): | |
| from pathlib import Path | |
| data = Path(ground_truth[0]["path"]).read_bytes() | |
| doc = parsing.parse(data, settings) | |
| assert doc.word_count > 50 | |
| assert doc.source_type.value == "pdf_text_layer" | |
| assert doc.page_count >= 1 | |
| def test_parse_cache_hits_on_repeat(corpus_dir, settings, ground_truth): | |
| from pathlib import Path | |
| data = Path(ground_truth[0]["path"]).read_bytes() | |
| first = parsing.parse(data, settings) | |
| before = parsing.cache_stats()["hits"] | |
| second = parsing.parse(data, settings) | |
| assert second.document_id == first.document_id | |
| assert parsing.cache_stats()["hits"] > before | |
| def test_blank_pdf_reports_no_words(tmp_path, settings): | |
| from reportlab.lib.pagesizes import A4 | |
| from reportlab.pdfgen import canvas | |
| pytest.importorskip("pytesseract") | |
| if not parsing.tesseract_available(): | |
| pytest.skip("Tesseract binary not on PATH") | |
| path = tmp_path / "blank.pdf" | |
| c = canvas.Canvas(str(path), pagesize=A4) | |
| c.showPage() | |
| c.save() | |
| doc = parsing.parse(path.read_bytes(), settings) | |
| # Either genuinely empty, or OCR of a blank page yields nothing usable. | |
| assert doc.word_count < 5 | |
| def test_tesseract_probe_does_not_raise(): | |
| """The availability probe must never raise, whatever the environment.""" | |
| assert isinstance(parsing.tesseract_available(), bool) | |
| # ---------------------------------------------------------------- validation | |
| def test_extract_request_rejects_empty_keys(): | |
| with pytest.raises(Exception): | |
| ExtractRequest(keys=["", " "]) | |
| def test_extract_request_deduplicates_preserving_order(): | |
| req = ExtractRequest(keys=["b", "a", "b", "c"]) | |
| assert req.keys == ["b", "a", "c"] | |
| def test_extract_request_rejects_overlong_key(): | |
| with pytest.raises(Exception): | |
| ExtractRequest(keys=["x" * 500]) | |
| # ---------------------------------------------------------------- status mapping | |
| def test_status_below_threshold_is_low_confidence(settings): | |
| from docxextract.engine import Span | |
| from docxextract.service import ExtractionService | |
| service = ExtractionService(settings) | |
| span = Span(key="k", answer="$10.00", confidence=0.2, page=1, start=0, end=0) | |
| field = service._status_for(span, threshold=0.5, page_has_words=True) | |
| assert field.status is FieldStatus.LOW_CONFIDENCE | |
| # The value is still returned so a human reviewer can judge it. | |
| assert field.value == "$10.00" | |
| def test_status_empty_answer_is_not_found(settings): | |
| from docxextract.engine import Span | |
| from docxextract.service import ExtractionService | |
| service = ExtractionService(settings) | |
| span = Span(key="k", answer="", confidence=0.99, page=1, start=-1, end=-1) | |
| field = service._status_for(span, threshold=0.5, page_has_words=True) | |
| assert field.status is FieldStatus.NOT_FOUND | |
| assert field.value is None | |
| def test_status_high_confidence_is_extracted(settings): | |
| from docxextract.engine import Span | |
| from docxextract.service import ExtractionService | |
| service = ExtractionService(settings) | |
| span = Span(key="k", answer="INV-1", confidence=0.95, page=1, start=0, end=0) | |
| field = service._status_for(span, threshold=0.5, page_has_words=True) | |
| assert field.status is FieldStatus.EXTRACTED | |
| assert field.is_usable | |
| def test_service_rejects_no_keys(settings, sample_pdf): | |
| from docxextract.service import ExtractionService | |
| service = ExtractionService(settings) | |
| with pytest.raises(ValidationError): | |
| service.extract(sample_pdf, []) | |
| def test_service_rejects_blank_document(settings, tmp_path): | |
| from reportlab.lib.pagesizes import A4 | |
| from reportlab.pdfgen import canvas | |
| from docxextract.service import ExtractionService | |
| pytest.importorskip("pytesseract") | |
| if not parsing.tesseract_available(): | |
| pytest.skip("Tesseract binary not on PATH") | |
| path = tmp_path / "empty.pdf" | |
| c = canvas.Canvas(str(path), pagesize=A4) | |
| c.showPage() | |
| c.save() | |
| service = ExtractionService(settings) | |
| with pytest.raises(ValidationError): | |
| service.extract(path.read_bytes(), ["Total"]) | |