"""Unit tests for DOCX and PDF parsers. LangChain experiment branch: parsers now return ``List[langchain_core.documents.Document]`` instead of ``List[ParsedBlock]``. """ from pathlib import Path import pytest from langchain_core.documents import Document # ── DOCX parser tests ───────────────────────────────────────────────────────── def _make_docx(tmp_path: Path) -> Path: """Create a minimal .docx fixture in ``tmp_path``.""" from docx import Document as DocxDocument doc = DocxDocument() doc.add_heading("FIXTURE: Test Heading 1", level=1) doc.add_paragraph("FIXTURE: This is a normal paragraph for testing purposes.") doc.add_paragraph("FIXTURE: This is a second paragraph.", style="Normal") table = doc.add_table(rows=2, cols=2) table.cell(0, 0).text = "FIXTURE Cell A1" table.cell(0, 1).text = "FIXTURE Cell B1" table.cell(1, 0).text = "FIXTURE Cell A2" table.cell(1, 1).text = "FIXTURE Cell B2" out = tmp_path / "fixture.docx" doc.save(str(out)) return out def test_parse_docx_returns_documents(tmp_path: Path) -> None: """parse_docx should return a non-empty list of LangChain Documents.""" from app.ingest.parser_docx import parse_docx docx_path = _make_docx(tmp_path) docs = parse_docx(docx_path) assert isinstance(docs, list) assert len(docs) > 0 assert all(isinstance(d, Document) for d in docs) def test_parse_docx_has_text_content(tmp_path: Path) -> None: """Parsed DOCX documents must have non-empty page_content.""" from app.ingest.parser_docx import parse_docx docx_path = _make_docx(tmp_path) docs = parse_docx(docx_path) combined = " ".join(d.page_content for d in docs) assert "FIXTURE" in combined, "Expected fixture text not found in parsed content" def test_parse_docx_has_source_metadata(tmp_path: Path) -> None: """Parsed DOCX documents must carry a 'source' metadata key.""" from app.ingest.parser_docx import parse_docx docx_path = _make_docx(tmp_path) docs = parse_docx(docx_path) for doc in docs: assert "source" in doc.metadata, "Missing 'source' metadata" def test_parse_docx_missing_file() -> None: """parse_docx should raise FileNotFoundError for non-existent file.""" from app.ingest.parser_docx import parse_docx with pytest.raises(FileNotFoundError): parse_docx(Path("/nonexistent/file.docx")) # ── PDF parser tests ────────────────────────────────────────────────────────── def _make_pdf(tmp_path: Path) -> Path: """Create a minimal one-page PDF fixture.""" import fitz # type: ignore[import-untyped] doc = fitz.open() page = doc.new_page() page.insert_text((72, 72), "FIXTURE: PDF Heading Line", fontsize=18) page.insert_text((72, 110), "FIXTURE: This is a normal paragraph in the PDF.", fontsize=11) page.insert_text((72, 130), "FIXTURE: Second line of normal body text.", fontsize=11) out = tmp_path / "fixture.pdf" doc.save(str(out)) doc.close() return out def test_parse_pdf_returns_documents(tmp_path: Path) -> None: """parse_pdf should return a non-empty list of LangChain Documents.""" from app.ingest.parser_pdf import parse_pdf pdf_path = _make_pdf(tmp_path) docs = parse_pdf(pdf_path) assert isinstance(docs, list) assert len(docs) > 0 assert all(isinstance(d, Document) for d in docs) def test_parse_pdf_has_text_content(tmp_path: Path) -> None: """Parsed PDF documents must contain the fixture text.""" from app.ingest.parser_pdf import parse_pdf pdf_path = _make_pdf(tmp_path) docs = parse_pdf(pdf_path) combined = " ".join(d.page_content for d in docs) assert "FIXTURE" in combined, "Expected fixture text not found in parsed content" def test_parse_pdf_page_metadata(tmp_path: Path) -> None: """PyMuPDFLoader attaches page metadata to each document.""" from app.ingest.parser_pdf import parse_pdf pdf_path = _make_pdf(tmp_path) docs = parse_pdf(pdf_path) for doc in docs: assert "page" in doc.metadata, "Missing 'page' metadata from PyMuPDFLoader" def test_parse_pdf_missing_file() -> None: """parse_pdf should raise FileNotFoundError for non-existent file.""" from app.ingest.parser_pdf import parse_pdf with pytest.raises(FileNotFoundError): parse_pdf(Path("/nonexistent/file.pdf"))