RandomZ / app /tests /test_parser.py
StormShadow308's picture
- Replaced timezone.utc with UTC from datetime for consistent datetime handling.
faa8fb3
Raw
History Blame Contribute Delete
4.62 kB
"""Unit tests for DOCX and PDF parsers.
LangChain experiment branch: parsers now return
``List[langchain_core.documents.Document]`` instead of ``List[ParsedBlock]``.
"""
from pathlib import Path
import pytest
from langchain_core.documents import Document
# ── DOCX parser tests ─────────────────────────────────────────────────────────
def _make_docx(tmp_path: Path) -> Path:
"""Create a minimal .docx fixture in ``tmp_path``."""
from docx import Document as DocxDocument
doc = DocxDocument()
doc.add_heading("FIXTURE: Test Heading 1", level=1)
doc.add_paragraph("FIXTURE: This is a normal paragraph for testing purposes.")
doc.add_paragraph("FIXTURE: This is a second paragraph.", style="Normal")
table = doc.add_table(rows=2, cols=2)
table.cell(0, 0).text = "FIXTURE Cell A1"
table.cell(0, 1).text = "FIXTURE Cell B1"
table.cell(1, 0).text = "FIXTURE Cell A2"
table.cell(1, 1).text = "FIXTURE Cell B2"
out = tmp_path / "fixture.docx"
doc.save(str(out))
return out
def test_parse_docx_returns_documents(tmp_path: Path) -> None:
"""parse_docx should return a non-empty list of LangChain Documents."""
from app.ingest.parser_docx import parse_docx
docx_path = _make_docx(tmp_path)
docs = parse_docx(docx_path)
assert isinstance(docs, list)
assert len(docs) > 0
assert all(isinstance(d, Document) for d in docs)
def test_parse_docx_has_text_content(tmp_path: Path) -> None:
"""Parsed DOCX documents must have non-empty page_content."""
from app.ingest.parser_docx import parse_docx
docx_path = _make_docx(tmp_path)
docs = parse_docx(docx_path)
combined = " ".join(d.page_content for d in docs)
assert "FIXTURE" in combined, "Expected fixture text not found in parsed content"
def test_parse_docx_has_source_metadata(tmp_path: Path) -> None:
"""Parsed DOCX documents must carry a 'source' metadata key."""
from app.ingest.parser_docx import parse_docx
docx_path = _make_docx(tmp_path)
docs = parse_docx(docx_path)
for doc in docs:
assert "source" in doc.metadata, "Missing 'source' metadata"
def test_parse_docx_missing_file() -> None:
"""parse_docx should raise FileNotFoundError for non-existent file."""
from app.ingest.parser_docx import parse_docx
with pytest.raises(FileNotFoundError):
parse_docx(Path("/nonexistent/file.docx"))
# ── PDF parser tests ──────────────────────────────────────────────────────────
def _make_pdf(tmp_path: Path) -> Path:
"""Create a minimal one-page PDF fixture."""
import fitz # type: ignore[import-untyped]
doc = fitz.open()
page = doc.new_page()
page.insert_text((72, 72), "FIXTURE: PDF Heading Line", fontsize=18)
page.insert_text((72, 110), "FIXTURE: This is a normal paragraph in the PDF.", fontsize=11)
page.insert_text((72, 130), "FIXTURE: Second line of normal body text.", fontsize=11)
out = tmp_path / "fixture.pdf"
doc.save(str(out))
doc.close()
return out
def test_parse_pdf_returns_documents(tmp_path: Path) -> None:
"""parse_pdf should return a non-empty list of LangChain Documents."""
from app.ingest.parser_pdf import parse_pdf
pdf_path = _make_pdf(tmp_path)
docs = parse_pdf(pdf_path)
assert isinstance(docs, list)
assert len(docs) > 0
assert all(isinstance(d, Document) for d in docs)
def test_parse_pdf_has_text_content(tmp_path: Path) -> None:
"""Parsed PDF documents must contain the fixture text."""
from app.ingest.parser_pdf import parse_pdf
pdf_path = _make_pdf(tmp_path)
docs = parse_pdf(pdf_path)
combined = " ".join(d.page_content for d in docs)
assert "FIXTURE" in combined, "Expected fixture text not found in parsed content"
def test_parse_pdf_page_metadata(tmp_path: Path) -> None:
"""PyMuPDFLoader attaches page metadata to each document."""
from app.ingest.parser_pdf import parse_pdf
pdf_path = _make_pdf(tmp_path)
docs = parse_pdf(pdf_path)
for doc in docs:
assert "page" in doc.metadata, "Missing 'page' metadata from PyMuPDFLoader"
def test_parse_pdf_missing_file() -> None:
"""parse_pdf should raise FileNotFoundError for non-existent file."""
from app.ingest.parser_pdf import parse_pdf
with pytest.raises(FileNotFoundError):
parse_pdf(Path("/nonexistent/file.pdf"))