RICS / app /tests /test_chunker.py
StormShadow308's picture
- Updated .gitignore to exclude proprietary RICS documents.
d611f45
Raw
History Blame Contribute Delete
3.05 kB
"""Unit tests for the LangChain-based text splitter.
LangChain experiment branch: the splitter now operates on
``List[langchain_core.documents.Document]`` and returns the same type.
The old ``split_chunks()`` / ``ParsedBlock`` / ``Chunk`` API is replaced
by ``split_documents()`` using ``RecursiveCharacterTextSplitter``.
"""
from langchain_core.documents import Document
from app.chunking.splitter import count_tokens, split_documents
def _make_docs(texts: list[str]) -> list[Document]:
return [Document(page_content=t, metadata={"source": "test"}) for t in texts]
def test_count_tokens_basic() -> None:
"""count_tokens should return a positive integer for non-empty strings."""
assert count_tokens("hello world") > 0
assert count_tokens("") == 0
def test_split_documents_returns_documents() -> None:
"""split_documents should return a list of LangChain Document objects."""
docs = _make_docs(["This is a short test paragraph."])
chunks = split_documents(docs)
assert isinstance(chunks, list)
assert len(chunks) >= 1
assert all(isinstance(c, Document) for c in chunks)
def test_split_documents_inherits_metadata() -> None:
"""Each chunk Document should carry the parent's metadata."""
docs = _make_docs(["Short paragraph."])
docs[0].metadata["custom_key"] = "custom_value"
chunks = split_documents(docs)
for chunk in chunks:
assert chunk.metadata.get("custom_key") == "custom_value"
assert chunk.metadata.get("source") == "test"
def test_split_documents_respects_chunk_size() -> None:
"""No chunk should substantially exceed the configured chunk_size in tokens."""
long_text = " ".join(["word"] * 1000)
docs = _make_docs([long_text])
chunk_size = 100
chunks = split_documents(docs, chunk_size=chunk_size, chunk_overlap=10)
for chunk in chunks:
actual_tokens = count_tokens(chunk.page_content)
# Allow a small overshoot due to overlap prepending
assert actual_tokens <= chunk_size + 25, (
f"Chunk too large: {actual_tokens} tokens (limit {chunk_size + 25})"
)
def test_split_documents_empty_input() -> None:
"""split_documents with an empty list should return an empty list."""
assert split_documents([]) == []
def test_split_documents_large_document() -> None:
"""split_documents should handle a large document without error."""
docs = _make_docs([" ".join(["word"] * 200) for _ in range(50)])
chunks = split_documents(docs, chunk_size=300)
assert len(chunks) > 0
def test_split_documents_multiple_pages() -> None:
"""Each page document should be independently split."""
docs = [
Document(page_content="Page one content. " * 10, metadata={"page": 1}),
Document(page_content="Page two content. " * 10, metadata={"page": 2}),
]
chunks = split_documents(docs, chunk_size=50, chunk_overlap=10)
assert len(chunks) >= 2
# Each chunk should have non-empty page_content
assert all(c.page_content.strip() for c in chunks)