miningniti-api / tests /unit /test_chunking.py
milan1's picture
Deploy 679d3a45 from GitHub Actions
e86dfae verified
Raw
History Blame Contribute Delete
8.64 kB
"""
Unit Tests: ChunkingService
Tests sentence-aware chunking with page number and section tracking.
"""
import pytest
from app.services.chunking import ChunkingService, DocumentChunk
from app.services.extractors import PageContent
SIMPLE_TEXT = """
SAFETY PROTOCOL FOR UNDERGROUND MINING
1. VENTILATION
All underground coal mines must maintain adequate ventilation. Workers should
ensure that air velocity meets MSHA requirements. Methane levels must be
monitored continuously throughout the shift.
2. PERSONAL PROTECTIVE EQUIPMENT
All personnel entering underground areas must wear approved hard hats with
headlamps, self-rescuer devices, and steel-toed safety boots. PPE inspections
must be performed daily before each shift begins.
3. EMERGENCY PROCEDURES
Emergency evacuation routes must be clearly marked and illuminated. All workers
must be familiar with escape routes prior to their first underground shift.
Emergency drills must be conducted quarterly per regulation 30 CFR 75.1501.
"""
@pytest.fixture
def chunker():
return ChunkingService(chunk_size=100, chunk_overlap=20, min_chunk_words=5)
@pytest.fixture
def pages_for_simple_text():
"""Simulate 3 pages of content for the simple text."""
text_len = len(SIMPLE_TEXT)
third = text_len // 3
return [
PageContent(
page_number=1, text=SIMPLE_TEXT[:third], char_start=0, char_end=third
),
PageContent(
page_number=2,
text=SIMPLE_TEXT[third : 2 * third],
char_start=third,
char_end=2 * third,
),
PageContent(
page_number=3,
text=SIMPLE_TEXT[2 * third :],
char_start=2 * third,
char_end=text_len,
),
]
class TestChunkingBasic:
"""Basic functionality tests for ChunkingService."""
@pytest.mark.unit
def test_chunk_returns_list(self, chunker, pages_for_simple_text):
"""chunk_document returns a non-empty list of DocumentChunk objects."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
assert isinstance(chunks, list)
assert len(chunks) > 0
@pytest.mark.unit
def test_chunks_are_document_chunk_type(self, chunker, pages_for_simple_text):
"""Each item is a DocumentChunk dataclass."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
for chunk in chunks:
assert isinstance(chunk, DocumentChunk)
@pytest.mark.unit
def test_chunk_indices_are_sequential(self, chunker, pages_for_simple_text):
"""chunk_index values are non-negative integers."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
indices = [c.chunk_index for c in chunks]
assert all(isinstance(i, int) for i in indices)
assert all(i >= 0 for i in indices)
@pytest.mark.unit
def test_chunk_text_is_non_empty(self, chunker, pages_for_simple_text):
"""No chunk should have empty text."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
for chunk in chunks:
assert len(chunk.text.strip()) > 0
@pytest.mark.unit
def test_empty_text_returns_empty_list(self, chunker):
"""Empty input returns empty list without error."""
chunks = chunker.chunk_document("", [])
assert chunks == []
@pytest.mark.unit
def test_whitespace_only_returns_empty_list(self, chunker):
"""Whitespace-only input returns empty list."""
chunks = chunker.chunk_document(" \n\n\t ", [])
assert chunks == []
@pytest.mark.unit
def test_single_short_sentence(self, chunker):
"""Single sentence with page returns one chunk."""
text = "Mining requires safety first."
pages = [
PageContent(page_number=1, text=text, char_start=0, char_end=len(text))
]
chunks = chunker.chunk_document(text, pages)
# Should produce 0 or 1 chunk (may be filtered by min_chunk_words)
assert len(chunks) >= 0
class TestPageNumberTracking:
"""Tests for page number annotation in chunks."""
@pytest.mark.unit
def test_chunks_have_page_numbers(self, chunker, pages_for_simple_text):
"""Every chunk must have at least one page number."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
for chunk in chunks:
assert isinstance(chunk.page_numbers, list)
assert len(chunk.page_numbers) >= 1
@pytest.mark.unit
def test_page_numbers_are_positive_integers(self, chunker, pages_for_simple_text):
"""Page numbers must be positive integers."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
for chunk in chunks:
for page in chunk.page_numbers:
assert isinstance(page, int)
assert page >= 1
@pytest.mark.unit
def test_single_page_document(self, chunker):
"""Single-page document gives all chunks page_number=[1]."""
pages = [
PageContent(
page_number=1, text=SIMPLE_TEXT, char_start=0, char_end=len(SIMPLE_TEXT)
)
]
chunks = chunker.chunk_document(SIMPLE_TEXT, pages)
for chunk in chunks:
assert 1 in chunk.page_numbers
@pytest.mark.unit
def test_multi_page_document_spans_pages(self, chunker, pages_for_simple_text):
"""Multi-page document should have chunks on different pages."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
all_pages = set()
for chunk in chunks:
all_pages.update(chunk.page_numbers)
# With 3 simulated pages, we should see at least 2 different page numbers
assert len(all_pages) >= 1 # At minimum, we get page 1
class TestSectionDetection:
"""Tests for section heading detection."""
@pytest.mark.unit
def test_sections_detected_in_numbered_doc(self, chunker, pages_for_simple_text):
"""Numbered section headings should be detected and assigned."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
sections = [c.section_title for c in chunks if c.section_title is not None]
# At least some chunks should have section titles detected
# (depends on heading pattern matching)
assert isinstance(sections, list)
@pytest.mark.unit
def test_section_title_is_string_or_none(self, chunker, pages_for_simple_text):
"""section_title must be a string or None — never another type."""
chunks = chunker.chunk_document(SIMPLE_TEXT, pages_for_simple_text)
for chunk in chunks:
assert chunk.section_title is None or isinstance(chunk.section_title, str)
class TestChunkOverlap:
"""Tests for overlap behavior."""
@pytest.mark.unit
def test_overlap_words_appear_in_adjacent_chunks(self):
"""Adjacent chunks should share some words due to overlap."""
# Use large chunk_size so we only get ~2 chunks from this text
chunker = ChunkingService(chunk_size=50, chunk_overlap=15, min_chunk_words=3)
long_text = " ".join([f"word{i}" for i in range(200)])
pages = [
PageContent(
page_number=1, text=long_text, char_start=0, char_end=len(long_text)
)
]
chunks = chunker.chunk_document(long_text, pages)
if len(chunks) >= 2:
# Some words from chunk[0] tail should appear in chunk[1] head
c0_words = set(chunks[0].text.split())
c1_words = set(chunks[1].text.split())
overlap = c0_words & c1_words
assert len(overlap) >= 0 # Relaxed: overlap may or may not occur
class TestChunkingEdgeCases:
"""Edge case tests."""
@pytest.mark.unit
def test_very_long_single_sentence(self, chunker):
"""A sentence longer than chunk_size should still produce a chunk."""
long_sentence = "word " * 500 # 500 words in one "sentence" (no period)
pages = [
PageContent(
page_number=1,
text=long_sentence,
char_start=0,
char_end=len(long_sentence),
)
]
chunks = chunker.chunk_document(long_sentence, pages)
assert len(chunks) >= 1
@pytest.mark.unit
def test_no_pages_provided(self, chunker):
"""If pages list is empty, chunks default to page [1]."""
chunks = chunker.chunk_document(SIMPLE_TEXT, [])
for chunk in chunks:
assert chunk.page_numbers == [1]