File size: 2,905 Bytes
9f6ffb8
 
 
 
 
 
 
 
 
 
 
83dc8bf
9f6ffb8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
"""
Sentence Window Chunking Strategy.

For long documents:
1. Splits text into sentences.
2. For each sentence, attaches a ±1 sentence window of surrounding context.
3. Incorporates 10-20% token overlap across sentence boundaries so cross-boundary queries retain continuity.
"""

from typing import Any, Dict, List
import config
from doc_chunking.metadata import (
    Chunk,
    split_sentences_multilingual,
    calculate_overlap_tokens,
    estimate_token_count,
)


def chunk_document_sentence_window(
    doc_dict: Dict[str, Any], window_size: int = config.SENTENCE_WINDOW_SIZE
) -> List[Chunk]:
    """
    Split a long document into sentence-window chunks with ±window_size surrounding context
    and 10-20% boundary overlap.
    """
    full_text = doc_dict.get("text", "")
    doc_id = doc_dict.get("doc_id", "doc_unknown")
    source_lang = doc_dict.get("source_lang", "en")
    title = doc_dict.get("title", "")
    
    sentences = split_sentences_multilingual(full_text)
    if not sentences:
        return []
        
    chunks: List[Chunk] = []
    prev_tail_overlap = ""
    
    for i, center_sent in enumerate(sentences):
        # Determine window boundaries
        start_idx = max(0, i - window_size)
        end_idx = min(len(sentences), i + window_size + 1)
        
        # Build window context
        window_sentences = sentences[start_idx:end_idx]
        raw_window_text = " ".join(window_sentences)
        
        # Prepend overlap from previous sentence window boundary if available
        if prev_tail_overlap:
            stitched_text = f"{prev_tail_overlap} {raw_window_text}"
        else:
            stitched_text = raw_window_text
            
        chunk_id = f"{doc_id}_sw_{i:04d}"
        
        chunk = Chunk(
            chunk_id=chunk_id,
            text=stitched_text,
            embed_text=center_sent,  # Focused central sentence for precise embedding
            chunk_strategy="sentence_window",
            source_lang=source_lang,
            token_count=estimate_token_count(stitched_text),
            doc_id=doc_id,
            context_window=raw_window_text,
            metadata={
                "sentence_index": i,
                "total_sentences": len(sentences),
                "title": title,
                "center_sentence": center_sent,
            },
        )
        chunks.append(chunk)
        
        # Calculate 15% overlap for next chunk
        prev_tail_overlap = calculate_overlap_tokens(
            raw_window_text, overlap_percent=config.CHUNK_OVERLAP_PERCENT
        )
        
    return chunks


def process_longdocs_sentence_window(longdocs: List[Dict[str, Any]]) -> List[Chunk]:
    """
    Process a collection of long documents using sentence-window chunking.
    """
    all_chunks = []
    for doc in longdocs:
        all_chunks.extend(chunk_document_sentence_window(doc))
    return all_chunks