| from pathlib import Path |
|
|
| from backend.database.schemas import Chunk |
|
|
|
|
| DOC_LANGUAGES = {"markdown", "text", "yaml", "json", "env"} |
|
|
|
|
| def fallback_chunk_file(path: Path, repo_root: Path, repo_id: str, language: str) -> list[Chunk]: |
| text = path.read_text(encoding="utf-8", errors="ignore") |
| chunk_size = 700 |
| overlap = 80 if language in DOC_LANGUAGES else 20 |
| return split_text_into_chunks(text, path, repo_root, repo_id, language, chunk_size, overlap) |
|
|
|
|
| def split_text_into_chunks( |
| text: str, |
| path: Path, |
| repo_root: Path, |
| repo_id: str, |
| language: str, |
| chunk_size: int, |
| overlap: int, |
| ) -> list[Chunk]: |
| words = text.split() |
| lines = text.splitlines() |
| rel = path.relative_to(repo_root).as_posix() |
| if not words: |
| return [ |
| Chunk( |
| id=f"{repo_id}:{rel}:1", |
| repo_id=repo_id, |
| path=rel, |
| language=language, |
| start_line=1, |
| end_line=max(1, len(lines)), |
| content=text, |
| ) |
| ] |
|
|
| chunks: list[Chunk] = [] |
| step = max(1, chunk_size - overlap) |
| for index, start in enumerate(range(0, len(words), step), start=1): |
| window = words[start : start + chunk_size] |
| content = " ".join(window) |
| start_line = _estimate_line_for_word(text, start) |
| end_line = _estimate_line_for_word(text, min(start + len(window), len(words))) |
| chunks.append( |
| Chunk( |
| id=f"{repo_id}:{rel}:fallback:{index}", |
| repo_id=repo_id, |
| path=rel, |
| language=language, |
| symbol=f"chunk_{index}", |
| kind="text" if language in DOC_LANGUAGES else "module", |
| start_line=start_line, |
| end_line=max(start_line, end_line), |
| content=content, |
| ) |
| ) |
| return chunks |
|
|
|
|
| def _estimate_line_for_word(text: str, word_index: int) -> int: |
| if word_index <= 0: |
| return 1 |
| seen = 0 |
| for line_number, line in enumerate(text.splitlines(), start=1): |
| seen += len(line.split()) |
| if seen >= word_index: |
| return line_number |
| return max(1, len(text.splitlines())) |
|
|