Spaces:
Sleeping
Sleeping
| """Vector store for RAG — pure-Python, numpy cosine similarity. | |
| Deliberately dependency-light: no native build (chroma-hnswlib needs a C++ | |
| compiler) and no external service. Each document's chunks + embeddings are | |
| persisted to a single .npz/.json pair on disk and cached in memory. | |
| Chunks carry page/bbox provenance so retrieval returns citations the UI can | |
| highlight. To scale up, swap this module's body for pgvector or Chroma — the | |
| public interface (index_chunks / query / delete / Chunk) stays the same. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| from dataclasses import dataclass, asdict | |
| from pathlib import Path | |
| from typing import Optional | |
| from app.core.config import settings | |
| from app.core.logging import get_logger | |
| log = get_logger(__name__) | |
| try: | |
| import numpy as np | |
| except Exception: # pragma: no cover | |
| np = None | |
| class Chunk: | |
| id: str | |
| text: str | |
| page: int | |
| doc_id: str | |
| bbox: Optional[dict] = None | |
| # in-memory cache: doc_id -> (matrix[N,d], list[Chunk]) | |
| _cache: dict[str, tuple] = {} | |
| def _store_dir() -> Path: | |
| p = Path(settings.chroma_dir) | |
| p.mkdir(parents=True, exist_ok=True) | |
| return p | |
| def _paths(doc_id: str) -> tuple[Path, Path]: | |
| d = _store_dir() | |
| return d / f"{doc_id}.npy", d / f"{doc_id}.json" | |
| def _load(doc_id: str): | |
| if doc_id in _cache: | |
| return _cache[doc_id] | |
| vec_path, meta_path = _paths(doc_id) | |
| if not vec_path.exists() or not meta_path.exists(): | |
| return None | |
| try: | |
| mat = np.load(vec_path) | |
| chunks = [Chunk(**c) for c in json.loads(meta_path.read_text("utf-8"))] | |
| _cache[doc_id] = (mat, chunks) | |
| return _cache[doc_id] | |
| except Exception as e: # pragma: no cover | |
| log.warning("vectorstore load failed for %s: %s", doc_id, e) | |
| return None | |
| def _normalize(mat): | |
| norms = np.linalg.norm(mat, axis=1, keepdims=True) | |
| norms[norms == 0] = 1.0 | |
| return mat / norms | |
| async def index_chunks(doc_id: str, chunks: list[Chunk]) -> None: | |
| if not chunks: | |
| return | |
| if np is None: | |
| raise RuntimeError("numpy is not installed") | |
| from app.llm.registry import get_embedder | |
| embedder = get_embedder() | |
| vectors = await embedder.embed([c.text for c in chunks]) | |
| mat = np.asarray(vectors, dtype="float32") | |
| vec_path, meta_path = _paths(doc_id) | |
| np.save(vec_path, mat) | |
| meta_path.write_text( | |
| json.dumps([asdict(c) for c in chunks], ensure_ascii=False), "utf-8" | |
| ) | |
| _cache[doc_id] = (mat, chunks) | |
| log.info("Indexed %d chunks for doc %s (dim=%d)", len(chunks), doc_id, | |
| mat.shape[1] if mat.ndim == 2 else 0) | |
| async def query(doc_id: str, question: str, k: int = 5) -> list[Chunk]: | |
| loaded = _load(doc_id) | |
| if not loaded: | |
| return [] | |
| mat, chunks = loaded | |
| if mat.size == 0 or not chunks: | |
| return [] | |
| from app.llm.registry import get_embedder | |
| embedder = get_embedder() | |
| qvec = np.asarray((await embedder.embed([question]))[0], dtype="float32") | |
| sims = _normalize(mat) @ (qvec / (np.linalg.norm(qvec) or 1.0)) | |
| top = np.argsort(-sims)[: min(k, len(chunks))] | |
| return [chunks[int(i)] for i in top] | |
| def delete(doc_id: str) -> None: | |
| _cache.pop(doc_id, None) | |
| for p in _paths(doc_id): | |
| try: | |
| p.unlink(missing_ok=True) | |
| except Exception: | |
| pass | |