pay-equity-for-eu / scripts /build_index.py
Franskaman110's picture
RAG pipeline
5af2f86
Raw
History Blame Contribute Delete
1.93 kB
"""Offline index builder — run once on ZeroGPU, commit the output.
Embeds the full corpus (EU Directive articles + IDA salary rows) with the
pinned embedding model and persists a FAISS index under
``data/processed/index/``. The embedding step is wrapped in ``@spaces.GPU`` so,
when run on the ZeroGPU Space, it uses the GPU. Commit the resulting
``index.faiss`` + ``chunks.jsonl`` so the live app only *loads* them — runtime
never re-embeds the corpus (keeps generation smooth).
Usage (on the ZeroGPU Space / any GPU machine)::
uv run python scripts/build_index.py
Requires ``embeddings.repo`` set in ``configs/models.yaml``.
"""
from __future__ import annotations
import sys
from pathlib import Path
import spaces
# Allow running as a plain script (`python scripts/build_index.py`).
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from src.backend.indexing.embedder import Embedder # noqa: E402
from src.backend.indexing.ingest import ( # noqa: E402
build_corpus,
load_directive_chunks,
load_ida_chunks,
)
from src.backend.indexing.store import DEFAULT_INDEX_DIR, VectorStore # noqa: E402
@spaces.GPU(duration=300)
def _embed_corpus(texts: list[str]):
"""Embed all passages on ZeroGPU (model loads here, in GPU context)."""
embedder = Embedder.get()
print(f"embedding with {embedder.repo} (loader={embedder.loader}) …")
return embedder.embed_passages(texts)
def main() -> None:
n_dir = len(load_directive_chunks())
n_ida = len(load_ida_chunks())
chunks = build_corpus()
print(f"ingested: directive={n_dir} ida={n_ida} total={len(chunks)}")
vectors = _embed_corpus([c.text for c in chunks])
dim = vectors.shape[1] if vectors.size else 0
store = VectorStore()
store.add(chunks, vectors)
store.persist()
print(f"persisted {len(store)} vectors (dim={dim}) → {DEFAULT_INDEX_DIR}")
if __name__ == "__main__":
main()