| """Re-embed the corpus chunks with BGE-small on the GPU to regenerate escr_vectors.npy locally |
| (faster than transferring the 1.9GB float32 matrix over Thor's slow link). Embeds in FILE ORDER so |
| the vector index matches escr_chunks.jsonl line order, exactly as serve.py expects. Documents are |
| embedded PLAIN (no query instruction prefix — that's query-side only).""" |
| import json, time, os |
| import numpy as np |
| from sentence_transformers import SentenceTransformer |
| DATA = os.environ.get("THEMIS_DATA", ".") |
| DEV = os.environ.get("THEMIS_DEVICE", "cuda") |
| texts = [] |
| with open(os.path.join(DATA, "escr_chunks.jsonl"), encoding="utf-8") as f: |
| for l in f: |
| texts.append(json.loads(l)["text"]) |
| print(f"{len(texts)} chunks; embedding on {DEV} ...", flush=True) |
| st = SentenceTransformer("BAAI/bge-small-en-v1.5", device=DEV) |
| t0 = time.time() |
| M = st.encode(texts, batch_size=512, normalize_embeddings=True, convert_to_numpy=True, |
| show_progress_bar=True).astype(np.float32) |
| np.save(os.path.join(DATA, "escr_vectors.npy"), M) |
| print(f"done {M.shape} {M.dtype} in {time.time()-t0:.0f}s -> escr_vectors.npy", flush=True) |
|
|