themis / phase1 /scripts /build_held_vectors.py
vg15o2's picture
Moonley backend (HF Space build)
1d9bd9b
Raw
History Blame Contribute Delete
2.51 kB
#!/usr/bin/env python3
"""Build doc-level HELD-headnote vectors (the $0 clean-representation arm).
One BGE-small vector per judgment whose reporter headnote (`held`) exceeds 40 chars —
clean reporter English instead of OCR body chunks. Consumed by Corpus.held_search;
a top-rank hit earns a boost the cross-encoder cannot veto (agent.py doctrine lane).
Previously this script lived only on the Mac; recreated in-repo so the corpus build is
self-contained. Matches the shipped artifact's recipe: iterate meta rows IN FILE ORDER
(including duplicate rows — the shipped 24,327-vector artifact was built that way),
truncate held to 1,800 chars, embed PLAIN (passage side), L2-normalize.
If synthetic_headnotes.jsonl exists (backfill_headnotes.py output), synthetic held
texts are included for docs whose reporter headnote is missing — extending the arm
over the 1970s-80s crater.
Run: python phase1/scripts/build_held_vectors.py [data_dir] (CPU, ~15-30 min)
Out: <data_dir>/held_vectors.npy (float32 L2-normalized), held_docids.json
"""
import json, os, sys, time
import numpy as np
data_dir = sys.argv[1] if len(sys.argv) > 1 else os.environ.get("THEMIS_DATA", "phase1/data/thor_artifacts")
t0 = time.time()
texts, docids, seen_syn = [], [], set()
for line in open(os.path.join(data_dir, "escr_meta.jsonl"), encoding="utf-8"):
m = json.loads(line)
h = str(m.get("held") or "")
if len(h) > 40:
texts.append(h[:1800]); docids.append(m["doc_id"])
syn = os.path.join(data_dir, "synthetic_headnotes.jsonl")
if os.path.exists(syn):
have = set(docids)
for line in open(syn, encoding="utf-8"):
r = json.loads(line)
h = str(r.get("held") or "")
if len(h) > 40 and r["doc_id"] not in have and r["doc_id"] not in seen_syn:
texts.append(h[:1800]); docids.append(r["doc_id"]); seen_syn.add(r["doc_id"])
print(f"[held] +{len(seen_syn)} synthetic headnotes", flush=True)
print(f"[held] embedding {len(texts)} headnotes ...", flush=True)
from sentence_transformers import SentenceTransformer
st = SentenceTransformer("BAAI/bge-small-en-v1.5", device="cpu")
V = st.encode(texts, batch_size=256, normalize_embeddings=True,
convert_to_numpy=True, show_progress_bar=False).astype(np.float32)
np.save(os.path.join(data_dir, "held_vectors.npy"), V)
json.dump(docids, open(os.path.join(data_dir, "held_docids.json"), "w"))
print(f"[held] wrote {V.shape} -> held_vectors.npy + held_docids.json | {time.time()-t0:.0f}s", flush=True)