| |
| """Build doc-level HELD-headnote vectors (the $0 clean-representation arm). |
| |
| One BGE-small vector per judgment whose reporter headnote (`held`) exceeds 40 chars — |
| clean reporter English instead of OCR body chunks. Consumed by Corpus.held_search; |
| a top-rank hit earns a boost the cross-encoder cannot veto (agent.py doctrine lane). |
| |
| Previously this script lived only on the Mac; recreated in-repo so the corpus build is |
| self-contained. Matches the shipped artifact's recipe: iterate meta rows IN FILE ORDER |
| (including duplicate rows — the shipped 24,327-vector artifact was built that way), |
| truncate held to 1,800 chars, embed PLAIN (passage side), L2-normalize. |
| |
| If synthetic_headnotes.jsonl exists (backfill_headnotes.py output), synthetic held |
| texts are included for docs whose reporter headnote is missing — extending the arm |
| over the 1970s-80s crater. |
| |
| Run: python phase1/scripts/build_held_vectors.py [data_dir] (CPU, ~15-30 min) |
| Out: <data_dir>/held_vectors.npy (float32 L2-normalized), held_docids.json |
| """ |
| import json, os, sys, time |
|
|
| import numpy as np |
|
|
| data_dir = sys.argv[1] if len(sys.argv) > 1 else os.environ.get("THEMIS_DATA", "phase1/data/thor_artifacts") |
| t0 = time.time() |
|
|
| texts, docids, seen_syn = [], [], set() |
| for line in open(os.path.join(data_dir, "escr_meta.jsonl"), encoding="utf-8"): |
| m = json.loads(line) |
| h = str(m.get("held") or "") |
| if len(h) > 40: |
| texts.append(h[:1800]); docids.append(m["doc_id"]) |
|
|
| syn = os.path.join(data_dir, "synthetic_headnotes.jsonl") |
| if os.path.exists(syn): |
| have = set(docids) |
| for line in open(syn, encoding="utf-8"): |
| r = json.loads(line) |
| h = str(r.get("held") or "") |
| if len(h) > 40 and r["doc_id"] not in have and r["doc_id"] not in seen_syn: |
| texts.append(h[:1800]); docids.append(r["doc_id"]); seen_syn.add(r["doc_id"]) |
| print(f"[held] +{len(seen_syn)} synthetic headnotes", flush=True) |
|
|
| print(f"[held] embedding {len(texts)} headnotes ...", flush=True) |
| from sentence_transformers import SentenceTransformer |
| st = SentenceTransformer("BAAI/bge-small-en-v1.5", device="cpu") |
| V = st.encode(texts, batch_size=256, normalize_embeddings=True, |
| convert_to_numpy=True, show_progress_bar=False).astype(np.float32) |
| np.save(os.path.join(data_dir, "held_vectors.npy"), V) |
| json.dump(docids, open(os.path.join(data_dir, "held_docids.json"), "w")) |
| print(f"[held] wrote {V.shape} -> held_vectors.npy + held_docids.json | {time.time()-t0:.0f}s", flush=True) |
|
|