#!/usr/bin/env python3 """Build doc-level HELD-headnote vectors (the $0 clean-representation arm). One BGE-small vector per judgment whose reporter headnote (`held`) exceeds 40 chars — clean reporter English instead of OCR body chunks. Consumed by Corpus.held_search; a top-rank hit earns a boost the cross-encoder cannot veto (agent.py doctrine lane). Previously this script lived only on the Mac; recreated in-repo so the corpus build is self-contained. Matches the shipped artifact's recipe: iterate meta rows IN FILE ORDER (including duplicate rows — the shipped 24,327-vector artifact was built that way), truncate held to 1,800 chars, embed PLAIN (passage side), L2-normalize. If synthetic_headnotes.jsonl exists (backfill_headnotes.py output), synthetic held texts are included for docs whose reporter headnote is missing — extending the arm over the 1970s-80s crater. Run: python phase1/scripts/build_held_vectors.py [data_dir] (CPU, ~15-30 min) Out: /held_vectors.npy (float32 L2-normalized), held_docids.json """ import json, os, sys, time import numpy as np data_dir = sys.argv[1] if len(sys.argv) > 1 else os.environ.get("THEMIS_DATA", "phase1/data/thor_artifacts") t0 = time.time() texts, docids, seen_syn = [], [], set() for line in open(os.path.join(data_dir, "escr_meta.jsonl"), encoding="utf-8"): m = json.loads(line) h = str(m.get("held") or "") if len(h) > 40: texts.append(h[:1800]); docids.append(m["doc_id"]) syn = os.path.join(data_dir, "synthetic_headnotes.jsonl") if os.path.exists(syn): have = set(docids) for line in open(syn, encoding="utf-8"): r = json.loads(line) h = str(r.get("held") or "") if len(h) > 40 and r["doc_id"] not in have and r["doc_id"] not in seen_syn: texts.append(h[:1800]); docids.append(r["doc_id"]); seen_syn.add(r["doc_id"]) print(f"[held] +{len(seen_syn)} synthetic headnotes", flush=True) print(f"[held] embedding {len(texts)} headnotes ...", flush=True) from sentence_transformers import SentenceTransformer st = SentenceTransformer("BAAI/bge-small-en-v1.5", device="cpu") V = st.encode(texts, batch_size=256, normalize_embeddings=True, convert_to_numpy=True, show_progress_bar=False).astype(np.float32) np.save(os.path.join(data_dir, "held_vectors.npy"), V) json.dump(docids, open(os.path.join(data_dir, "held_docids.json"), "w")) print(f"[held] wrote {V.shape} -> held_vectors.npy + held_docids.json | {time.time()-t0:.0f}s", flush=True)