| |
| """Build corpus_ledger.jsonl — one row of identity + health facts per unique judgment. |
| |
| Fixes recorded (not applied to the source artifacts — the ledger is a sidecar the |
| serving layer reads): |
| - decision_year vs scr_volume_year (meta `year` is the reporter-volume year; wrong |
| for filters/boosts in ~24% of docs) |
| - duplicate meta rows (43,175 lines -> 37,898 unique doc_ids; dup_rows counted) |
| - text health: stored chars vs official SCR page span (from escr_pdfmap path |
| YYYY_VOL_START_END) + garbled-token noise rate |
| - bench_n: bench_strength string -> int (the runtime previously did int("division") |
| -> always 0, silently killing the bench boost/filter) |
| - sibling clusters: referral orders / main judgments / reviews of the same case share |
| party names within a few years; cluster them and mark the CANONICAL member (longest |
| text) so lookups prefer the judgment over its 5-page order. |
| |
| Run: python phase1/scripts/build_ledger.py [data_dir] (~3 min, CPU only) |
| Out: <data_dir>/corpus_ledger.jsonl |
| """ |
| import json, os, re, sys, time |
| from collections import Counter, defaultdict |
|
|
| data_dir = sys.argv[1] if len(sys.argv) > 1 else os.environ.get("THEMIS_DATA", "phase1/data/thor_artifacts") |
| t0 = time.time() |
|
|
| BENCH_N = {"single": 1, "division": 2, "full": 3, "constitution": 5, "larger": 7} |
| _STOP = {"the", "of", "and", "state", "union", "india", "others", "ors", "anr", "another", |
| "etc", "through", "lrs", "dead", "smt", "shri", "sri", "mst", "dr", "m/s", "vs", "v"} |
|
|
| def party_tokens(name): |
| toks = set() |
| for side in re.split(r"\bv\.?s?\b|\bversus\b", (name or "").lower())[:2]: |
| for w in re.findall(r"[a-z]+", side): |
| if len(w) >= 4 and w not in _STOP: |
| toks.add(w) |
| return toks |
|
|
| |
| meta, dup_rows = {}, Counter() |
| for line in open(os.path.join(data_dir, "escr_meta.jsonl"), encoding="utf-8"): |
| m = json.loads(line) |
| d = m["doc_id"] |
| if d in meta: dup_rows[d] += 1 |
| else: meta[d] = m |
| print(f"[ledger] meta: {len(meta)} unique docs, {sum(dup_rows.values())} duplicate rows", flush=True) |
|
|
| |
| span = {} |
| if os.path.exists(os.path.join(data_dir, "escr_pdfmap.jsonl")): |
| for line in open(os.path.join(data_dir, "escr_pdfmap.jsonl"), encoding="utf-8"): |
| r = json.loads(line) |
| p = r["path"].split("_") |
| if len(p) == 4 and p[2].isdigit() and p[3].isdigit(): |
| pages = int(p[3]) - int(p[2]) + 1 |
| if 0 < pages < 3000: span[r["doc_id"]] = pages |
|
|
| |
| chars, noise_num, noise_den = Counter(), Counter(), Counter() |
| pat = re.compile(r'"doc_id":\s*"([^"]+)"') |
| mixed = re.compile(r"[a-zA-Z]\d|\d[a-zA-Z]") |
| ctrl = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f~·]") |
| ok2 = {"a","an","is","of","to","in","by","on","at","or","as","it","be","we","he","no","so","if","do","us","up"} |
| for line in open(os.path.join(data_dir, "escr_chunks.jsonl"), encoding="utf-8"): |
| d = pat.search(line[:120]).group(1) |
| i = line.find('"text":') |
| t = line[i + 9:-3] |
| chars[d] += len(t) |
| toks = t.split() |
| if toks: |
| bad = 0 |
| for w in toks: |
| if mixed.search(w) or ctrl.search(w): bad += 1 |
| elif len(w) <= 2 and w.isalpha() and w.lower() not in ok2: bad += 1 |
| noise_num[d] += bad; noise_den[d] += len(toks) |
| print(f"[ledger] chunk pass done, {time.time()-t0:.0f}s", flush=True) |
|
|
| |
| ptoks = {d: party_tokens(m.get("case_name")) for d, m in meta.items()} |
| def dyear(d): |
| dt = meta[d].get("date") or "" |
| return int(dt[:4]) if dt[:4].isdigit() else 0 |
|
|
| block = defaultdict(list) |
| for d, ts in ptoks.items(): |
| for t in ts: block[t].append(d) |
|
|
| parent = {d: d for d in meta} |
| def find(x): |
| while parent[x] != x: |
| parent[x] = parent[parent[x]]; x = parent[x] |
| return x |
| def union(a, b): |
| ra, rb = find(a), find(b) |
| if ra != rb: parent[rb] = ra |
|
|
| pairs_checked = 0 |
| for t, docs in block.items(): |
| if len(docs) > 40: continue |
| for i in range(len(docs)): |
| for j in range(i + 1, len(docs)): |
| a, b = docs[i], docs[j] |
| if abs(dyear(a) - dyear(b)) > 6: continue |
| ta, tb = ptoks[a], ptoks[b] |
| if not ta or not tb: continue |
| ov = len(ta & tb) / min(len(ta), len(tb)) |
| if ov >= 0.8: |
| union(a, b); pairs_checked += 1 |
|
|
| clusters = defaultdict(list) |
| for d in meta: clusters[find(d)].append(d) |
| cluster_id, canonical = {}, {} |
| n_multi = 0 |
| for root, members in clusters.items(): |
| cid = f"c{abs(hash(root)) % 10**9}" if len(members) > 1 else None |
| if len(members) > 1: |
| n_multi += 1 |
| canon = max(members, key=lambda d: chars.get(d, 0)) |
| for d in members: |
| cluster_id[d] = cid; canonical[d] = (d == canon) |
| print(f"[ledger] sibling clusters: {n_multi} multi-doc clusters " |
| f"({sum(len(v) for v in clusters.values() if len(v)>1)} docs), {time.time()-t0:.0f}s", flush=True) |
|
|
| |
| out = os.path.join(data_dir, "corpus_ledger.jsonl") |
| n_mismatch = 0 |
| with open(out, "w", encoding="utf-8") as f: |
| for d, m in meta.items(): |
| dy = dyear(d) |
| try: vy = int(str(m.get("year") or "")[:4]) |
| except Exception: vy = 0 |
| ch = chars.get(d, 0) |
| pg = span.get(d) |
| cpp = round(ch / pg, 1) if pg else None |
| health = "empty" if ch == 0 else ("thin" if (cpp is not None and cpp < 800 and pg >= 5) else "ok") |
| mism = bool(dy and vy and dy != vy); n_mismatch += mism |
| row = {"doc_id": d, "decision_year": dy or None, "scr_volume_year": vy or None, |
| "year_mismatch": mism, "dup_rows": dup_rows.get(d, 0), |
| "chars": ch, "official_pages": pg, "chars_per_page": cpp, "text_health": health, |
| "held_len": len(str(m.get("held") or "")), |
| "noise_rate": round(noise_num.get(d, 0) / noise_den[d], 4) if noise_den.get(d) else None, |
| "bench_n": BENCH_N.get(str(m.get("bench_strength") or "").lower(), 0), |
| "cluster_id": cluster_id.get(d), "canonical": canonical.get(d, True)} |
| f.write(json.dumps(row, ensure_ascii=False) + "\n") |
|
|
| print(f"[ledger] wrote {len(meta)} rows -> {out}", flush=True) |
| print(f"[ledger] year_mismatch: {n_mismatch} ({n_mismatch/len(meta):.1%}) | " |
| f"empty: {sum(1 for d in meta if chars.get(d,0)==0)} | " |
| f"dup rows total: {sum(dup_rows.values())} | {time.time()-t0:.0f}s", flush=True) |
|
|