File size: 15,138 Bytes
9e07f5b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
"""CP5 + Stage-2 β€” deterministic good-law citator AND citation-graph builder (no DeepSeek).

Tier-1 (good-law): scan each judgment for explicit overruling/doubting DECLARATIONS, resolve the
named case via the citation index, record a NEGATIVE edge (bench + date), derive status by latching
on the strongest *competent* negative (fail-safe to 'doubted'; default 'unknown'; never good-law from
absence).

Stage-2 (the citation GRAPH that feeds search): harvest every inbound edge and PERSIST it.
  - cite edges: in-text reporter citations (expanded regex: bare/no-vol SCR, Supp. volumes, year-first
    AIR), resolved cross-reporter via parallel-citation union-find.
  - NAME edges: cases cited BY NAME ("Khushal Rao v State of Bombay") β€” the dominant way foundational
    cases are referenced (Khushal Rao: graph cited_by 0 via regex, but named in ~39 bodies). Resolved
    via a distinctive-party name index. THIS is the foundational-recall fix.
  - dedup by (from,target); a body cite that resolves to NO corpus doc is dropped, never a dangling edge.

In:  escr_corpus_full.jsonl
Out: good_law.jsonl  (doc_id, status, provenance, as_of, cited_by, treatment_breakdown)
     edges.jsonl     (from, target, treatment, method, para)   <- the graph
Run on Thor:  CORPUS=escr_corpus_full.jsonl python3 21_citator.py
"""
import json, os, re
from collections import defaultdict, Counter

CORPUS = os.getenv("CORPUS", "escr_corpus_full.jsonl")
OUT = os.getenv("OUT", "good_law.jsonl")
EDGES_OUT = os.getenv("EDGES_OUT", "edges.jsonl")

BENCH_RANK = {"single": 1, "division": 2, "full": 3, "constitution": 5, "larger": 7}
def rank(b):
    if b is None: return None
    if isinstance(b, int): return b
    if b in BENCH_RANK: return BENCH_RANK[b]
    m = re.match(r"\d+", str(b)); return int(m.group()) if m else None

NEG_PATTERNS = [
    (re.compile(r"\b(?:is|are|stand[s]?|hereby|hereby\s+)?\s*overrul(?:e|ed|es|ing)\b", re.I), "overruled"),
    (re.compile(r"\bwe\s+overrule\b", re.I), "overruled"),
    (re.compile(r"\bno longer\s+(?:good law|the law|holds the field|holds good)\b", re.I), "overruled"),
    (re.compile(r"\b(?:cannot|can no longer)\s+be\s+(?:considered|treated|regarded)\s+(?:as\s+)?good law\b", re.I), "overruled"),
    (re.compile(r"\bdoes not (?:lay down|state) the correct law\b", re.I), "overruled"),
    (re.compile(r"\b(?:partly|partially)\s+overrul", re.I), "partly_overruled"),
    (re.compile(r"\bper incuriam\b", re.I), "per_incuriam"),
    (re.compile(r"\bdoubt(?:ed|s)?\s+the correctness\b", re.I), "doubted"),
    (re.compile(r"\bcorrectness\s+(?:of|.{0,40}?)\s+(?:is\s+)?doubt", re.I), "doubted"),
    (re.compile(r"\breferred?\s+to\s+a\s+larger\s+bench\b", re.I), "doubted"),
]
SEVERITY = {"overruled": 4, "partly_overruled": 3, "per_incuriam": 3, "doubted": 2}
POS_LEX = [("relied", "relied_on"), ("followed", "followed"), ("approved", "approved"),
           ("affirmed", "affirmed"), ("reiterated", "followed"), ("distinguish", "distinguished"),
           ("referred", "referred")]
TREAT_PRI = {"relied_on": 5, "followed": 4, "approved": 4, "affirmed": 4,
             "distinguished": 3, "referred": 2, "cited": 1, "named": 1}

# expanded citation pattern β€” catches the forms the old regex missed (bare/no-vol SCR, Supp., year-first AIR)
_C = (r"\[?\d{4}\]?\s*(?:supp\.?\s*)?\d*\s*S\.?C\.?R\.?\s*\d+"
      r"|\(\d{4}\)\s*(?:supp\.?\s*)?\d+\s*SCC\s*\d+"
      r"|AIR\s+\d{4}\s+SC\s+\d+|\d{4}\s+AIR\s+(?:SC\s+)?\d+|\d{4}\s+INSC\s+\d+")
CITE = re.compile(_C, re.I)
ONE = r"(?:" + _C + r")"
PARALLEL = re.compile(ONE + r"(?:\s*[:;]\s*" + ONE + r")+", re.I)

def norm_cite(c): return re.sub(r"\s+", " ", (c or "").replace(".", "")).strip().upper()  # dots stripped: S.C.R.==SCR

# name-index: distinctive party tokens -> doc (drops generic govt/state/place parties so 'State of X' is not a key)
NAME_STOP = set("v vs versus of the and in re m s smt sri shri dr ms mr kumari ors anr etc another others "
                "state union india govt government through rep by its secretary ministry law justice department "
                "maharashtra punjab gujarat rajasthan bombay delhi kerala karnataka bihar uttar pradesh madhya "
                "tamil nadu andhra telangana bengal west haryana assam odisha orissa jharkhand chhattisgarh "
                "himachal uttarakhand goa manipur tripura nagaland mizoram sikkim meghalaya nct calcutta madras "
                "allahabad company ltd limited pvt private corporation board authority commissioner".split())
def distinctive_key(party):
    toks = [t for t in re.findall(r"[a-z]+", (party or "").lower()) if t not in NAME_STOP and len(t) >= 3]
    if not toks: return None
    if len(toks) >= 2 or len(toks[0]) >= 7:       # 2+ distinctive tokens, or one long surname/entity
        return " ".join(toks[:4])
    return None
# Require a CITATION CUE before the name β€” a real case reference reads "in/following/see/decision in
# <Name> v <Name>", not bare "X v Y" prose (which gave ~50% false matches). The cue gates the harvest.
NAMECITE = re.compile(
    r"(?:\bin|\bsee|\bper|\bfollowing|\breiterated in|\breaffirmed in|\brelied (?:up)?on(?: in)?|"
    r"\bdecisions?\s+in|\bjudgments?\s+in|\bheld\s+in|\blaid\s+down\s+in|\bobserved\s+in|\bcase\s+of|\bratio\s+(?:in|of))\s+"
    r"([A-Z][A-Za-z.&'’\-]+(?:\s+[A-Z][A-Za-z.&'’\-]+){0,4})\s+v(?:s|ersus)?\.?\s+"
    r"([A-Z][A-Za-z.&'’\-]+(?:\s+[A-Za-z.&'’\-]+){0,3})", re.I)
def second_party(nm):
    p = re.split(r"\s+v[s.]?\s+|\s+versus\s+", nm or "", 1, flags=re.I)
    return p[1] if len(p) > 1 else ""
def party_tokens(s): return {t for t in re.findall(r"[a-z]+", (s or "").lower()) if t not in NAME_STOP and len(t) >= 4}

print("loading corpus + building indexes...", flush=True)
docs, seen_ids = [], set()
cite2doc = {}
for l in open(CORPUS):
    r = json.loads(l); did = r.get("doc_id")
    if did in seen_ids: continue                  # dedup duplicate doc_id lines (root-cause of node double-counting)
    seen_ids.add(did); docs.append(r)
    for key in ([r.get("neutral_citation")] + (r.get("equivalent_citations") or [])):
        if key: cite2doc[norm_cite(key)] = did
by_id = {r.get("doc_id"): r for r in docs}

keydocs = defaultdict(list)                        # distinctive key -> [doc_ids]; name_index resolved later by cite-popularity
for r in docs:
    first = re.split(r"\s+v[s.]?\s+|\s+versus\s+", r.get("case_name") or "", 1, flags=re.I)[0]
    k = distinctive_key(first)
    if k: keydocs[k].append(r.get("doc_id"))
print(f"{len(docs)} docs (deduped), {len(cite2doc)} citation keys, {len(keydocs)} name keys", flush=True)

# Snapshot the RELIABLE base index (each doc's own neutral+equivalent cites) BEFORE union-find
# enrichment. High-stakes Tier-1 negative-edge resolution uses ONLY this β€” the union-find can
# over-merge different cases that co-occur in ';'-separated citation lists, and a false overruling
# (e.g. Royappa) is catastrophic. The enriched index is used only for the low-stakes recall graph.
cite2doc_base = dict(cite2doc)

# --- cross-reporter enrichment via parallel-citation union-find ---
parent = {}
def _year(c):
    m = re.search(r"\d{4}", c or ""); return m.group() if m else None
def _find(x):
    parent.setdefault(x, x)
    while parent[x] != x: parent[x] = parent[parent[x]]; x = parent[x]
    return x
def _union(a, b): parent[_find(a)] = _find(b)
def _group(cites):
    cs = [norm_cite(c) for c in cites if c]
    for c in cs[1:]: _union(cs[0], c)
for r in docs:
    for m in PARALLEL.finditer(r.get("full_text") or ""):       # only merge SAME-YEAR cites (true parallel cites
        cs = re.split(r"\s*[:;]\s*", m.group(0))                 # share a year; a ';'-list of different cases does not)
        byy = defaultdict(list)
        for c in cs:
            y = _year(c)
            if y: byy[y].append(c)
        for grp in byy.values(): _group(grp)
    for ed in (r.get("cases_cited") or []):                      # cases_cited[].citations ARE one case's parallels β€” safe
        _group(ed.get("citations") or [])
    _group([r.get("neutral_citation")] + (r.get("equivalent_citations") or []))
groups = defaultdict(list)
for c in list(parent): groups[_find(c)].append(c)
added = 0
for g in groups.values():
    doc = next((cite2doc[c] for c in g if c in cite2doc), None)
    if doc:
        for c in g:
            if c not in cite2doc: cite2doc[c] = doc; added += 1
print(f"cross-reporter enrichment: +{added} keys -> {len(cite2doc)} total", flush=True)

def resolve_neg(ft, vs, ve, radius=70):
    """Tier-1 negative-edge targeting β€” UNIQUENESS guard (precision over recall). The overruled case
    must be the SINGLE case unambiguously tied to the verb: collect every resolvable target within a
    tight window (base citation OR distinctive case-name) and flag it ONLY if exactly one distinct doc
    qualifies. Ambiguous (a good-law case cited near the overruled one, e.g. Maneka near ADM Jabalpur
    in Puttaswamy) or none -> attribute NOTHING. Asymmetric cost: a false overruling is catastrophic;
    a missed one defaults to honest 'unknown' (which now renders as nothing)."""
    # DIRECTION guard: "X was overruled IN/BY <Name>" names the OVERRULER after in/by β€” never the target.
    overruler_after = re.match(r"\s*(?:in|by)\b", ft[ve: ve + 40].lower())
    base = max(0, vs - radius)
    seg = ft[base: ve + radius]
    targets = set()
    def consider(pos, d):
        if not d: return
        if overruler_after and pos >= ve: return        # a case after "overruled in/by" is the overruler, skip
        targets.add(d)
    for m in CITE.finditer(seg):
        consider(base + m.start(), cite2doc_base.get(norm_cite(m.group(0))))  # base index only, cite-anchored (high precision)
    return (next(iter(targets)), None) if len(targets) == 1 else (None, None)

def window_treatment(win):
    w = win.lower()
    for kw, lab in POS_LEX:
        if kw in w: return lab
    return "cited"

# --- harvest edges: cite + name ---
edges = {}                                        # (from, target) -> {treatment, method, para}
def add_edge(frm, tgt, tre, method, para):
    if not tgt or frm == tgt: return
    k = (frm, tgt); cur = edges.get(k)
    if cur is None:
        edges[k] = {"treatment": tre, "method": method, "para": para[:160]}
    else:
        if TREAT_PRI.get(tre, 0) > TREAT_PRI.get(cur["treatment"], 0): cur["treatment"] = tre  # keep strongest (fixes sticky bug)
        if cur["method"] == "name" and method == "cite": cur["method"] = "cite"; cur["para"] = para[:160]

neg_edges = defaultdict(list)
cite_indeg = Counter()
n_decl = n_name = 0
# PASS A β€” Tier-1 negatives + CITE edges (reliable; also yields cite-popularity to disambiguate names)
for r in docs:
    ft = r.get("full_text") or ""
    self_id = r.get("doc_id")
    cb, cd = rank(r.get("bench_strength")), r.get("date")
    for pat, kind in NEG_PATTERNS:
        for m in pat.finditer(ft):
            tgt, c = resolve_neg(ft, m.start(), m.end())
            if tgt and tgt != self_id:
                neg_edges[tgt].append({"from": self_id, "from_cite": r.get("neutral_citation"),
                                       "kind": kind, "bench": cb, "date": cd,
                                       "passage": re.sub(r"\s+", " ", ft[max(0, m.start() - 170): m.end() + 170])}); n_decl += 1
    for m in CITE.finditer(ft):
        tgt = cite2doc.get(norm_cite(m.group(0)))
        if not tgt or tgt == self_id: continue
        win = ft[max(0, m.start() - 120): m.end() + 120]
        add_edge(self_id, tgt, window_treatment(win), "cite", re.sub(r"\s+", " ", win))
        cite_indeg[tgt] += 1

# resolve each name key to the MOST cite-cited candidate (the famous case among namesakes) β€” fixes the 53% ambiguity
name_index = {k: (ds[0] if len(ds) == 1 else max(ds, key=lambda d: (cite_indeg.get(d, 0), d))) for k, ds in keydocs.items()}

# PASS B β€” NAME edges (cue-required NAMECITE gates false matches; ambiguous keys -> most-cited candidate)
for r in docs:
    ft = r.get("full_text") or ""
    self_id = r.get("doc_id")
    for m in NAMECITE.finditer(ft):
        k = distinctive_key(m.group(1)); tgt = name_index.get(k) if k else None
        if not tgt or tgt == self_id: continue
        cs = party_tokens(m.group(2)); ts = party_tokens(second_party(by_id.get(tgt, {}).get("case_name")))
        if cs and ts and not (cs & ts): continue   # second parties both distinctive but disjoint -> wrong same-surname case
        win = ft[max(0, m.start() - 60): m.start() + 200]; tre = window_treatment(win)
        add_edge(self_id, tgt, tre if tre != "cited" else "named", "name", re.sub(r"\s+", " ", win)); n_name += 1

cited_by = defaultdict(dict)
for (frm, tgt), e in edges.items():
    cited_by[tgt][frm] = e["treatment"]
print(f"tier-1 neg edges: {n_decl} | total graph edges: {len(edges)} (name-mentions scanned: {n_name})", flush=True)

def derive(did):
    tgt = by_id[did]; tb = rank(tgt.get("bench_strength")); td = tgt.get("date")
    best = None
    for e in neg_edges.get(did, []):
        if td and e["date"] and e["date"] <= td: continue
        competent = (e["bench"] is not None and tb is not None and e["bench"] >= tb)
        sev = SEVERITY.get(e["kind"], 0)
        if best is None or sev > best[0]: best = (sev, e["kind"], e, competent)
    if best is None:
        return {"good_law_status": "unknown", "provenance": "no negative treatment found", "as_of": None}
    sev, kind, e, competent = best; asof = e["date"]
    extra = {"passage": e.get("passage", ""), "overruling": e.get("from"), "overruling_cite": e.get("from_cite")}
    if kind == "doubted":
        return {"good_law_status": "doubted", "provenance": f"doubted by {e['from_cite']}", "as_of": asof, **extra}
    if kind == "per_incuriam":
        return {"good_law_status": "per_incuriam", "provenance": f"held per incuriam in {e['from_cite']}", "as_of": asof, **extra}
    if not competent:
        return {"good_law_status": "doubted", "provenance": f"negative treatment in {e['from_cite']} β€” overruling bench not confirmed >= target",
                "as_of": asof, "needs_review": True, **extra}
    return {"good_law_status": kind, "provenance": f"{kind} by {e['from_cite']} (bench {e['bench']} >= {tb})", "as_of": asof, **extra}

status_counts = Counter(); nonzero = 0
with open(OUT, "w") as f:
    for r in docs:
        did = r.get("doc_id"); gl = derive(did); cb = cited_by.get(did, {})
        nonzero += 1 if cb else 0
        rec = {"doc_id": did, "neutral_citation": r.get("neutral_citation"), "case_name": r.get("case_name"), **gl,
               "cited_by": len(cb), "treatment_breakdown": dict(Counter(cb.values()))}
        f.write(json.dumps(rec, ensure_ascii=False) + "\n")
        status_counts[gl["good_law_status"]] += 1

with open(EDGES_OUT, "w") as f:
    for (frm, tgt), e in edges.items():
        f.write(json.dumps({"from": frm, "target": tgt, **e}, ensure_ascii=False) + "\n")

print("status distribution:", dict(status_counts), flush=True)
print(f"cited_by nonzero: {nonzero}/{len(docs)} ({100*nonzero//len(docs)}%)", flush=True)
print(f"wrote {OUT} and {EDGES_OUT} ({len(edges)} edges)", flush=True)