"""Ingest CanLex's curated commentary datasets into searchable chunks. Commentary is the one doc_type CanLex AUTHORS rather than mirrors: structured legal-analysis datasets (currently the US-dispositions helper -- whether each kind of US state criminal disposition is a "conviction" for IRPA s. 36 purposes). Every chunk is banner-labelled as commentary, every proposition carries its authorities, and entries with no authority say so explicitly and flag their reasoning as interpretation. The source of truth is data/curated/us_dispositions.json, which is reviewed by the user before it ships; this module just renders it into corpus chunks. py -m canlex.commentary [--allow-shrink] """ import json import sys from ._common import write_corpus from .config import DATA_DIR, PROCESSED_DIR CURATED = DATA_DIR / "curated" / "us_dispositions.json" EQUIV = DATA_DIR / "curated" / "us_equivalency.json" OUT = PROCESSED_DIR / "commentary.json" ACT_CODE = "US-DISP" ACT_SHORT = "US Dispositions Helper" ACT_NAME = ("US criminal dispositions and the IRPA 'conviction' concept " "(curated CanLex commentary)") BANNER = ("CURATED ANALYSIS -- commentary compiled for CanLex, not a source " "of law. Verify against the cited authorities before relying on it.") _STATUS_LABEL = { "settled": "Settled by binding authority", "judicially-considered": "Judicially considered (persuasive authority)", "guidance-only": "IRCC guidance only -- no judicial authority located", "no-authority": "NO AUTHORITY LOCATED -- reasoned interpretation only", } # Verdicts assess the COMPLETED disposition; the methodology chunk explains # the rule each one derives from. 'depends' is the legacy spelling of # fact-specific, kept for backwards compatibility. _VERDICT_LABEL = { "yes": "YES", "likely-yes": "LIKELY YES", "likely-no": "LIKELY NO", "no": "NO", "fact-specific": "FACT-SPECIFIC", "depends": "FACT-SPECIFIC", } def _vlabel(v): return _VERDICT_LABEL.get(v, str(v).upper()) def _entry_text(e, max_states=None): """Render one disposition entry as a readable, retrieval-friendly block. With the 51-jurisdiction survey an entry can carry dozens of state rows; max_states caps how many render inline (the corpus keeps full per-state detail in separate per-state chunks so retrieval still reaches every row). None renders everything -- the tool uses that for direct lookups.""" lines = [BANNER, ""] lines.append(f"Disposition: {'; '.join(e['names'])}") lines.append(f"Is the COMPLETED disposition a conviction for IRPA s. 36: " f"{_vlabel(e['is_conviction'])}") lines.append(f"Authority status: {_STATUS_LABEL[e['status']]}") if e.get("bottom_line"): lines.append("") lines.append(f"BOTTOM LINE: {e['bottom_line']}") lines.append("") lines.append(e["analysis"]) if e.get("state_variations"): named = [v for v in e["state_variations"] if v["state"].lower() != "general"] shown = named if max_states is None else named[:max_states] lines.append("") if max_states is not None and len(named) > len(shown): from collections import Counter counts = Counter(v.get("is_conviction", "depends") for v in named) lines.append(f"State-by-state coverage: {len(named)} jurisdictions " f"({', '.join(f'{k}: {n}' for k, n in counts.most_common())}) " f"-- full per-state detail in the per-state entries.") else: lines.append("State variations:") for v in shown: flag = (f" (conviction: {_vlabel(v['is_conviction'])})" if v.get("is_conviction") else "") lines.append(f"- {v['state']}{flag}: {v['note']}") for v in e["state_variations"]: if v["state"].lower() == "general": lines.append(f"- General: {v['note']}") if e.get("authorities"): lines.append("") lines.append("Authorities:") for a in e["authorities"]: pin = f", {a['pin']}" if a.get("pin") else "" lines.append(f"- {a['cite']} ({a['court']}{pin}): {a['holding']}") if e.get("guidance"): lines.append("") lines.append("IRCC guidance (cited by reference; not reproduced here):") for g in e["guidance"]: lines.append(f"- {g['ref']}: {g['note']}") if e.get("interpretation"): lines.append("") lines.append("INTERPRETATION (no direct authority -- this is CanLex's " "reasoned view from the governing principles; treat it as " "a starting point, not an answer): " + e["interpretation"]) return "\n".join(lines) def build(allow_shrink=False): data = json.loads(CURATED.read_text(encoding="utf-8")) chunks = [] for e in data.get("methodology", []): chunks.append({ "id": f"commentary-method-{e['id']}", "doc_type": "commentary", "act_code": ACT_CODE, "act_short": ACT_SHORT, "act_name": ACT_NAME, "section": e["id"], "marginal_note": e["title"], "part": "Methodology", "division": "", "heading": e["title"], "text": BANNER + "\n\n" + e["text"], "history": "", "last_amended": "", "current_to": data.get("reviewed", ""), "citation": f"{ACT_SHORT} — {e['title']}", "source_url": "", }) for e in data.get("dispositions", []): chunks.append({ "id": f"commentary-disp-{e['id']}", "doc_type": "commentary", "act_code": ACT_CODE, "act_short": ACT_SHORT, "act_name": ACT_NAME, "section": e["id"], "marginal_note": e["names"][0], "part": "US dispositions", "division": "", "heading": (f"Is a US {e['names'][0]} a conviction for IRPA " f"s. 36? ({_vlabel(e['is_conviction'])})"), "text": _entry_text(e, max_states=6), "history": "", "last_amended": "", "current_to": data.get("reviewed", ""), "citation": f"{ACT_SHORT} — {e['names'][0]}", "source_url": "", }) # One chunk per jurisdiction: every disposition row for that state, so a # query naming a state ("Georgia first offender act", "Missouri SIS") # retrieves that state's page directly. by_state = {} for e in data.get("dispositions", []): for v in e.get("state_variations", []): st = v["state"] if st.lower() == "general": continue flag = _vlabel(v.get("is_conviction") or e["is_conviction"]) by_state.setdefault(st, []).append( f"- {e['names'][0]} (conviction: {flag}): {v['note']}") for st in sorted(by_state): body = (BANNER + "\n\n" + f"US dispositions — {st}: whether each disposition type is a " f"conviction for IRPA s. 36, under {st} law.\n\n" + "\n".join(by_state[st]) + "\n\nThe act branch (IRPA s. 36(1)(c)/(2)(c)) can apply even " "where a disposition is not a conviction. See the " "per-disposition entries for the governing analysis and " "authorities.") slug = st.lower().replace(" ", "-") chunks.append({ "id": f"commentary-state-{slug}", "doc_type": "commentary", "act_code": ACT_CODE, "act_short": ACT_SHORT, "act_name": ACT_NAME, "section": f"state-{slug}", "marginal_note": f"US dispositions — {st}", "part": "US dispositions by state", "division": "", "heading": (f"{st}: criminal dispositions vs the IRPA " f"'conviction' concept"), "text": body, "history": "", "last_amended": "", "current_to": data.get("reviewed", ""), "citation": f"{ACT_SHORT} — {st}", "source_url": "", }) # --- equivalency pairings (step 2), same banner discipline n_pairings = 0 if EQUIV.exists(): eq = json.loads(EQUIV.read_text(encoding="utf-8")) for m in eq.get("methodology", []): chunks.append({ "id": f"commentary-method-{m['id']}", "doc_type": "commentary", "act_code": ACT_CODE, "act_short": ACT_SHORT, "act_name": ACT_NAME, "section": m["id"], "marginal_note": m["title"], "part": "Methodology", "division": "", "heading": m["title"], "text": BANNER + "\n\n" + m["text"], "history": "", "last_amended": "", "current_to": eq.get("reviewed", ""), "citation": f"{ACT_SHORT} — {m['title']}", "source_url": "", }) for p in eq.get("pairings", []): n_pairings += 1 lines = [BANNER, "", f"US offence: {'; '.join(p['us_terms'][:5])}", f"Canadian equivalent: {p['canadian_offence']}", f"Maximum penalty (verified): {p['penalty']}", f"Inadmissibility branch: {p['branch']}", "", p["analysis"]] if p.get("caveats"): lines.append("") lines.append("Caveats:") lines += [f"- {c}" for c in p["caveats"]] if p.get("authorities"): lines.append("") lines.append("Authorities:") lines += [f"- {a['cite']} ({a['court']}): {a['holding']}" for a in p["authorities"]] chunks.append({ "id": f"commentary-equiv-{p['id']}", "doc_type": "commentary", "act_code": ACT_CODE, "act_short": ACT_SHORT, "act_name": ACT_NAME, "section": f"equiv-{p['id']}", "marginal_note": f"Equivalency: {p['us_terms'][0]}", "part": "US offence equivalency", "division": "", "heading": (f"What does a US {p['us_terms'][0]} conviction " f"equate to in Canada?"), "text": "\n".join(lines), "history": "", "last_amended": "", "current_to": eq.get("reviewed", ""), "citation": f"{ACT_SHORT} — equivalency: {p['us_terms'][0]}", "source_url": "", }) # A full re-render of the curated files, so a truncated or half-edited # us_dispositions.json (or an equivalency file that moved) collapses the # chunk count and the write would replace good analysis with the remnant. # indent=1 is the stored file's format -- changing it rewrites every line. if not write_corpus(OUT, chunks, indent=1, allow_shrink=allow_shrink): return False print(f"{len(chunks)} commentary chunks " f"({len(data.get('dispositions', []))} dispositions, " f"{len(by_state)} state pages, {n_pairings} equivalency pairings, " f"{len(data.get('methodology', []))}+ methodology) -> {OUT}") return True def main(): sys.exit(0 if build(allow_shrink="--allow-shrink" in sys.argv) else 1) if __name__ == "__main__": main()