Spaces:
Sleeping
Sleeping
File size: 6,561 Bytes
685cc60 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 | import os
import json
import hashlib
import pandas as pd
def process_xrefs(xref_str, current_act):
"""
Process cross_references JSON string from Phase 1.
Splits into internal_refs (e.g. 'S64') and cross_act_refs (e.g. {'act': 'BNSS', 'section': '173'}).
"""
internal = []
cross = []
if not xref_str:
return internal, cross
try:
refs = json.loads(xref_str)
except Exception:
refs = []
for ref in refs:
if ref.startswith(f"{current_act}_"):
# Internal reference, strip act prefix (e.g. 'BNS_S85' -> 'S85')
stripped = ref[len(current_act)+1:]
internal.append(stripped)
else:
# Cross-act reference (e.g. 'BNSS_S173' -> {'act': 'BNSS', 'section': '173'})
parts = ref.split("_", 1)
if len(parts) == 2:
act, sec = parts
if sec.startswith("S"):
sec = sec[1:]
cross.append({"act": act, "section": sec})
return internal, cross
def build_unsummarized_trees(output_dir="output"):
"""
Loads Parquet files and constructs the unsummarized tree structures for BNS, BNSS, and BSA.
Returns a dict mapping act_code -> root_node_dict.
"""
toc_path = os.path.join(output_dir, "toc_df.parquet")
line_path = os.path.join(output_dir, "line_df.parquet")
schedule_path = os.path.join(output_dir, "schedule_df.parquet")
if not os.path.exists(toc_path) or not os.path.exists(line_path):
raise FileNotFoundError("Required Parquet files are missing in output directory.")
toc_df = pd.read_parquet(toc_path)
line_df = pd.read_parquet(line_path)
schedule_df = pd.read_parquet(schedule_path) if os.path.exists(schedule_path) else pd.DataFrame()
# Concatenate text per section_id in line_df
print("Concatenating section line contents...")
section_texts = line_df.groupby("section_id")["text"].apply(lambda lines: " ".join(lines)).to_dict()
acts = ["BNS", "BNSS", "BSA"]
if "SOP" in toc_df["act_code"].values:
acts.append("SOP")
trees = {}
for act in acts:
print(f"Building skeleton tree for {act}...")
act_toc = toc_df[toc_df["act_code"] == act].copy()
# Instantiate all nodes as dicts
node_map = {}
root_node = None
for _, row in act_toc.iterrows():
node_id = row["section_id"]
level = int(row["level"])
node_type = row["node_type"]
title = row["title"]
# Content population
content = None
if node_type in ["section", "sop_procedure", "sop_form", "sop_reference", "sop_table"]:
content = section_texts.get(node_id, "")
# Cross references
internal_refs, cross_act_refs = process_xrefs(row["cross_references"], act)
token_est = len(content) // 4 if content else 0
node = {
"node_id": node_id,
"level": level,
"node_type": node_type,
"title": title,
"summary": None,
"content": content,
"children": [],
"metadata": {
"act_code": act,
"page_range": [int(row["start_page"]), int(row["end_page"])],
"internal_refs": internal_refs,
"cross_act_refs": cross_act_refs,
"token_estimate": token_est,
"stable_hash": row["stable_hash"]
}
}
node_map[node_id] = node
if level == 0 and node_type == "root":
root_node = node
# Build parent-child relationships
for _, row in act_toc.iterrows():
node_id = row["section_id"]
parent_id = row["parent_id"]
if pd.isna(parent_id) or parent_id is None:
continue
if parent_id in node_map and node_id in node_map:
node_map[parent_id]["children"].append(node_map[node_id])
else:
print(f"Warning: parent_id {parent_id} or node_id {node_id} not found in node_map.")
# Special schedule handling for BNSS
if act == "BNSS" and not schedule_df.empty:
schedule_node_id = "BNSS_SCH1"
if schedule_node_id in node_map:
print("Appending First Schedule table rows as leaf nodes...")
for idx, row in schedule_df.iterrows():
row_content = (
f"Section: {row['section']}\n"
f"Offence: {row['offence']}\n"
f"Punishment: {row['punishment']}\n"
f"Cognizable: {row['cognizable']}\n"
f"Bailable: {row['bailable']}\n"
f"Court: {row['court']}"
)
row_id = f"BNSS_SCH1_R{idx}"
# Try to parse section number for internal reference
section_num = str(row['section']).strip()
internal_refs = []
if section_num:
internal_refs.append(f"S{section_num}")
row_node = {
"node_id": row_id,
"level": 2,
"node_type": "schedule_row",
"title": f"Schedule Row: Offence under Section {row['section']}",
"summary": None,
"content": row_content,
"children": [],
"metadata": {
"act_code": "BNSS",
"page_range": [int(row["page_no"]), int(row["page_no"])],
"internal_refs": internal_refs,
"cross_act_refs": [],
"token_estimate": len(row_content) // 4,
"stable_hash": hashlib.sha1(row_id.encode()).hexdigest()
}
}
node_map[schedule_node_id]["children"].append(row_node)
if root_node:
trees[act] = root_node
else:
raise ValueError(f"Root node for {act} was not found in TOC.")
return trees
|