Spaces:
Runtime error
Runtime error
File size: 2,680 Bytes
993fce6 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 | import os, re
import pandas as pd
from tqdm import tqdm
import spacy
nlp = spacy.load("en_core_web_sm")
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PROJECT_ROOT = os.path.dirname(os.path.dirname(SCRIPT_DIR))
SEG = os.path.join(PROJECT_ROOT, "data", "segmented")
df = pd.read_csv(os.path.join(SEG, "clauses_balanced.csv"))
print(f"Loaded {len(df)} clauses")
# ββ Keyword lists (reused from Step 6) ββββββββββββββββββββββ
MODALS = {
"must":0.9,"shall":0.9,"will":0.8,"required":0.8,
"should":0.6,"may":0.3,"can":0.3,"encouraged":0.1,"optional":0.1
}
CONSEQUENCES = {
"termination":1.0,"dismissal":1.0,"criminal":0.95,"imprisonment":1.0,
"prosecution":1.0,"damages":0.85,"penalty":0.85,"fine":0.85,
"suspension":0.75,"prohibited":0.7,"disciplinary":0.7,
"breach":0.8,"liable":0.85,"liability":0.85,"sanction":0.85,
"warning":0.5,"deduction":0.45,"comply":0.4,"obligation":0.45
}
CONDITIONALS = {
"failure to":1.0,"violation of":1.0,"breach of":1.0,
"result in":0.85,"constitute":0.8,"shall not":0.85,
"must not":0.85,"in the event of":0.85,"without consent":0.85,
"may result":0.6,"provided that":0.6,"unless":0.5
}
NEGATIONS = ["must not","shall not","not permitted","not allowed",
"without consent","without authorization","strictly prohibited"]
def extract_features(text):
t = text.lower()
tok = [x.text.lower() for x in nlp(t)]
modal = max((MODALS[w] for w in tok if w in MODALS), default=0.0)
cons = 0.0
for p,w in sorted(CONSEQUENCES.items(), key=lambda x:-len(x[0])):
if p in t: cons = max(cons, w); break
cond = 0.0
for p,w in sorted(CONDITIONALS.items(), key=lambda x:-len(x[0])):
if p in t: cond = max(cond, w); break
neg = int(any(n in t for n in NEGATIONS))
wc = len(text.split())
# Obligation strength β counts must/shall/will tokens
oblig = sum(1 for w in tok if w in {"must","shall","will","required"})
# Penalty flag β binary: does clause contain any penalty keyword
pflag = int(cons >= 0.7)
return modal, cons, cond, neg, wc, oblig, pflag
print("Extracting features...")
rows = [extract_features(str(t)) for t in tqdm(df["raw_text"])]
cols = ["modal_score","consequence_score","conditional_score",
"has_negation","word_count_f","obligation_count","penalty_flag"]
df[cols] = pd.DataFrame(rows, index=df.index)
out = os.path.join(SEG, "clauses_features.csv")
df.to_csv(out, index=False)
print(f"β Saved β {out}")
print(df[cols].describe().round(3).to_string()) |