Spaces:
Runtime error
Runtime error
File size: 2,901 Bytes
993fce6 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 | import os, pickle
import numpy as np
import pandas as pd
from scipy.sparse import save_npz
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.preprocessing import LabelEncoder
from sentence_transformers import SentenceTransformer
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
PROJECT_ROOT = os.path.dirname(os.path.dirname(SCRIPT_DIR))
SEG = os.path.join(PROJECT_ROOT, "data", "segmented")
EMB = os.path.join(PROJECT_ROOT, "data", "embeddings")
os.makedirs(EMB, exist_ok=True)
df = pd.read_csv(os.path.join(SEG, "clauses_features.csv"))
print(f"Loaded {len(df)} clauses")
# ββ TF-IDF on clean_text βββββββββββββββββββββββββββββββββββββ
print("\n[1/3] TF-IDF vectorization...")
tfidf = TfidfVectorizer(max_features=3000, ngram_range=(1, 2),
sublinear_tf=True, min_df=2)
X_tfidf = tfidf.fit_transform(df["clean_text"].fillna(""))
save_npz(os.path.join(EMB, "tfidf_matrix.npz"), X_tfidf)
with open(os.path.join(EMB, "tfidf_vectorizer.pkl"), "wb") as f:
pickle.dump(tfidf, f)
print(f" β TF-IDF shape : {X_tfidf.shape}")
# ββ SBERT on raw_text ββββββββββββββββββββββββββββββββββββββββ
print("\n[2/3] SBERT embeddings (all-MiniLM-L6-v2)...")
model = SentenceTransformer("all-MiniLM-L6-v2")
embeddings = model.encode(
df["raw_text"].tolist(),
batch_size=64,
show_progress_bar=True,
convert_to_numpy=True
)
np.save(os.path.join(EMB, "sbert_embeddings.npy"), embeddings)
print(f" β SBERT shape : {embeddings.shape}")
# ββ Label encoding βββββββββββββββββββββββββββββββββββββββββββ
print("\n[3/3] Encoding labels...")
le = LabelEncoder()
y = le.fit_transform(df["risk_label"])
np.save(os.path.join(EMB, "labels_encoded.npy"), y)
with open(os.path.join(EMB, "label_encoder.pkl"), "wb") as f:
pickle.dump(le, f)
print(f" β Classes : {list(le.classes_)}")
print(f" β Labels : {dict(zip(le.classes_, le.transform(le.classes_)))}")
# ββ Save clause IDs for alignment ββββββββββββββββββββββββββββ
df[["clause_id","risk_label","domain"]].to_csv(
os.path.join(EMB, "embedding_index.csv"), index=False)
print(f"\nβ All saved β {EMB}/")
print(" tfidf_matrix.npz β sparse TF-IDF (3000 features)")
print(" tfidf_vectorizer.pkl β fitted vectorizer for inference")
print(" sbert_embeddings.npy β dense SBERT (384 dims)")
print(" labels_encoded.npy β integer labels")
print(" label_encoder.pkl β LabelEncoder for inverse transform")
print(" embedding_index.csv β clause_id alignment file") |