File size: 2,901 Bytes
993fce6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
import os, pickle
import numpy as np
import pandas as pd
from scipy.sparse import save_npz
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.preprocessing import LabelEncoder
from sentence_transformers import SentenceTransformer

SCRIPT_DIR   = os.path.dirname(os.path.abspath(__file__))
PROJECT_ROOT = os.path.dirname(os.path.dirname(SCRIPT_DIR))
SEG          = os.path.join(PROJECT_ROOT, "data", "segmented")
EMB          = os.path.join(PROJECT_ROOT, "data", "embeddings")
os.makedirs(EMB, exist_ok=True)

df = pd.read_csv(os.path.join(SEG, "clauses_features.csv"))
print(f"Loaded {len(df)} clauses")

# ── TF-IDF on clean_text ─────────────────────────────────────
print("\n[1/3] TF-IDF vectorization...")
tfidf = TfidfVectorizer(max_features=3000, ngram_range=(1, 2),
                        sublinear_tf=True, min_df=2)
X_tfidf = tfidf.fit_transform(df["clean_text"].fillna(""))

save_npz(os.path.join(EMB, "tfidf_matrix.npz"), X_tfidf)
with open(os.path.join(EMB, "tfidf_vectorizer.pkl"), "wb") as f:
    pickle.dump(tfidf, f)
print(f"  βœ“ TF-IDF shape : {X_tfidf.shape}")

# ── SBERT on raw_text ────────────────────────────────────────
print("\n[2/3] SBERT embeddings (all-MiniLM-L6-v2)...")
model = SentenceTransformer("all-MiniLM-L6-v2")
embeddings = model.encode(
    df["raw_text"].tolist(),
    batch_size=64,
    show_progress_bar=True,
    convert_to_numpy=True
)
np.save(os.path.join(EMB, "sbert_embeddings.npy"), embeddings)
print(f"  βœ“ SBERT shape  : {embeddings.shape}")

# ── Label encoding ───────────────────────────────────────────
print("\n[3/3] Encoding labels...")
le = LabelEncoder()
y  = le.fit_transform(df["risk_label"])
np.save(os.path.join(EMB, "labels_encoded.npy"), y)
with open(os.path.join(EMB, "label_encoder.pkl"), "wb") as f:
    pickle.dump(le, f)
print(f"  βœ“ Classes : {list(le.classes_)}")
print(f"  βœ“ Labels  : {dict(zip(le.classes_, le.transform(le.classes_)))}")

# ── Save clause IDs for alignment ────────────────────────────
df[["clause_id","risk_label","domain"]].to_csv(
    os.path.join(EMB, "embedding_index.csv"), index=False)

print(f"\nβœ“ All saved β†’ {EMB}/")
print("  tfidf_matrix.npz      β€” sparse TF-IDF (3000 features)")
print("  tfidf_vectorizer.pkl  β€” fitted vectorizer for inference")
print("  sbert_embeddings.npy  β€” dense SBERT (384 dims)")
print("  labels_encoded.npy    β€” integer labels")
print("  label_encoder.pkl     β€” LabelEncoder for inverse transform")
print("  embedding_index.csv   β€” clause_id alignment file")