IDS-Multi-Agents-Graph-RAG / utils /rules_extractor.py
tarekfer8's picture
Upload folder using huggingface_hub
bf8df4f verified
Raw
History Blame Contribute Delete
9.77 kB
"""
Hierarchical rule extraction from XAI attributions.
Two-phase clustering:
Phase 1: cluster instances by which features are in their top-K (Jaccard).
Each Phase-1 cluster represents a "reasoning mode."
Phase 2: within each Phase-1 cluster, re-cluster members in feature-value
space, using the cluster's defining features only. Each Phase-2
sub-cluster represents a "value variant" of that reasoning mode.
The class is XAI-agnostic: it consumes a DataFrame of pre-computed attributions
(index = instance_idx, columns = 'class', 'z0', then one column per feature).
"""
import numpy as np
import pandas as pd
from scipy.spatial.distance import pdist, squareform
from scipy.cluster.hierarchy import linkage, fcluster
from sklearn.cluster import KMeans
META_COLS = {"class", "z0"}
class HierarchicalRuleExtractor:
def __init__(
self,
attributions_df, # index = instance_idx; cols: class, z0, + features
reference_data, # numpy (n_rows, n_columns)
feature_specs, # list of dicts (name, type, columns, ...)
K_top=10,
higher_is_better=True,
defining_threshold=0.7,
):
self.attributions = attributions_df.copy()
self.reference_data = np.asarray(reference_data, dtype=np.float32)
self.feature_specs = feature_specs
self._spec_by_name = {spec["name"]: spec for spec in feature_specs}
self.K_top = K_top
self.higher_is_better = higher_is_better
self.defining_threshold = defining_threshold
# Feature columns in the attributions
self.feature_names = [
c for c in self.attributions.columns if c not in META_COLS
]
# Results filled in by fit()
self.phase1_labels = None
self.phase2_labels = None
self.defining_features_per_cluster = None
self.cluster_descriptions = None
# ------------------------------------------------------------
# Phase 1: cluster by top-K feature sets (Jaccard)
# ------------------------------------------------------------
def _build_top_k_matrix(self):
"""Return binary matrix (n_instances, n_features): 1 if feature is top-K."""
n = len(self.attributions)
f = len(self.feature_names)
mat = np.zeros((n, f), dtype=np.int8)
name_to_idx = {name: i for i, name in enumerate(self.feature_names)}
for row_pos, (_, row) in enumerate(self.attributions.iterrows()):
scores = row[self.feature_names].dropna()
ranked = scores.sort_values(ascending=not self.higher_is_better)
top = ranked.index[: self.K_top]
for name in top:
mat[row_pos, name_to_idx[name]] = 1
return mat
def _cluster_phase1(self, top_k_matrix, n_clusters):
"""Hierarchical clustering with Jaccard distance, cut at n_clusters."""
# pdist returns condensed distance vector
dists = pdist(top_k_matrix, metric="jaccard")
Z = linkage(dists, method="average")
labels = fcluster(Z, t=n_clusters, criterion="maxclust")
return labels # 1-indexed cluster labels
# ------------------------------------------------------------
# Defining features per Phase-1 cluster
# ------------------------------------------------------------
def _compute_defining_features(self, top_k_matrix, phase1_labels):
"""For each cluster, list features in top-K for >= threshold of members."""
result = {}
for cid in np.unique(phase1_labels):
members_mask = phase1_labels == cid
cluster_top_k = top_k_matrix[members_mask]
freqs = cluster_top_k.mean(axis=0)
defining_idx = np.where(freqs >= self.defining_threshold)[0]
result[int(cid)] = [self.feature_names[i] for i in defining_idx]
return result
# ------------------------------------------------------------
# Phase 2: cluster on feature values within each Phase-1 cluster
# ------------------------------------------------------------
def _cluster_phase2(self, phase1_labels, defining_features, n_subclusters):
"""
For each Phase-1 cluster, k-means on its members' feature values
(using the cluster's defining features only).
Returns array of sub-cluster labels (one per instance).
"""
n = len(self.attributions)
sub_labels = np.zeros(n, dtype=int)
instance_indices = self.attributions.index.values
for cid in np.unique(phase1_labels):
members_mask = phase1_labels == cid
members_idx_pos = np.where(members_mask)[0]
n_members = len(members_idx_pos)
features = defining_features[int(cid)]
if len(features) == 0 or n_members < 2:
# No defining features or too few members: single sub-cluster
sub_labels[members_idx_pos] = 0
continue
# Get column indices in reference_data for defining features
cols = []
for fname in features:
cols.extend(self._spec_by_name[fname]["columns"])
cols = np.array(cols, dtype=int)
# Pull feature values
row_indices = instance_indices[members_idx_pos]
X_sub = self.reference_data[row_indices][:, cols]
# Number of sub-clusters: user-specified, but capped at members
k = min(n_subclusters, n_members)
if k < 2:
sub_labels[members_idx_pos] = 0
continue
km = KMeans(n_clusters=k, random_state=0, n_init=10)
labels = km.fit_predict(X_sub)
sub_labels[members_idx_pos] = labels
return sub_labels
# ------------------------------------------------------------
# Describe each (phase1, phase2) cluster
# ------------------------------------------------------------
def _describe(self, phase1_labels, phase2_labels, defining_features):
rows = []
instance_indices = self.attributions.index.values
for cid in np.unique(phase1_labels):
members_mask = phase1_labels == cid
features = defining_features[int(cid)]
for sub in np.unique(phase2_labels[members_mask]):
sub_mask = members_mask & (phase2_labels == sub)
row_indices = instance_indices[sub_mask]
sub_classes = self.attributions.loc[row_indices, "class"].values
# Feature value summary for defining features
feature_ranges = {}
for fname in features:
spec = self._spec_by_name[fname]
cols = spec["columns"]
vals = self.reference_data[row_indices][:, cols]
if spec["type"] == "numerical":
v = vals[:, 0]
feature_ranges[fname] = {
"type": "numerical",
"q10": float(np.quantile(v, 0.10)),
"median": float(np.median(v)),
"q90": float(np.quantile(v, 0.90)),
}
else:
# Categorical/ordinal group: report most common active category
active_idx = vals.argmax(axis=1)
most_common = int(np.bincount(active_idx).argmax())
modal_col = cols[most_common]
feature_ranges[fname] = {
"type": spec["type"],
"modal_column": modal_col,
"modal_fraction": float((active_idx == most_common).mean()),
}
rows.append({
"phase1_id": int(cid),
"phase2_id": int(sub),
"n_instances": int(sub_mask.sum()),
"n_TP": int((sub_classes == "TP").sum()),
"n_FP": int((sub_classes == "FP").sum()),
"n_FN": int((sub_classes == "FN").sum()),
"n_TN": int((sub_classes == "TN").sum()),
"defining_features": features,
"feature_ranges": feature_ranges,
})
return pd.DataFrame(rows)
# ------------------------------------------------------------
# Main entry
# ------------------------------------------------------------
def fit(self, n_clusters_phase1=5, n_subclusters_phase2=3):
"""
Run the two-phase clustering.
Parameters
----------
n_clusters_phase1 : int, number of reasoning modes to extract.
n_subclusters_phase2 : int, max number of value variants per mode.
"""
# Phase 1
top_k_matrix = self._build_top_k_matrix()
phase1_labels = self._cluster_phase1(top_k_matrix, n_clusters_phase1)
defining_features = self._compute_defining_features(top_k_matrix, phase1_labels)
# Phase 2
phase2_labels = self._cluster_phase2(phase1_labels, defining_features, n_subclusters_phase2)
# Store
self.phase1_labels = phase1_labels
self.phase2_labels = phase2_labels
self.defining_features_per_cluster = defining_features
self.cluster_descriptions = self._describe(
phase1_labels, phase2_labels, defining_features
)
return self.cluster_descriptions