ai-text-meta-classifier / src /meta /train_eval.py
BenoDugo's picture
Upload 35 files
ce3aedb verified
Raw
History Blame Contribute Delete
13.8 kB
"""Train and evaluate meta-classifiers over base detector scores.
Loads the score CSV produced by scripts/run_inference.py, performs a
stratified 60/20/20 split on (domain, generator, label), trains three
meta-classifier variants (logistic regression, XGBoost, small MLP), and
reports the evaluation metrics from Section 3.6 of the dissertation:
accuracy, macro-F1, AUROC, FPR at TPR=0.95 and TPR=0.99, and Expected
Calibration Error.
Outputs:
outputs/metrics_summary.csv -- one row per system, all metrics
outputs/per_domain.csv -- breakdown by domain
outputs/per_generator.csv -- breakdown by generator
outputs/predictions_test.csv -- raw test-set predictions for plotting
outputs/models/*.pkl -- saved trained models
"""
from __future__ import annotations
import argparse
import pickle
from pathlib import Path
import numpy as np
import pandas as pd
import xgboost as xgb
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import (
accuracy_score,
f1_score,
roc_auc_score,
roc_curve,
)
from sklearn.model_selection import train_test_split
from sklearn.neural_network import MLPClassifier
from sklearn.preprocessing import StandardScaler
# ---------------------------------------------------------------------------
# Metrics
# ---------------------------------------------------------------------------
def fpr_at_tpr(y_true: np.ndarray, y_score: np.ndarray, target_tpr: float) -> float:
"""Return the false positive rate at a fixed true positive rate."""
fpr, tpr, _ = roc_curve(y_true, y_score)
# Find smallest threshold that achieves at least target_tpr.
idx = np.searchsorted(tpr, target_tpr, side="left")
if idx >= len(fpr):
idx = len(fpr) - 1
return float(fpr[idx])
def expected_calibration_error(y_true: np.ndarray, y_prob: np.ndarray, n_bins: int = 15) -> float:
"""Compute ECE with equal-width probability bins."""
bins = np.linspace(0.0, 1.0, n_bins + 1)
bin_ids = np.digitize(y_prob, bins) - 1
bin_ids = np.clip(bin_ids, 0, n_bins - 1)
ece = 0.0
n = len(y_true)
for b in range(n_bins):
mask = bin_ids == b
if not mask.any():
continue
bin_acc = y_true[mask].mean()
bin_conf = y_prob[mask].mean()
ece += (mask.sum() / n) * abs(bin_acc - bin_conf)
return float(ece)
def all_metrics(y_true: np.ndarray, y_prob: np.ndarray, threshold: float = 0.5) -> dict[str, float]:
"""Compute the full set of metrics from Section 3.6."""
y_pred = (y_prob >= threshold).astype(int)
return {
"accuracy": float(accuracy_score(y_true, y_pred)),
"f1_macro": float(f1_score(y_true, y_pred, average="macro")),
"auroc": float(roc_auc_score(y_true, y_prob)),
"fpr_at_tpr_95": fpr_at_tpr(y_true, y_prob, 0.95),
"fpr_at_tpr_99": fpr_at_tpr(y_true, y_prob, 0.99),
"ece": expected_calibration_error(y_true, y_prob),
}
# ---------------------------------------------------------------------------
# Data loading and splitting
# ---------------------------------------------------------------------------
def load_scores(path: str) -> pd.DataFrame:
df = pd.read_csv(path)
# Drop rows where any detector failed (None -> NaN).
before = len(df)
df = df.dropna(subset=["binoculars", "fast_detect_gpt", "roberta"]).reset_index(drop=True)
after = len(df)
if before != after:
print(f"Dropped {before - after} rows with NaN scores. {after} remain.")
return df
def stratified_three_way_split(
df: pd.DataFrame,
test_frac: float = 0.20,
val_frac: float = 0.20,
seed: int = 42,
) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:
"""60/20/20 split stratified on (domain, generator, label).
Sparsely populated joint strata (<3 samples) get merged into the dominant
stratum for that (domain, label) pair so each stratum can appear in all
splits.
"""
df = df.copy()
df["stratum"] = (
df["domain"].astype(str) + "_"
+ df["generator"].astype(str) + "_"
+ df["label"].astype(str)
)
counts = df["stratum"].value_counts()
sparse = counts[counts < 3].index
df.loc[df["stratum"].isin(sparse), "stratum"] = (
df.loc[df["stratum"].isin(sparse), "domain"].astype(str)
+ "_merged_"
+ df.loc[df["stratum"].isin(sparse), "label"].astype(str)
)
train_val, test = train_test_split(
df, test_size=test_frac, stratify=df["stratum"], random_state=seed
)
val_relative = val_frac / (1.0 - test_frac)
train, val = train_test_split(
train_val, test_size=val_relative, stratify=train_val["stratum"], random_state=seed
)
return (
train.reset_index(drop=True),
val.reset_index(drop=True),
test.reset_index(drop=True),
)
# ---------------------------------------------------------------------------
# Models
# ---------------------------------------------------------------------------
FEATURES = ["binoculars", "fast_detect_gpt", "roberta"]
def train_logistic(X_train, y_train) -> LogisticRegression:
model = LogisticRegression(max_iter=1000, random_state=42)
model.fit(X_train, y_train)
return model
def train_xgboost(X_train, y_train, X_val, y_val) -> xgb.XGBClassifier:
model = xgb.XGBClassifier(
n_estimators=300,
max_depth=4,
learning_rate=0.1,
objective="binary:logistic",
eval_metric="logloss",
random_state=42,
early_stopping_rounds=20,
)
model.fit(X_train, y_train, eval_set=[(X_val, y_val)], verbose=False)
return model
def train_mlp(X_train, y_train) -> MLPClassifier:
model = MLPClassifier(
hidden_layer_sizes=(32, 32),
activation="relu",
solver="adam",
max_iter=500,
early_stopping=True,
validation_fraction=0.15,
random_state=42,
)
model.fit(X_train, y_train)
return model
# ---------------------------------------------------------------------------
# Evaluation orchestration
# ---------------------------------------------------------------------------
def evaluate_system(name: str, y_true: np.ndarray, y_prob: np.ndarray) -> dict:
metrics = all_metrics(y_true, y_prob)
metrics["system"] = name
return metrics
def per_group_metrics(
df_test: pd.DataFrame, y_prob: np.ndarray, group_col: str, system_name: str
) -> pd.DataFrame:
rows = []
for value, group in df_test.groupby(group_col):
y_true = group["label"].to_numpy()
scores = y_prob[group.index]
if len(np.unique(y_true)) < 2:
# Need both classes present for AUROC/FPR.
row = {"system": system_name, group_col: value, "n": len(group)}
row.update({k: float("nan") for k in [
"accuracy", "f1_macro", "auroc", "fpr_at_tpr_95", "fpr_at_tpr_99", "ece"
]})
else:
row = {"system": system_name, group_col: value, "n": len(group)}
row.update(all_metrics(y_true, scores))
rows.append(row)
return pd.DataFrame(rows)
# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--scores", type=str, default="data/raid_scores.csv")
parser.add_argument("--outdir", type=str, default="outputs")
parser.add_argument("--seed", type=int, default=42)
args = parser.parse_args()
outdir = Path(args.outdir)
(outdir / "models").mkdir(parents=True, exist_ok=True)
print(f"Loading {args.scores}...")
df = load_scores(args.scores)
print(f"Loaded {len(df)} samples.")
print("\nSplitting 60/20/20 stratified on (domain, generator, label)...")
train_df, val_df, test_df = stratified_three_way_split(df, seed=args.seed)
print(f" train: {len(train_df)} | val: {len(val_df)} | test: {len(test_df)}")
X_train = train_df[FEATURES].to_numpy()
y_train = train_df["label"].to_numpy()
X_val = val_df[FEATURES].to_numpy()
y_val = val_df["label"].to_numpy()
X_test = test_df[FEATURES].to_numpy()
y_test = test_df["label"].to_numpy()
# Per-detector standardisation using training distribution only.
scaler = StandardScaler().fit(X_train)
X_train_s = scaler.transform(X_train)
X_val_s = scaler.transform(X_val)
X_test_s = scaler.transform(X_test)
# Save the scaler for reproducibility.
with (outdir / "models" / "scaler.pkl").open("wb") as f:
pickle.dump(scaler, f)
summary_rows = []
per_domain_rows = []
per_generator_rows = []
test_predictions = test_df[["id", "domain", "generator", "label"]].copy()
# ----- Base detectors as standalone systems -----
# For each base detector we treat its raw score as the prediction. We
# min-max normalise to [0, 1] so it can be compared against the meta
# classifier probabilities. Polarity differs across detectors:
# - Binoculars: high = human (label 0). Invert.
# - Fast-DetectGPT: high = AI. Keep as-is.
# - RoBERTa: P(AI). Keep as-is.
base_score_funcs = {
"binoculars_solo": lambda x: 1.0 - _minmax(x[:, 0], train_df["binoculars"]),
"fast_detect_gpt_solo": lambda x: _minmax(x[:, 1], train_df["fast_detect_gpt"]),
"roberta_solo": lambda x: x[:, 2], # already 0-1
}
for name, fn in base_score_funcs.items():
probs = fn(X_test)
summary_rows.append(evaluate_system(name, y_test, probs))
test_predictions[name] = probs
per_domain_rows.append(per_group_metrics(test_df, probs, "domain", name))
per_generator_rows.append(per_group_metrics(test_df, probs, "generator", name))
print(f" {name}: AUROC={summary_rows[-1]['auroc']:.4f} "
f"FPR@TPR95={summary_rows[-1]['fpr_at_tpr_95']:.4f}")
# ----- Meta classifiers -----
print("\nTraining logistic regression...")
lr = train_logistic(X_train_s, y_train)
lr_prob = lr.predict_proba(X_test_s)[:, 1]
summary_rows.append(evaluate_system("meta_logistic", y_test, lr_prob))
test_predictions["meta_logistic"] = lr_prob
per_domain_rows.append(per_group_metrics(test_df, lr_prob, "domain", "meta_logistic"))
per_generator_rows.append(per_group_metrics(test_df, lr_prob, "generator", "meta_logistic"))
with (outdir / "models" / "logistic.pkl").open("wb") as f:
pickle.dump(lr, f)
print(f" meta_logistic: AUROC={summary_rows[-1]['auroc']:.4f} "
f"FPR@TPR95={summary_rows[-1]['fpr_at_tpr_95']:.4f}")
print(f" coefficients: {dict(zip(FEATURES, lr.coef_[0].tolist()))}")
print("\nTraining XGBoost (with early stopping on val)...")
xgb_model = train_xgboost(X_train_s, y_train, X_val_s, y_val)
xgb_prob = xgb_model.predict_proba(X_test_s)[:, 1]
summary_rows.append(evaluate_system("meta_xgboost", y_test, xgb_prob))
test_predictions["meta_xgboost"] = xgb_prob
per_domain_rows.append(per_group_metrics(test_df, xgb_prob, "domain", "meta_xgboost"))
per_generator_rows.append(per_group_metrics(test_df, xgb_prob, "generator", "meta_xgboost"))
with (outdir / "models" / "xgboost.pkl").open("wb") as f:
pickle.dump(xgb_model, f)
print(f" meta_xgboost: AUROC={summary_rows[-1]['auroc']:.4f} "
f"FPR@TPR95={summary_rows[-1]['fpr_at_tpr_95']:.4f}")
print("\nTraining MLP...")
mlp = train_mlp(X_train_s, y_train)
mlp_prob = mlp.predict_proba(X_test_s)[:, 1]
summary_rows.append(evaluate_system("meta_mlp", y_test, mlp_prob))
test_predictions["meta_mlp"] = mlp_prob
per_domain_rows.append(per_group_metrics(test_df, mlp_prob, "domain", "meta_mlp"))
per_generator_rows.append(per_group_metrics(test_df, mlp_prob, "generator", "meta_mlp"))
with (outdir / "models" / "mlp.pkl").open("wb") as f:
pickle.dump(mlp, f)
print(f" meta_mlp: AUROC={summary_rows[-1]['auroc']:.4f} "
f"FPR@TPR95={summary_rows[-1]['fpr_at_tpr_95']:.4f}")
# ----- Save outputs -----
pd.DataFrame(summary_rows).to_csv(outdir / "metrics_summary.csv", index=False)
pd.concat(per_domain_rows).to_csv(outdir / "per_domain.csv", index=False)
pd.concat(per_generator_rows).to_csv(outdir / "per_generator.csv", index=False)
test_predictions.to_csv(outdir / "predictions_test.csv", index=False)
print(f"\nSaved metrics summary to {outdir / 'metrics_summary.csv'}")
print(f"Saved per-domain to {outdir / 'per_domain.csv'}")
print(f"Saved per-generator to {outdir / 'per_generator.csv'}")
print(f"Saved test predictions to {outdir / 'predictions_test.csv'}")
print(f"Saved models to {outdir / 'models/'}")
print("\n=== SUMMARY (test set) ===")
print(pd.DataFrame(summary_rows)[["system", "auroc", "fpr_at_tpr_95",
"fpr_at_tpr_99", "f1_macro", "ece"]]
.to_string(index=False, float_format=lambda x: f"{x:.4f}"))
def _minmax(values: np.ndarray, train_values: pd.Series) -> np.ndarray:
"""Min-max normalise values to [0, 1] using the training distribution."""
lo = float(train_values.min())
hi = float(train_values.max())
if hi - lo < 1e-12:
return np.zeros_like(values, dtype=float)
out = (values - lo) / (hi - lo)
return np.clip(out, 0.0, 1.0)
if __name__ == "__main__":
main()