| """ |
| evaluate.py |
| =========== |
| Evaluation suite for the LSTM-Autoencoder anomaly detector. |
| |
| Benchmarks against three baselines: |
| 1. Isolation Forest (statistical, no sequence awareness) |
| 2. Random Forest (supervised, context-blind) |
| 3. WAF simulation (pattern matching, signature-based) |
| |
| Metrics reported: |
| - Accuracy, Precision, Recall, F1-Score |
| - False Positive Rate (FPR) |
| - Inference latency (ms per session) |
| - Throughput (sessions per second) |
| |
| Author : K.A.D.S.D. Kandanaarachchi (2020/ICT/19) |
| Project: Detecting Anomalous REST API Traffic - IT4216 |
| """ |
|
|
| import argparse |
| import json |
| import logging |
| import time |
| from pathlib import Path |
|
|
| import numpy as np |
| import torch |
| from sklearn.ensemble import IsolationForest, RandomForestClassifier |
| from sklearn.metrics import ( |
| accuracy_score, |
| confusion_matrix, |
| f1_score, |
| precision_score, |
| recall_score, |
| ) |
| from torch.utils.data import DataLoader, TensorDataset |
|
|
| from model import ( |
| build_model_cicids2018, |
| build_model_csic2010, |
| build_model_unsw, |
| ) |
|
|
| logging.basicConfig( |
| level=logging.INFO, |
| format="%(asctime)s %(levelname)s %(message)s", |
| datefmt="%H:%M:%S", |
| ) |
| log = logging.getLogger(__name__) |
|
|
|
|
| def compute_metrics(y_true: np.ndarray, y_pred: np.ndarray, model_name: str) -> dict: |
| """ |
| Compute and log all evaluation metrics. |
| Returns a dict of results for saving. |
| """ |
| acc = accuracy_score(y_true, y_pred) |
| prec = precision_score(y_true, y_pred, zero_division=0) |
| rec = recall_score(y_true, y_pred, zero_division=0) |
| f1 = f1_score(y_true, y_pred, zero_division=0) |
|
|
| cm = confusion_matrix(y_true, y_pred) |
| tn, fp, fn, tp = cm.ravel() |
| fpr = fp / (fp + tn) if (fp + tn) > 0 else 0.0 |
|
|
| log.info("── %s ──────────────────────────", model_name) |
| log.info(" Accuracy : %.4f", acc) |
| log.info(" Precision : %.4f", prec) |
| log.info(" Recall : %.4f", rec) |
| log.info(" F1-Score : %.4f", f1) |
| log.info(" FPR : %.4f", fpr) |
| log.info(" TP=%d FP=%d TN=%d FN=%d", tp, fp, tn, fn) |
|
|
| return { |
| "model": model_name, |
| "accuracy": round(acc, 4), |
| "precision": round(prec, 4), |
| "recall": round(rec, 4), |
| "f1": round(f1, 4), |
| "fpr": round(fpr, 4), |
| "tp": int(tp), |
| "fp": int(fp), |
| "tn": int(tn), |
| "fn": int(fn), |
| } |
|
|
|
|
| def measure_latency(fn, data, n_runs: int = 100) -> tuple[float, float]: |
| """ |
| Measure average inference latency and throughput. |
| |
| Returns (latency_ms_per_session, sessions_per_second) |
| """ |
| |
| for _ in range(5): |
| fn(data[:32]) |
|
|
| times = [] |
| for _ in range(n_runs): |
| start = time.perf_counter() |
| fn(data[:32]) |
| times.append(time.perf_counter() - start) |
|
|
| avg_ms = np.mean(times) * 1000 / 32 |
| throughput = 32 / np.mean(times) |
| return round(avg_ms, 4), round(throughput, 1) |
|
|
|
|
| def evaluate_lstm( |
| model, |
| X_test: np.ndarray, |
| y_test: np.ndarray, |
| threshold: float, |
| dataset: str, |
| device: str = "cpu", |
| ) -> tuple[dict, float, float]: |
| """ |
| Run the LSTM-Autoencoder on the test set. |
| Flag sessions where reconstruction error > threshold. |
| """ |
| model.eval() |
| model = model.to(device) |
|
|
| if dataset == "csic2010": |
| tensor = torch.tensor(X_test, dtype=torch.long) |
| else: |
| X_test = np.nan_to_num(X_test, nan=0.0, posinf=0.0, neginf=0.0) |
| tensor = torch.tensor(X_test, dtype=torch.float32) |
|
|
| loader = DataLoader( |
| TensorDataset(tensor), |
| batch_size=512, |
| shuffle=False, |
| ) |
|
|
| all_errors = [] |
| with torch.no_grad(): |
| for (batch,) in loader: |
| batch = batch.to(device) |
| errors = model.reconstruction_error(batch) |
| all_errors.extend(errors.cpu().numpy()) |
|
|
| all_errors = np.array(all_errors) |
| y_pred = (all_errors > threshold).astype(int) |
|
|
| metrics = compute_metrics(y_test, y_pred, "LSTM-Autoencoder") |
|
|
| |
| def infer(x): |
| with torch.no_grad(): |
| if dataset == "csic2010": |
| t = torch.tensor(x, dtype=torch.long).to(device) |
| else: |
| t = torch.tensor(x, dtype=torch.float32).to(device) |
| return model.reconstruction_error(t) |
|
|
| latency, throughput = measure_latency(infer, X_test) |
| log.info(" Latency : %.4f ms/session", latency) |
| log.info(" Throughput: %.1f sessions/sec", throughput) |
|
|
| metrics["latency_ms"] = latency |
| metrics["throughput"] = throughput |
| metrics["threshold"] = threshold |
| metrics["error_mean"] = round(float(all_errors.mean()), 6) |
| metrics["error_std"] = round(float(all_errors.std()), 6) |
|
|
| return metrics, all_errors |
|
|
|
|
| def save_errors(errors: np.ndarray, y_test: np.ndarray, dataset: str, res_dir: Path) -> None: |
| """ |
| Save raw per-sample reconstruction errors and their true labels to disk. |
| |
| This is what lets visualise.py plot the REAL error distribution |
| (Figures 2 and 7) instead of simulating one from mean_error/std_error. |
| Call this right after evaluate_lstm() — errors and y_test are |
| already aligned since evaluate_lstm uses shuffle=False. |
| """ |
| res_dir = Path(res_dir) |
| res_dir.mkdir(parents=True, exist_ok=True) |
| np.save(res_dir / f"errors_{dataset}.npy", errors) |
| np.save(res_dir / f"errors_labels_{dataset}.npy", y_test) |
| log.info("Saved raw errors → %s", res_dir / f"errors_{dataset}.npy") |
|
|
|
|
| def evaluate_isolation_forest( |
| X_train: np.ndarray, |
| X_test: np.ndarray, |
| y_test: np.ndarray, |
| dataset: str, |
| ) -> dict: |
| """ |
| Isolation Forest baseline. |
| Treats each session as a flat feature vector — no sequence awareness. |
| This is the 'lightweight but blind' baseline from proposal. |
| """ |
| log.info("Training Isolation Forest...") |
|
|
| |
| if dataset == "csic2010": |
| |
| X_tr_flat = X_train.astype(np.float32).reshape(len(X_train), -1) |
| X_te_flat = X_test.astype(np.float32).reshape(len(X_test), -1) |
| else: |
| X_train = np.nan_to_num(X_train, nan=0.0, posinf=0.0, neginf=0.0) |
| X_test = np.nan_to_num(X_test, nan=0.0, posinf=0.0, neginf=0.0) |
| X_tr_flat = X_train.reshape(len(X_train), -1) |
| X_te_flat = X_test.reshape(len(X_test), -1) |
|
|
| |
| max_train = min(50_000, len(X_tr_flat)) |
| idx = np.random.choice(len(X_tr_flat), max_train, replace=False) |
|
|
| clf = IsolationForest( |
| n_estimators=100, |
| contamination=0.05, |
| random_state=42, |
| n_jobs=-1, |
| ) |
| clf.fit(X_tr_flat[idx]) |
|
|
| |
| raw_pred = clf.predict(X_te_flat) |
| y_pred = (raw_pred == -1).astype(int) |
|
|
| metrics = compute_metrics(y_test, y_pred, "Isolation Forest") |
|
|
| def infer(x): |
| xf = x.astype(np.float32).reshape(len(x), -1) |
| return clf.predict(xf) |
|
|
| latency, throughput = measure_latency(infer, X_test) |
| log.info(" Latency : %.4f ms/session", latency) |
| log.info(" Throughput: %.1f sessions/sec", throughput) |
|
|
| metrics["latency_ms"] = latency |
| metrics["throughput"] = throughput |
| return metrics |
|
|
|
|
| def evaluate_random_forest( |
| X_train: np.ndarray, |
| y_train: np.ndarray, |
| X_test: np.ndarray, |
| y_test: np.ndarray, |
| dataset: str, |
| ) -> dict: |
| """ |
| Random Forest baseline — supervised, context-blind. |
| Given labels during training (unlike our unsupervised model). |
| This represents the best-case supervised approach. |
| """ |
| log.info("Training Random Forest...") |
|
|
| if dataset == "csic2010": |
| X_tr_flat = X_train.astype(np.float32).reshape(len(X_train), -1) |
| X_te_flat = X_test.astype(np.float32).reshape(len(X_test), -1) |
| else: |
| X_train = np.nan_to_num(X_train, nan=0.0, posinf=0.0, neginf=0.0) |
| X_test = np.nan_to_num(X_test, nan=0.0, posinf=0.0, neginf=0.0) |
| X_tr_flat = X_train.reshape(len(X_train), -1) |
| X_te_flat = X_test.reshape(len(X_test), -1) |
|
|
| |
| max_train = min(50_000, len(X_tr_flat)) |
| idx = np.random.choice(len(X_tr_flat), max_train, replace=False) |
| y_sub = y_train[idx] if len(y_train) > max_train else y_train |
|
|
| clf = RandomForestClassifier( |
| n_estimators=100, |
| random_state=42, |
| n_jobs=-1, |
| ) |
| clf.fit(X_tr_flat[idx], y_sub) |
| y_pred = clf.predict(X_te_flat) |
|
|
| metrics = compute_metrics(y_test, y_pred, "Random Forest") |
|
|
| def infer(x): |
| xf = x.astype(np.float32).reshape(len(x), -1) |
| return clf.predict(xf) |
|
|
| latency, throughput = measure_latency(infer, X_test) |
| log.info(" Latency : %.4f ms/session", latency) |
| log.info(" Throughput: %.1f sessions/sec", throughput) |
|
|
| metrics["latency_ms"] = latency |
| metrics["throughput"] = throughput |
| return metrics |
|
|
|
|
| def evaluate_waf( |
| X_test: np.ndarray, |
| y_test: np.ndarray, |
| dataset: str, |
| data_dir: Path, |
| ) -> dict: |
| """ |
| WAF simulation baseline — signature/pattern matching only. |
| |
| For CSIC 2010: checks if any token in session is UNK (proxy for suspicious/unseen URL pattern) |
| For flow datasets: flags sessions where Dst Port is in the known attack port list (very basic rule). |
| |
| This demonstrates why WAFs alone are insufficient.* |
| """ |
| log.info("Running WAF simulation...") |
|
|
| if dataset == "csic2010": |
| |
| |
| y_pred = (X_test == 1).any(axis=1).astype(int) |
|
|
| else: |
| |
| |
| SUSPICIOUS_PORTS = {21, 22, 23, 25, 53, 3306, 3389, 4444, 8080, 8443} |
| |
| |
| |
| |
| port_vals = X_test[:, :, 0] |
| y_pred = (np.abs(port_vals) > 2.0).any(axis=1).astype(int) |
|
|
| metrics = compute_metrics(y_test, y_pred, "WAF Simulation") |
|
|
| def infer(x): |
| if dataset == "csic2010": |
| return (x == 1).any(axis=1).astype(int) |
| else: |
| return (np.abs(x[:, :, 0]) > 2.0).any(axis=1).astype(int) |
|
|
| latency, throughput = measure_latency(infer, X_test) |
| log.info(" Latency : %.4f ms/session", latency) |
| log.info(" Throughput: %.1f sessions/sec", throughput) |
|
|
| metrics["latency_ms"] = latency |
| metrics["throughput"] = throughput |
| return metrics |
|
|
|
|
| def main(): |
| parser = argparse.ArgumentParser() |
| parser.add_argument( |
| "--dataset", required=True, choices=["csic2010", "cicids2018", "unsw"] |
| ) |
| parser.add_argument( |
| "--skip_baselines", |
| action="store_true", |
| help="Only evaluate LSTM model, skip baselines", |
| ) |
| parser.add_argument( |
| "--window", |
| type=int, |
| default=5, |
| help="Sliding window size (must match training)", |
| ) |
| args = parser.parse_args() |
|
|
| device = "cuda" if torch.cuda.is_available() else "cpu" |
| data_dir = Path("data/processed") |
| mdl_dir = Path("models") |
| res_dir = Path("results") |
| res_dir.mkdir(exist_ok=True) |
|
|
| run_id = f"{args.dataset}_w{args.window}" |
|
|
| log.info("=" * 55) |
| log.info("EVALUATION — %s (window=%d)", args.dataset.upper(), args.window) |
| log.info("=" * 55) |
|
|
| |
| X_test = np.load(data_dir / f"X_test_{run_id}.npy") |
| y_test = np.load(data_dir / f"y_test_{run_id}.npy") |
| X_train = np.load(data_dir / f"X_train_{run_id}.npy") |
|
|
| log.info("X_test shape: %s", X_test.shape) |
| log.info( |
| "y_test shape: %s (normal=%d attack=%d)", |
| y_test.shape, |
| (y_test == 0).sum(), |
| (y_test == 1).sum(), |
| ) |
|
|
| |
| |
| n_normal = min(len(X_train), 50_000) |
| n_attack = min((y_test == 1).sum(), 10_000) |
|
|
| X_rf_normal = X_train[:n_normal] |
| y_rf_normal = np.zeros(n_normal, dtype=int) |
|
|
| X_rf_attack = X_test[y_test == 1][:n_attack] |
| y_rf_attack = np.ones(n_attack, dtype=int) |
|
|
| X_rf_train = np.concatenate([X_rf_normal, X_rf_attack], axis=0) |
| y_rf_train = np.concatenate([y_rf_normal, y_rf_attack], axis=0) |
|
|
| |
| thresh_path = mdl_dir / f"threshold_{run_id}.json" |
| thresh_data = json.load(open(thresh_path)) |
| threshold = thresh_data["threshold"] |
| saved_window = thresh_data.get("window") |
| if saved_window is not None and saved_window != args.window: |
| raise ValueError( |
| f"Window mismatch: {thresh_path.name} was trained with " |
| f"window={saved_window}, but --window={args.window} was passed. " |
| f"Pass --window {saved_window} to match the trained model." |
| ) |
| log.info("Threshold: %.6f", threshold) |
|
|
| |
| if args.dataset == "csic2010": |
| checkpoint = torch.load( |
| mdl_dir / f"best_{run_id}.pt", map_location=device |
| ) |
| embed_weight = checkpoint["embedding.weight"] |
| vocab_size = embed_weight.shape[0] |
| model = build_model_csic2010(vocab_size=vocab_size, seq_len=args.window) |
| elif args.dataset == "cicids2018": |
| model = build_model_cicids2018(n_features=X_test.shape[2], seq_len=args.window) |
| else: |
| model = build_model_unsw(n_features=X_test.shape[2], seq_len=args.window) |
|
|
| model.load_state_dict( |
| torch.load(mdl_dir / f"best_{run_id}.pt", map_location=device) |
| ) |
|
|
| |
| all_results = [] |
|
|
| |
| lstm_metrics, errors = evaluate_lstm( |
| model, X_test, y_test, threshold, args.dataset, device |
| ) |
| all_results.append(lstm_metrics) |
|
|
| |
| |
| save_errors(errors, y_test, run_id, res_dir) |
|
|
| if not args.skip_baselines: |
| |
| if_metrics = evaluate_isolation_forest(X_train, X_test, y_test, args.dataset) |
| all_results.append(if_metrics) |
|
|
| |
| rf_metrics = evaluate_random_forest( |
| X_rf_train, y_rf_train, X_test, y_test, args.dataset |
| ) |
| all_results.append(rf_metrics) |
|
|
| |
| waf_metrics = evaluate_waf(X_test, y_test, args.dataset, data_dir) |
| all_results.append(waf_metrics) |
|
|
| |
| log.info("") |
| log.info("=" * 55) |
| log.info("COMPARISON TABLE — %s", args.dataset.upper()) |
| log.info("=" * 55) |
| log.info( |
| "%-22s %6s %6s %6s %6s %8s", "Model", "Prec", "Rec", "F1", "FPR", "Lat(ms)" |
| ) |
| log.info("-" * 55) |
| for r in all_results: |
| log.info( |
| "%-22s %6.4f %6.4f %6.4f %6.4f %8.4f", |
| r["model"], |
| r["precision"], |
| r["recall"], |
| r["f1"], |
| r["fpr"], |
| r["latency_ms"], |
| ) |
|
|
| |
| out = { |
| "dataset": args.dataset, |
| "window": args.window, |
| "results": all_results, |
| "error_distribution": { |
| "mean": lstm_metrics["error_mean"], |
| "std": lstm_metrics["error_std"], |
| "threshold": threshold, |
| }, |
| } |
|
|
| out_path = res_dir / f"evaluation_{run_id}.json" |
| with open(out_path, "w") as f: |
| json.dump(out, f, indent=2) |
| log.info("Results saved → %s", out_path) |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|