""" Evaluation metrics shared by both baselines (and, later, any Whisper-based model), so every experiment in experiments/EXPERIMENTS.md is measured the same way. Definitions used throughout this project (stated once, here, so every doc and script agrees): - Positive class = END (endpoint_bool = True / turn complete). - False END = model predicted END but the true label was CONTINUE. This is a "premature endpoint" — the costly error in a live voice agent (interrupts the user). - False CONTINUE = model predicted CONTINUE but the true label was END. This is a "delayed endpoint" — the agent waits too long before responding. True wall-clock endpoint latency cannot be computed from this dataset (no conversation-level timestamps — docs/INITIAL_ANALYSIS.md §7). Anywhere that number would normally go, code in this module reports `"endpoint_latency_ms": None` with a reason string, rather than a fabricated number or a silently missing field. """ from __future__ import annotations import time from typing import Callable import numpy as np from sklearn.metrics import confusion_matrix, precision_recall_fscore_support def classification_metrics(y_true: np.ndarray, y_pred: np.ndarray) -> dict: y_true = np.asarray(y_true, dtype=bool) y_pred = np.asarray(y_pred, dtype=bool) if len(y_true) == 0: return {"n": 0, "note": "empty evaluation set"} accuracy = float(np.mean(y_true == y_pred)) precision, recall, f1, _ = precision_recall_fscore_support( y_true, y_pred, average="binary", zero_division=0 ) cm = confusion_matrix(y_true, y_pred, labels=[False, True]) tn, fp, fn, tp = cm.ravel() n_actual_continue = tn + fp # true label = CONTINUE n_actual_end = fn + tp # true label = END false_end_rate = float(fp / n_actual_continue) if n_actual_continue > 0 else None false_continue_rate = float(fn / n_actual_end) if n_actual_end > 0 else None return { "n": int(len(y_true)), "accuracy": accuracy, "precision": float(precision), "recall": float(recall), "f1": float(f1), "confusion_matrix": {"tn": int(tn), "fp": int(fp), "fn": int(fn), "tp": int(tp)}, "false_end_rate": false_end_rate, # premature endpoint rate "false_continue_rate": false_continue_rate, # delayed endpoint rate "endpoint_latency_ms": None, "endpoint_latency_note": ( "Not measurable: dataset has no conversation-level timestamps " "(see docs/INITIAL_ANALYSIS.md §7)." ), } def slice_metrics( y_true: np.ndarray, y_pred: np.ndarray, slice_labels: np.ndarray, min_samples: int = 20, ) -> dict: """Per-slice metrics (e.g. by language, by synthetic flag). Slices with fewer than `min_samples` are excluded and reported as such, per the Phase 2 brief's "only report slices with sufficient samples" rule — exclusion is explicit, not silent. """ y_true = np.asarray(y_true, dtype=bool) y_pred = np.asarray(y_pred, dtype=bool) slice_labels = np.asarray(slice_labels) results = {} excluded = [] for val in sorted(set(slice_labels.tolist()), key=str): mask = slice_labels == val n = int(mask.sum()) if n < min_samples: excluded.append({"slice": str(val), "n": n}) continue results[str(val)] = classification_metrics(y_true[mask], y_pred[mask]) return {"slices": results, "excluded_insufficient_n": excluded, "min_samples": min_samples} def measure_inference_latency( predict_fn: Callable, inputs: list, n_warmup: int = 3, n_repeats: int = 1, ) -> dict: """Wall-clock latency per single-item inference call, on THIS machine (CPU, exact spec unstated — see caveat below). Reported per-call, not batched, since production turn-detection is a per-utterance call. Honesty caveat this function always attaches: measured latency is environment-specific (CPU model, load, Python/library versions all matter). It should be reported as "measured on this dev/eval machine", never presented as a universal number, and never compared directly to a different environment's numbers (e.g. the upstream Smart Turn project's reported ~65ms Pipecat Cloud figure) without noting that caveat explicitly in the write-up. """ for x in inputs[:n_warmup]: predict_fn(x) times_ms = [] for _ in range(n_repeats): for x in inputs: t0 = time.perf_counter() predict_fn(x) t1 = time.perf_counter() times_ms.append((t1 - t0) * 1000.0) times_ms = np.array(times_ms) return { "n_calls": int(len(times_ms)), "mean_ms": float(np.mean(times_ms)), "median_ms": float(np.median(times_ms)), "p95_ms": float(np.percentile(times_ms, 95)), "p99_ms": float(np.percentile(times_ms, 99)), "min_ms": float(np.min(times_ms)), "max_ms": float(np.max(times_ms)), "caveat": "Measured on this dev/eval machine; not directly comparable across environments.", }