Spaces:
Running on Zero
Running on Zero
| """ | |
| Evaluation metrics shared by both baselines (and, later, any Whisper-based | |
| model), so every experiment in experiments/EXPERIMENTS.md is measured the | |
| same way. | |
| Definitions used throughout this project (stated once, here, so every doc | |
| and script agrees): | |
| - Positive class = END (endpoint_bool = True / turn complete). | |
| - False END = model predicted END but the true label was CONTINUE. | |
| This is a "premature endpoint" — the costly error in a live | |
| voice agent (interrupts the user). | |
| - False CONTINUE = model predicted CONTINUE but the true label was END. | |
| This is a "delayed endpoint" — the agent waits too long | |
| before responding. | |
| True wall-clock endpoint latency cannot be computed from this dataset (no | |
| conversation-level timestamps — docs/INITIAL_ANALYSIS.md §7). Anywhere that | |
| number would normally go, code in this module reports | |
| `"endpoint_latency_ms": None` with a reason string, rather than a fabricated | |
| number or a silently missing field. | |
| """ | |
| from __future__ import annotations | |
| import time | |
| from typing import Callable | |
| import numpy as np | |
| from sklearn.metrics import confusion_matrix, precision_recall_fscore_support | |
| def classification_metrics(y_true: np.ndarray, y_pred: np.ndarray) -> dict: | |
| y_true = np.asarray(y_true, dtype=bool) | |
| y_pred = np.asarray(y_pred, dtype=bool) | |
| if len(y_true) == 0: | |
| return {"n": 0, "note": "empty evaluation set"} | |
| accuracy = float(np.mean(y_true == y_pred)) | |
| precision, recall, f1, _ = precision_recall_fscore_support( | |
| y_true, y_pred, average="binary", zero_division=0 | |
| ) | |
| cm = confusion_matrix(y_true, y_pred, labels=[False, True]) | |
| tn, fp, fn, tp = cm.ravel() | |
| n_actual_continue = tn + fp # true label = CONTINUE | |
| n_actual_end = fn + tp # true label = END | |
| false_end_rate = float(fp / n_actual_continue) if n_actual_continue > 0 else None | |
| false_continue_rate = float(fn / n_actual_end) if n_actual_end > 0 else None | |
| return { | |
| "n": int(len(y_true)), | |
| "accuracy": accuracy, | |
| "precision": float(precision), | |
| "recall": float(recall), | |
| "f1": float(f1), | |
| "confusion_matrix": {"tn": int(tn), "fp": int(fp), "fn": int(fn), "tp": int(tp)}, | |
| "false_end_rate": false_end_rate, # premature endpoint rate | |
| "false_continue_rate": false_continue_rate, # delayed endpoint rate | |
| "endpoint_latency_ms": None, | |
| "endpoint_latency_note": ( | |
| "Not measurable: dataset has no conversation-level timestamps " | |
| "(see docs/INITIAL_ANALYSIS.md §7)." | |
| ), | |
| } | |
| def slice_metrics( | |
| y_true: np.ndarray, | |
| y_pred: np.ndarray, | |
| slice_labels: np.ndarray, | |
| min_samples: int = 20, | |
| ) -> dict: | |
| """Per-slice metrics (e.g. by language, by synthetic flag). Slices with | |
| fewer than `min_samples` are excluded and reported as such, per the | |
| Phase 2 brief's "only report slices with sufficient samples" rule — | |
| exclusion is explicit, not silent. | |
| """ | |
| y_true = np.asarray(y_true, dtype=bool) | |
| y_pred = np.asarray(y_pred, dtype=bool) | |
| slice_labels = np.asarray(slice_labels) | |
| results = {} | |
| excluded = [] | |
| for val in sorted(set(slice_labels.tolist()), key=str): | |
| mask = slice_labels == val | |
| n = int(mask.sum()) | |
| if n < min_samples: | |
| excluded.append({"slice": str(val), "n": n}) | |
| continue | |
| results[str(val)] = classification_metrics(y_true[mask], y_pred[mask]) | |
| return {"slices": results, "excluded_insufficient_n": excluded, "min_samples": min_samples} | |
| def measure_inference_latency( | |
| predict_fn: Callable, | |
| inputs: list, | |
| n_warmup: int = 3, | |
| n_repeats: int = 1, | |
| ) -> dict: | |
| """Wall-clock latency per single-item inference call, on THIS machine | |
| (CPU, exact spec unstated — see caveat below). Reported per-call, not | |
| batched, since production turn-detection is a per-utterance call. | |
| Honesty caveat this function always attaches: measured latency is | |
| environment-specific (CPU model, load, Python/library versions all | |
| matter). It should be reported as "measured on this dev/eval machine", | |
| never presented as a universal number, and never compared directly to | |
| a different environment's numbers (e.g. the upstream Smart Turn | |
| project's reported ~65ms Pipecat Cloud figure) without noting that | |
| caveat explicitly in the write-up. | |
| """ | |
| for x in inputs[:n_warmup]: | |
| predict_fn(x) | |
| times_ms = [] | |
| for _ in range(n_repeats): | |
| for x in inputs: | |
| t0 = time.perf_counter() | |
| predict_fn(x) | |
| t1 = time.perf_counter() | |
| times_ms.append((t1 - t0) * 1000.0) | |
| times_ms = np.array(times_ms) | |
| return { | |
| "n_calls": int(len(times_ms)), | |
| "mean_ms": float(np.mean(times_ms)), | |
| "median_ms": float(np.median(times_ms)), | |
| "p95_ms": float(np.percentile(times_ms, 95)), | |
| "p99_ms": float(np.percentile(times_ms, 99)), | |
| "min_ms": float(np.min(times_ms)), | |
| "max_ms": float(np.max(times_ms)), | |
| "caveat": "Measured on this dev/eval machine; not directly comparable across environments.", | |
| } | |