""" EXP-002/EXP-003: Lightweight classical classifier on acoustic features. Logistic regression first, per the Phase 2 brief — small, fast, and its coefficients are directly inspectable for error analysis (which features drive END vs CONTINUE decisions). A small GBM (e.g. sklearn GradientBoostingClassifier or HistGradientBoostingClassifier) is a natural next step if logistic regression underfits, but per the "don't assume the final architecture" rule, that's a decision for after real results, not before. """ from __future__ import annotations from dataclasses import dataclass, field import numpy as np from sklearn.linear_model import LogisticRegression from sklearn.preprocessing import StandardScaler from sklearn.pipeline import Pipeline from .features import FeatureConfig, extract_features FEATURE_ORDER_CACHE_KEY = "_feature_order" def features_dict_to_vector(feats: dict, feature_order: list[str]) -> np.ndarray: """Convert a feature dict to a fixed-order vector, filling missing keys (e.g. a recent-window feature unavailable because the clip was shorter than that window) with NaN, which the sklearn pipeline's imputer step then handles explicitly — never silently zero-filled, since zero is a meaningful RMS value and would bias the model. """ return np.array([feats.get(k, np.nan) for k in feature_order], dtype=np.float64) def build_feature_matrix( feature_dicts: list[dict], ) -> tuple[np.ndarray, list[str]]: """Build an (n_samples, n_features) matrix from a list of per-clip feature dicts (as produced by features.extract_features). Feature order is the sorted union of all keys seen, excluding non-numeric bookkeeping keys (e.g. `*_available`, `too_short_for_framing` bool flags are kept as 0/1 numeric since availability itself may be predictive — e.g. very short clips may correlate with the label). """ all_keys = set() for f in feature_dicts: all_keys.update(f.keys()) feature_order = sorted(all_keys) rows = [] for f in feature_dicts: row = [] for k in feature_order: v = f.get(k, np.nan) if isinstance(v, bool): v = float(v) row.append(v if v is not None else np.nan) rows.append(row) return np.array(rows, dtype=np.float64), feature_order @dataclass class ClassifierBaselineConfig: feature_cfg: FeatureConfig = field(default_factory=FeatureConfig) C: float = 1.0 max_iter: int = 1000 class_weight: str | None = "balanced" random_state: int = 42 class ClassifierBaseline: """sklearn Pipeline: median-impute -> standardize -> logistic regression. Median imputation (not mean) chosen because several features are ratios/durations that can have skewed distributions (e.g. trailing_silence_sec, energy_slope) where the median is a more robust fill value than the mean for the small number of missing entries expected (e.g. recent-window features on very short clips). """ def __init__(self, cfg: ClassifierBaselineConfig = ClassifierBaselineConfig()): self.cfg = cfg self.feature_order_: list[str] | None = None self.pipeline_: Pipeline | None = None def _featurize(self, audios: list[np.ndarray]) -> list[dict]: return [extract_features(a, self.cfg.feature_cfg) for a in audios] def fit(self, audios: list[np.ndarray], labels: list[bool]): from sklearn.impute import SimpleImputer feats = self._featurize(audios) X, feature_order = build_feature_matrix(feats) self.feature_order_ = feature_order y = np.array(labels, dtype=int) self.pipeline_ = Pipeline([ ("impute", SimpleImputer(strategy="median")), ("scale", StandardScaler()), ("clf", LogisticRegression( C=self.cfg.C, max_iter=self.cfg.max_iter, class_weight=self.cfg.class_weight, random_state=self.cfg.random_state, )), ]) self.pipeline_.fit(X, y) return self def predict_proba(self, audios: list[np.ndarray]) -> np.ndarray: if self.pipeline_ is None: raise RuntimeError("Call fit() before predict_proba().") feats = self._featurize(audios) X = np.array( [[f.get(k, np.nan) for k in self.feature_order_] for f in feats], dtype=np.float64, ) return self.pipeline_.predict_proba(X)[:, 1] def predict(self, audios: list[np.ndarray]) -> np.ndarray: return self.predict_proba(audios) >= 0.5 def param_count(self) -> int: if self.pipeline_ is None: raise RuntimeError("Call fit() before param_count().") clf: LogisticRegression = self.pipeline_.named_steps["clf"] # coefficients + intercept return int(clf.coef_.size + clf.intercept_.size) def model_size_bytes(self) -> int: # Rough estimate: coefficients + intercept + scaler mean/scale + # imputer statistics, all float64. if self.pipeline_ is None: raise RuntimeError("Call fit() before model_size_bytes().") clf = self.pipeline_.named_steps["clf"] scaler = self.pipeline_.named_steps["scale"] imputer = self.pipeline_.named_steps["impute"] n_params = ( clf.coef_.size + clf.intercept_.size + scaler.mean_.size + scaler.scale_.size + imputer.statistics_.size ) return int(n_params * 8) def top_features(self, k: int = 15) -> list[tuple[str, float]]: """Feature name -> coefficient, sorted by |coefficient| descending. Directly useful for error analysis / explaining what the model actually learned (docs/ERROR_ANALYSIS.md). """ if self.pipeline_ is None or self.feature_order_ is None: raise RuntimeError("Call fit() first.") coefs = self.pipeline_.named_steps["clf"].coef_[0] pairs = list(zip(self.feature_order_, coefs)) pairs.sort(key=lambda p: abs(p[1]), reverse=True) return pairs[:k]