Spaces:
Running on Zero
Running on Zero
| """ | |
| EXP-002/EXP-003: Lightweight classical classifier on acoustic features. | |
| Logistic regression first, per the Phase 2 brief — small, fast, and its | |
| coefficients are directly inspectable for error analysis (which features | |
| drive END vs CONTINUE decisions). A small GBM (e.g. sklearn | |
| GradientBoostingClassifier or HistGradientBoostingClassifier) is a natural | |
| next step if logistic regression underfits, but per the "don't assume the | |
| final architecture" rule, that's a decision for after real results, not | |
| before. | |
| """ | |
| from __future__ import annotations | |
| from dataclasses import dataclass, field | |
| import numpy as np | |
| from sklearn.linear_model import LogisticRegression | |
| from sklearn.preprocessing import StandardScaler | |
| from sklearn.pipeline import Pipeline | |
| from .features import FeatureConfig, extract_features | |
| FEATURE_ORDER_CACHE_KEY = "_feature_order" | |
| def features_dict_to_vector(feats: dict, feature_order: list[str]) -> np.ndarray: | |
| """Convert a feature dict to a fixed-order vector, filling missing keys | |
| (e.g. a recent-window feature unavailable because the clip was shorter | |
| than that window) with NaN, which the sklearn pipeline's imputer step | |
| then handles explicitly — never silently zero-filled, since zero is a | |
| meaningful RMS value and would bias the model. | |
| """ | |
| return np.array([feats.get(k, np.nan) for k in feature_order], dtype=np.float64) | |
| def build_feature_matrix( | |
| feature_dicts: list[dict], | |
| ) -> tuple[np.ndarray, list[str]]: | |
| """Build an (n_samples, n_features) matrix from a list of per-clip | |
| feature dicts (as produced by features.extract_features). Feature order | |
| is the sorted union of all keys seen, excluding non-numeric bookkeeping | |
| keys (e.g. `*_available`, `too_short_for_framing` bool flags are kept | |
| as 0/1 numeric since availability itself may be predictive — e.g. very | |
| short clips may correlate with the label). | |
| """ | |
| all_keys = set() | |
| for f in feature_dicts: | |
| all_keys.update(f.keys()) | |
| feature_order = sorted(all_keys) | |
| rows = [] | |
| for f in feature_dicts: | |
| row = [] | |
| for k in feature_order: | |
| v = f.get(k, np.nan) | |
| if isinstance(v, bool): | |
| v = float(v) | |
| row.append(v if v is not None else np.nan) | |
| rows.append(row) | |
| return np.array(rows, dtype=np.float64), feature_order | |
| class ClassifierBaselineConfig: | |
| feature_cfg: FeatureConfig = field(default_factory=FeatureConfig) | |
| C: float = 1.0 | |
| max_iter: int = 1000 | |
| class_weight: str | None = "balanced" | |
| random_state: int = 42 | |
| class ClassifierBaseline: | |
| """sklearn Pipeline: median-impute -> standardize -> logistic regression. | |
| Median imputation (not mean) chosen because several features are | |
| ratios/durations that can have skewed distributions (e.g. | |
| trailing_silence_sec, energy_slope) where the median is a more robust | |
| fill value than the mean for the small number of missing entries | |
| expected (e.g. recent-window features on very short clips). | |
| """ | |
| def __init__(self, cfg: ClassifierBaselineConfig = ClassifierBaselineConfig()): | |
| self.cfg = cfg | |
| self.feature_order_: list[str] | None = None | |
| self.pipeline_: Pipeline | None = None | |
| def _featurize(self, audios: list[np.ndarray]) -> list[dict]: | |
| return [extract_features(a, self.cfg.feature_cfg) for a in audios] | |
| def fit(self, audios: list[np.ndarray], labels: list[bool]): | |
| from sklearn.impute import SimpleImputer | |
| feats = self._featurize(audios) | |
| X, feature_order = build_feature_matrix(feats) | |
| self.feature_order_ = feature_order | |
| y = np.array(labels, dtype=int) | |
| self.pipeline_ = Pipeline([ | |
| ("impute", SimpleImputer(strategy="median")), | |
| ("scale", StandardScaler()), | |
| ("clf", LogisticRegression( | |
| C=self.cfg.C, | |
| max_iter=self.cfg.max_iter, | |
| class_weight=self.cfg.class_weight, | |
| random_state=self.cfg.random_state, | |
| )), | |
| ]) | |
| self.pipeline_.fit(X, y) | |
| return self | |
| def predict_proba(self, audios: list[np.ndarray]) -> np.ndarray: | |
| if self.pipeline_ is None: | |
| raise RuntimeError("Call fit() before predict_proba().") | |
| feats = self._featurize(audios) | |
| X = np.array( | |
| [[f.get(k, np.nan) for k in self.feature_order_] for f in feats], | |
| dtype=np.float64, | |
| ) | |
| return self.pipeline_.predict_proba(X)[:, 1] | |
| def predict(self, audios: list[np.ndarray]) -> np.ndarray: | |
| return self.predict_proba(audios) >= 0.5 | |
| def param_count(self) -> int: | |
| if self.pipeline_ is None: | |
| raise RuntimeError("Call fit() before param_count().") | |
| clf: LogisticRegression = self.pipeline_.named_steps["clf"] | |
| # coefficients + intercept | |
| return int(clf.coef_.size + clf.intercept_.size) | |
| def model_size_bytes(self) -> int: | |
| # Rough estimate: coefficients + intercept + scaler mean/scale + | |
| # imputer statistics, all float64. | |
| if self.pipeline_ is None: | |
| raise RuntimeError("Call fit() before model_size_bytes().") | |
| clf = self.pipeline_.named_steps["clf"] | |
| scaler = self.pipeline_.named_steps["scale"] | |
| imputer = self.pipeline_.named_steps["impute"] | |
| n_params = ( | |
| clf.coef_.size + clf.intercept_.size | |
| + scaler.mean_.size + scaler.scale_.size | |
| + imputer.statistics_.size | |
| ) | |
| return int(n_params * 8) | |
| def top_features(self, k: int = 15) -> list[tuple[str, float]]: | |
| """Feature name -> coefficient, sorted by |coefficient| descending. | |
| Directly useful for error analysis / explaining what the model | |
| actually learned (docs/ERROR_ANALYSIS.md). | |
| """ | |
| if self.pipeline_ is None or self.feature_order_ is None: | |
| raise RuntimeError("Call fit() first.") | |
| coefs = self.pipeline_.named_steps["clf"].coef_[0] | |
| pairs = list(zip(self.feature_order_, coefs)) | |
| pairs.sort(key=lambda p: abs(p[1]), reverse=True) | |
| return pairs[:k] | |