from __future__ import annotations import librosa import numpy as np SR_TARGET = 16000 # common processing rate for MFCCs (also silero-vad's native rate) _VAD_MODEL = None def _vad_model(): global _VAD_MODEL if _VAD_MODEL is None: from silero_vad import load_silero_vad # lazy import — heavy dep _VAD_MODEL = load_silero_vad() return _VAD_MODEL def load_audio(path: str, sr: int = SR_TARGET) -> tuple[np.ndarray, int]: y, real_sr = librosa.load(path, sr=sr, mono=True) return y.astype(np.float32), real_sr def slice_word(y: np.ndarray, sr: int, start_s: float, end_s: float) -> np.ndarray: s = max(0, int(round(start_s * sr))) e = min(len(y), int(round(end_s * sr))) return y[s:e] def trim_silence(y: np.ndarray, sr: int, top_db: float = 30.0) -> np.ndarray: """[fallback] Energy-threshold silence trim. Used when VAD returns empty.""" if len(y) < int(0.02 * sr): return y yt, _ = librosa.effects.trim(y, top_db=top_db) return yt if len(yt) > 0 else y def trim_silence_vad( y: np.ndarray, sr: int, threshold: float = 0.5, min_speech_ms: int = 30, min_silence_ms: int = 80, ) -> np.ndarray: """VAD-based trim. Crops leading/trailing non-speech around a word slice. Robust to ambient noise, breath, and audible-but-non-phonetic sound that the energy-threshold trim would keep (e.g., the deliberate pause attached to "a" in rec.3 — measured 0.8 s at top_db=30, ~0.05 s with VAD). For slices shorter than ~100 ms VAD is unreliable; we fall back to the energy trim (which mostly does nothing on such short slices anyway). """ import torch if len(y) < int(0.1 * sr): return trim_silence(y, sr) # silero-vad operates on 8 kHz or 16 kHz only if sr not in (8000, 16000): y_vad = librosa.resample(y, orig_sr=sr, target_sr=SR_TARGET) scale = sr / SR_TARGET vad_sr = SR_TARGET else: y_vad = y scale = 1.0 vad_sr = sr from silero_vad import get_speech_timestamps audio_t = torch.from_numpy(y_vad.astype(np.float32)) speech = get_speech_timestamps( audio_t, _vad_model(), sampling_rate=vad_sr, threshold=threshold, min_speech_duration_ms=min_speech_ms, min_silence_duration_ms=min_silence_ms, ) if not speech: return trim_silence(y, sr) # fallback to energy trim # Concatenate speech segments, dropping any inter-segment silence. # Rationale: silero's min_silence_duration_ms=80 ms means any inter-segment gap # is ≥80 ms, which is longer than typical stop closures (40–70 ms in fast Czech), # so a multi-segment split signals a real pause/hesitation that should be excluded. pieces: list[np.ndarray] = [] for seg in speech: s = max(0, int(round(seg["start"] * scale))) e = min(len(y), int(round(seg["end"] * scale))) if e > s: pieces.append(y[s:e]) if not pieces: return trim_silence(y, sr) return pieces[0] if len(pieces) == 1 else np.concatenate(pieces) def _mfcc_cmn(y: np.ndarray, sr: int, n_mfcc: int = 13, hop_ms: float = 10.0) -> np.ndarray: hop = max(1, int(sr * hop_ms / 1000.0)) n_fft = 512 if sr <= 16000 else 1024 M = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=n_mfcc, hop_length=hop, n_fft=n_fft) # cepstral mean normalization — removes channel / microphone bias per stream M = M - M.mean(axis=1, keepdims=True) return M def spectral_distance( orig_wav: np.ndarray, orig_sr: int, tts_wav: np.ndarray, tts_sr: int, target_sr: int = SR_TARGET, ) -> float: """DTW-aligned mean MFCC distance between original word and its TTS rendering. Higher = more spectrally distant from canonical TTS = more reduced/different. Returns NaN for too-short slices (cannot compute MFCC reliably). """ if orig_sr != target_sr: orig_wav = librosa.resample(orig_wav, orig_sr=orig_sr, target_sr=target_sr) if tts_sr != target_sr: tts_wav = librosa.resample(tts_wav, orig_sr=tts_sr, target_sr=target_sr) # need at least ~30 ms of audio for a meaningful MFCC sequence if len(orig_wav) < int(0.03 * target_sr) or len(tts_wav) < int(0.03 * target_sr): return float("nan") M_orig = _mfcc_cmn(orig_wav, target_sr) M_tts = _mfcc_cmn(tts_wav, target_sr) if M_orig.shape[1] < 2 or M_tts.shape[1] < 2: return float("nan") D, wp = librosa.sequence.dtw(M_orig, M_tts, metric="euclidean") return float(D[-1, -1] / max(1, len(wp))) def center_scores(rows: list[dict]) -> list[dict]: """Add centered scores (mean = 1) to each row. Convention: higher score = less reduced (closer to canonical / longer). duration_score = (orig_dur / tts_dur) / mean(orig_dur / tts_dur) spectral_score = mean(spec_dist) / spec_dist (inverted, same direction as duration) combined_score = arithmetic mean of the two (NaN-safe) """ dur_ratios = np.array([r["duration_ratio"] for r in rows], dtype=float) spec_dists = np.array([r["spectral_distance"] for r in rows], dtype=float) mean_dr = np.nanmean(dur_ratios) if np.any(~np.isnan(dur_ratios)) else float("nan") mean_sd = np.nanmean(spec_dists) if np.any(~np.isnan(spec_dists)) else float("nan") for r in rows: dr = r["duration_ratio"] sd = r["spectral_distance"] r["duration_score"] = (dr / mean_dr) if (mean_dr and not np.isnan(dr)) else float("nan") r["spectral_score"] = (mean_sd / sd) if (sd and not np.isnan(sd) and not np.isnan(mean_sd)) else float("nan") components = [v for v in (r["duration_score"], r["spectral_score"]) if not np.isnan(v)] r["combined_score"] = float(np.mean(components)) if components else float("nan") return rows