recognizer-api / raaga_id /features.py
Sathya-UM's picture
Deploy headless recognizer API (Docker)
c25f760 verified
Raw
History Blame Contribute Delete
12.4 kB
"""Feature extraction.
Modeling ladder (D16): tonic-normalized pitch-class/swara histogram + XGBoost floor
-> TDMS -> CNN on mel/CQT. The floor (D16 step 1) is a librosa chroma histogram rolled
so the tonic (Sa) sits at bin 0 (D5). The tonic comes from Saraga's ctonic annotation
when available (precise), else a drone-argmax estimate; essentia TonicIndianArtMusic is
the inference-time upgrade for unlabelled clips (Phase 2).
"""
from __future__ import annotations
import numpy as np
from .config import CLIP_SECONDS, MIN_CLIP_SECONDS, MIN_VOICED_FRAC, SAMPLE_RATE
def _voiced_fraction(seg) -> float:
"""Fraction of a window's predominant-melody frames that carry a pitch (f0>0). Low for
percussion/speech/applause/silence (the junk gate, D6/D8); high for real melody."""
seg = np.asarray(seg, dtype=float)
return float((seg > 0).mean()) if seg.size else 0.0
def load_audio(path: str, sr: int = SAMPLE_RATE, duration: float | None = None) -> np.ndarray:
"""Decode any format to mono float32 at `sr` (ffmpeg handles mp3/m4a).
`duration` caps the decode length — Saraga tracks run 20-30 min, so decoding the
whole file to use only the first N windows wastes most of the work.
"""
import librosa
y, _ = librosa.load(path, sr=sr, mono=True, duration=duration)
return y.astype(np.float32)
def estimate_tonic_pc(y: np.ndarray, sr: int = SAMPLE_RATE) -> int:
"""Cheap tonic (Sa) estimate: the most energetic pitch class over the clip.
Carnatic performances carry a continuous tanpura drone on Sa, so the summed
chromagram peaks at the tonic. This is the librosa-only stand-in for the precise
essentia `TonicIndianArtMusic` used in Phase 2 (D5).
"""
import librosa
chroma = librosa.feature.chroma_cqt(y=y, sr=sr)
return int(chroma.mean(axis=1).argmax())
def tonic_pc_from_hz(hz: float) -> int:
"""Map a tonic frequency (Hz) to its chroma pitch class (0-11). Used with a
precise tonic — Saraga's ctonic annotation, or essentia at inference."""
import librosa
return int(round(librosa.hz_to_midi(hz))) % 12
# 12 semitone positions relative to Sa (common Carnatic labels for the swara-sthanas).
SWARA_LABELS = ["S", "R1", "R2", "G2", "G3", "M1", "M2", "P", "D1", "D2", "N2", "N3"]
# The seven swaras every learner knows. For the "how to hear this raaga" panel we fold the 12
# chromatic positions down to these names (sa ri ga ma pa da ni) — the beginner's vocabulary,
# not the R1/R2/G2/G3 sthana subscripts. Each name groups its sthana variants.
SWARA7 = ["Sa", "Ri", "Ga", "Ma", "Pa", "Da", "Ni"]
_SWARA7_GROUPS = [[0], [1, 2], [3, 4], [5, 6], [7], [8, 9], [10, 11]] # indices into SWARA_LABELS
def to_swaras7(profile):
"""Fold a 12-position swara profile to the seven named swaras (SWARA7), summing each swara's
sthana variants. Returns (names, values). The friendly display unit for new learners."""
p = np.asarray(profile, dtype=float)
vals = np.array([p[g].sum() for g in _SWARA7_GROUPS])
return SWARA7, vals
def pcd_to_swaras(pcd) -> np.ndarray:
"""Fold a fine PCD (n_bins, a multiple of 12) to a 12-position swara profile
(one bin per semitone), normalized — the human-readable raaga fingerprint."""
a = np.asarray(pcd, dtype=float)
a = a.reshape(12, a.size // 12).sum(axis=1)
return a / (a.sum() + 1e-9)
def pitch_class_histogram(f0_hz, tonic_hz: float, n_bins: int = 120) -> np.ndarray:
"""Tonic-normalized pitch-class distribution (PCD) from a predominant-melody pitch
track — the classic raaga fingerprint (D16). Each voiced f0 becomes cents relative
to Sa, folded to one octave, binned and normalized; unvoiced frames (f0<=0) are
ignored. Works on any (pitch, tonic) source, so Saraga and IAMRRD pool cleanly.
Returns zeros when nothing is voiced.
"""
f0 = np.asarray(f0_hz, dtype=float)
voiced = f0[f0 > 0]
if voiced.size == 0:
return np.zeros(n_bins, dtype=np.float32)
cents = 1200.0 * np.log2(voiced / tonic_hz)
bins = np.mod((np.mod(cents, 1200.0) / (1200.0 / n_bins)).astype(int), n_bins)
hist = np.bincount(bins, minlength=n_bins).astype(np.float32)
return hist / hist.sum()
def tdms(times, f0_hz, tonic_hz: float, delay: float = 0.3, n_bins: int = 48) -> np.ndarray:
"""Time-Delayed Melody Surface (Gulati et al., ISMIR 2016) — the gamaka feature (refines
D16). A 2-D histogram of (pitch(t), pitch(t+delay)) over the tonic-normalized, octave-folded
melody: a held note sits on the diagonal, while gamaka/transitions smear off-diagonal in a
raaga-characteristic way. This keeps the *movement* the 1-D PCD discards — the information
that should separate allied raagas (Mōhanaṁ/Bilahari/Bēgaḍa) which share a scale.
Returns a flat (n_bins*n_bins,) surface normalized to sum 1, or zeros when too few voiced
pairs. `delay` is in seconds; the frame hop is inferred from `times`.
"""
times = np.asarray(times, dtype=float)
f0 = np.asarray(f0_hz, dtype=float)
zeros = np.zeros(n_bins * n_bins, dtype=np.float32)
if times.size < 2:
return zeros
voiced = f0 > 0
cents = np.zeros_like(f0)
cents[voiced] = np.mod(1200.0 * np.log2(f0[voiced] / tonic_hz), 1200.0)
bins = np.mod((cents / (1200.0 / n_bins)).astype(int), n_bins)
hop = float(np.median(np.diff(times)))
d = max(1, int(round(delay / hop)))
if d >= bins.size:
return zeros
a, b = bins[:-d], bins[d:]
both = voiced[:-d] & voiced[d:] # count a pair only if both endpoints are voiced
a, b = a[both], b[both]
if a.size == 0:
return zeros
surf = np.zeros((n_bins, n_bins), dtype=np.float64)
np.add.at(surf, (a, b), 1.0)
return (surf / surf.sum()).astype(np.float32).ravel()
def tdms_windows(
times,
f0_hz,
tonic_hz: float,
window_s: float = CLIP_SECONDS,
hop_s: float = CLIP_SECONDS,
max_windows: int | None = None,
delay: float = 0.3,
n_bins: int = 48,
min_voiced: float = MIN_VOICED_FRAC,
) -> list[np.ndarray]:
"""Slide a window over a predominant-melody pitch track -> one TDMS surface per window
(the delay-surface analog of pitch_windows, D28). Slices identically to pitch_windows so
surfaces aggregate exactly like PCD windows. Windows voiced less than `min_voiced` of the
time are dropped (junk gate, D6/D8 — percussion/speech/silence, no stable melody to track).
"""
times = np.asarray(times, dtype=float)
f0 = np.asarray(f0_hz, dtype=float)
if times.size == 0 or times[-1] < MIN_CLIP_SECONDS:
return []
end = float(times[-1])
out: list[np.ndarray] = []
start = 0.0
while start < end:
if min(start + window_s, end) - start < MIN_CLIP_SECONDS:
break
mask = (times >= start) & (times < start + window_s)
if _voiced_fraction(f0[mask]) >= min_voiced:
surf = tdms(times[mask], f0[mask], tonic_hz, delay=delay, n_bins=n_bins)
if surf.sum() > 0:
out.append(surf)
if max_windows and len(out) >= max_windows:
break
start += hop_s
return out
def model_windows(times, f0_hz, tonic_hz: float, max_windows: int | None = None,
hop_s: float | None = None) -> list[np.ndarray]:
"""THE production model feature (D28): windowed TDMS at the config's winning settings.
train / evaluate / calibrate / inference all call this one function, so the feature can
never drift between how the model is trained and how it's served. `hop_s` overrides only the
stride — inference passes a smaller hop so a short clip yields more windows to average (each
window is still classified independently, so overlap is safe and doesn't change the feature).
"""
from .config import TDMS_BINS, TDMS_DELAY, TDMS_HOP_S, TDMS_MAX_WINDOWS, TDMS_WINDOW_S
return tdms_windows(times, f0_hz, tonic_hz, window_s=TDMS_WINDOW_S, hop_s=hop_s or TDMS_HOP_S,
max_windows=max_windows or TDMS_MAX_WINDOWS, delay=TDMS_DELAY, n_bins=TDMS_BINS)
def frame_vector(y: np.ndarray, sr: int = SAMPLE_RATE, tonic_pc: int = 0) -> np.ndarray:
"""Tonic-relative pitch-class descriptor for one window (D16 step 1).
Raaga lives in the pitch classes *relative to Sa*, so we roll the chromagram so
the tonic pitch class -> bin 0, then summarize as a normalized 12-bin pitch-class
histogram (the raaga fingerprint) + per-bin spread. Timbre features (MFCC,
spectral shape) are deliberately EXCLUDED: they encode recording/instrument
identity and make the model memorize tracks instead of learning raaga.
"""
import librosa
if y.size == 0:
raise ValueError("empty audio")
chroma = librosa.feature.chroma_cqt(y=y, sr=sr)
chroma = np.roll(chroma, -tonic_pc, axis=0) # tonic -> bin 0
hist = chroma.mean(axis=1)
hist = hist / (hist.sum() + 1e-8) # pitch-class distribution
return np.concatenate([hist, chroma.std(axis=1)]).astype(np.float32) # 24-dim
def window_vectors(
y: np.ndarray,
sr: int = SAMPLE_RATE,
window_s: float = CLIP_SECONDS,
hop_s: float = CLIP_SECONDS,
max_windows: int | None = None,
tonic_hz: float | None = None,
) -> list[np.ndarray]:
"""Slide a `window_s` window over a (long) clip -> one frame vector per window.
Saraga tracks are full concert recordings, so one vector per track would be a
single over-smoothed sample. Windowing (D7: 10 s) turns each recording into many
training examples. Tail segments shorter than MIN_CLIP_SECONDS are dropped;
hop_s == window_s gives non-overlapping windows (the training default).
"""
win = int(round(window_s * sr))
hop = max(1, int(round(hop_s * sr)))
min_len = int(round(MIN_CLIP_SECONDS * sr))
if len(y) < min_len:
return []
if max_windows: # only look at the span we'll use (fast tonic estimate on long uploads)
y = y[: win + hop * (max_windows - 1)]
# Prefer a precise tonic (Saraga ctonic) when given; else the drone-argmax heuristic.
tonic_pc = tonic_pc_from_hz(tonic_hz) if tonic_hz else estimate_tonic_pc(y, sr)
out: list[np.ndarray] = []
start = 0
while start < len(y):
seg = y[start : start + win]
if len(seg) < min_len:
break
out.append(frame_vector(seg, sr, tonic_pc))
if max_windows and len(out) >= max_windows:
break
start += hop
return out
def pitch_windows(
times,
f0_hz,
tonic_hz: float,
window_s: float = CLIP_SECONDS,
hop_s: float = CLIP_SECONDS,
max_windows: int | None = None,
n_bins: int = 120,
min_voiced: float = MIN_VOICED_FRAC,
) -> list[np.ndarray]:
"""Slide a window over a predominant-melody pitch track -> one tonic-normalized PCD
per window (the pitch-track analog of window_vectors, D7). Windows that are voiced less
than `min_voiced` of the time are dropped (junk gate, D6/D8 — percussion/speech/silence);
a trailing window >= MIN_CLIP_SECONDS is kept. Works on any (pitch, tonic) source.
"""
times = np.asarray(times, dtype=float)
f0 = np.asarray(f0_hz, dtype=float)
if times.size == 0 or times[-1] < MIN_CLIP_SECONDS:
return []
end = float(times[-1])
out: list[np.ndarray] = []
start = 0.0
while start < end:
if min(start + window_s, end) - start < MIN_CLIP_SECONDS:
break
seg = f0[(times >= start) & (times < start + window_s)]
if _voiced_fraction(seg) >= min_voiced:
hist = pitch_class_histogram(seg, tonic_hz, n_bins)
if hist.sum() > 0:
out.append(hist)
if max_windows and len(out) >= max_windows:
break
start += hop_s
return out
def extract_windows(
path: str,
sr: int = SAMPLE_RATE,
max_windows: int | None = None,
tonic_hz: float | None = None,
) -> list[np.ndarray]:
"""Load a (long) recording and return one frame vector per 10 s window.
Pass tonic_hz (Saraga ctonic) for a precise tonic; otherwise it's estimated."""
duration = None if max_windows is None else max_windows * CLIP_SECONDS + 1.0
y = load_audio(path, sr, duration=duration)
return window_vectors(y, sr, max_windows=max_windows, tonic_hz=tonic_hz)