Spaces:
Sleeping
Sleeping
File size: 12,374 Bytes
c25f760 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 | """Feature extraction.
Modeling ladder (D16): tonic-normalized pitch-class/swara histogram + XGBoost floor
-> TDMS -> CNN on mel/CQT. The floor (D16 step 1) is a librosa chroma histogram rolled
so the tonic (Sa) sits at bin 0 (D5). The tonic comes from Saraga's ctonic annotation
when available (precise), else a drone-argmax estimate; essentia TonicIndianArtMusic is
the inference-time upgrade for unlabelled clips (Phase 2).
"""
from __future__ import annotations
import numpy as np
from .config import CLIP_SECONDS, MIN_CLIP_SECONDS, MIN_VOICED_FRAC, SAMPLE_RATE
def _voiced_fraction(seg) -> float:
"""Fraction of a window's predominant-melody frames that carry a pitch (f0>0). Low for
percussion/speech/applause/silence (the junk gate, D6/D8); high for real melody."""
seg = np.asarray(seg, dtype=float)
return float((seg > 0).mean()) if seg.size else 0.0
def load_audio(path: str, sr: int = SAMPLE_RATE, duration: float | None = None) -> np.ndarray:
"""Decode any format to mono float32 at `sr` (ffmpeg handles mp3/m4a).
`duration` caps the decode length — Saraga tracks run 20-30 min, so decoding the
whole file to use only the first N windows wastes most of the work.
"""
import librosa
y, _ = librosa.load(path, sr=sr, mono=True, duration=duration)
return y.astype(np.float32)
def estimate_tonic_pc(y: np.ndarray, sr: int = SAMPLE_RATE) -> int:
"""Cheap tonic (Sa) estimate: the most energetic pitch class over the clip.
Carnatic performances carry a continuous tanpura drone on Sa, so the summed
chromagram peaks at the tonic. This is the librosa-only stand-in for the precise
essentia `TonicIndianArtMusic` used in Phase 2 (D5).
"""
import librosa
chroma = librosa.feature.chroma_cqt(y=y, sr=sr)
return int(chroma.mean(axis=1).argmax())
def tonic_pc_from_hz(hz: float) -> int:
"""Map a tonic frequency (Hz) to its chroma pitch class (0-11). Used with a
precise tonic — Saraga's ctonic annotation, or essentia at inference."""
import librosa
return int(round(librosa.hz_to_midi(hz))) % 12
# 12 semitone positions relative to Sa (common Carnatic labels for the swara-sthanas).
SWARA_LABELS = ["S", "R1", "R2", "G2", "G3", "M1", "M2", "P", "D1", "D2", "N2", "N3"]
# The seven swaras every learner knows. For the "how to hear this raaga" panel we fold the 12
# chromatic positions down to these names (sa ri ga ma pa da ni) — the beginner's vocabulary,
# not the R1/R2/G2/G3 sthana subscripts. Each name groups its sthana variants.
SWARA7 = ["Sa", "Ri", "Ga", "Ma", "Pa", "Da", "Ni"]
_SWARA7_GROUPS = [[0], [1, 2], [3, 4], [5, 6], [7], [8, 9], [10, 11]] # indices into SWARA_LABELS
def to_swaras7(profile):
"""Fold a 12-position swara profile to the seven named swaras (SWARA7), summing each swara's
sthana variants. Returns (names, values). The friendly display unit for new learners."""
p = np.asarray(profile, dtype=float)
vals = np.array([p[g].sum() for g in _SWARA7_GROUPS])
return SWARA7, vals
def pcd_to_swaras(pcd) -> np.ndarray:
"""Fold a fine PCD (n_bins, a multiple of 12) to a 12-position swara profile
(one bin per semitone), normalized — the human-readable raaga fingerprint."""
a = np.asarray(pcd, dtype=float)
a = a.reshape(12, a.size // 12).sum(axis=1)
return a / (a.sum() + 1e-9)
def pitch_class_histogram(f0_hz, tonic_hz: float, n_bins: int = 120) -> np.ndarray:
"""Tonic-normalized pitch-class distribution (PCD) from a predominant-melody pitch
track — the classic raaga fingerprint (D16). Each voiced f0 becomes cents relative
to Sa, folded to one octave, binned and normalized; unvoiced frames (f0<=0) are
ignored. Works on any (pitch, tonic) source, so Saraga and IAMRRD pool cleanly.
Returns zeros when nothing is voiced.
"""
f0 = np.asarray(f0_hz, dtype=float)
voiced = f0[f0 > 0]
if voiced.size == 0:
return np.zeros(n_bins, dtype=np.float32)
cents = 1200.0 * np.log2(voiced / tonic_hz)
bins = np.mod((np.mod(cents, 1200.0) / (1200.0 / n_bins)).astype(int), n_bins)
hist = np.bincount(bins, minlength=n_bins).astype(np.float32)
return hist / hist.sum()
def tdms(times, f0_hz, tonic_hz: float, delay: float = 0.3, n_bins: int = 48) -> np.ndarray:
"""Time-Delayed Melody Surface (Gulati et al., ISMIR 2016) — the gamaka feature (refines
D16). A 2-D histogram of (pitch(t), pitch(t+delay)) over the tonic-normalized, octave-folded
melody: a held note sits on the diagonal, while gamaka/transitions smear off-diagonal in a
raaga-characteristic way. This keeps the *movement* the 1-D PCD discards — the information
that should separate allied raagas (Mōhanaṁ/Bilahari/Bēgaḍa) which share a scale.
Returns a flat (n_bins*n_bins,) surface normalized to sum 1, or zeros when too few voiced
pairs. `delay` is in seconds; the frame hop is inferred from `times`.
"""
times = np.asarray(times, dtype=float)
f0 = np.asarray(f0_hz, dtype=float)
zeros = np.zeros(n_bins * n_bins, dtype=np.float32)
if times.size < 2:
return zeros
voiced = f0 > 0
cents = np.zeros_like(f0)
cents[voiced] = np.mod(1200.0 * np.log2(f0[voiced] / tonic_hz), 1200.0)
bins = np.mod((cents / (1200.0 / n_bins)).astype(int), n_bins)
hop = float(np.median(np.diff(times)))
d = max(1, int(round(delay / hop)))
if d >= bins.size:
return zeros
a, b = bins[:-d], bins[d:]
both = voiced[:-d] & voiced[d:] # count a pair only if both endpoints are voiced
a, b = a[both], b[both]
if a.size == 0:
return zeros
surf = np.zeros((n_bins, n_bins), dtype=np.float64)
np.add.at(surf, (a, b), 1.0)
return (surf / surf.sum()).astype(np.float32).ravel()
def tdms_windows(
times,
f0_hz,
tonic_hz: float,
window_s: float = CLIP_SECONDS,
hop_s: float = CLIP_SECONDS,
max_windows: int | None = None,
delay: float = 0.3,
n_bins: int = 48,
min_voiced: float = MIN_VOICED_FRAC,
) -> list[np.ndarray]:
"""Slide a window over a predominant-melody pitch track -> one TDMS surface per window
(the delay-surface analog of pitch_windows, D28). Slices identically to pitch_windows so
surfaces aggregate exactly like PCD windows. Windows voiced less than `min_voiced` of the
time are dropped (junk gate, D6/D8 — percussion/speech/silence, no stable melody to track).
"""
times = np.asarray(times, dtype=float)
f0 = np.asarray(f0_hz, dtype=float)
if times.size == 0 or times[-1] < MIN_CLIP_SECONDS:
return []
end = float(times[-1])
out: list[np.ndarray] = []
start = 0.0
while start < end:
if min(start + window_s, end) - start < MIN_CLIP_SECONDS:
break
mask = (times >= start) & (times < start + window_s)
if _voiced_fraction(f0[mask]) >= min_voiced:
surf = tdms(times[mask], f0[mask], tonic_hz, delay=delay, n_bins=n_bins)
if surf.sum() > 0:
out.append(surf)
if max_windows and len(out) >= max_windows:
break
start += hop_s
return out
def model_windows(times, f0_hz, tonic_hz: float, max_windows: int | None = None,
hop_s: float | None = None) -> list[np.ndarray]:
"""THE production model feature (D28): windowed TDMS at the config's winning settings.
train / evaluate / calibrate / inference all call this one function, so the feature can
never drift between how the model is trained and how it's served. `hop_s` overrides only the
stride — inference passes a smaller hop so a short clip yields more windows to average (each
window is still classified independently, so overlap is safe and doesn't change the feature).
"""
from .config import TDMS_BINS, TDMS_DELAY, TDMS_HOP_S, TDMS_MAX_WINDOWS, TDMS_WINDOW_S
return tdms_windows(times, f0_hz, tonic_hz, window_s=TDMS_WINDOW_S, hop_s=hop_s or TDMS_HOP_S,
max_windows=max_windows or TDMS_MAX_WINDOWS, delay=TDMS_DELAY, n_bins=TDMS_BINS)
def frame_vector(y: np.ndarray, sr: int = SAMPLE_RATE, tonic_pc: int = 0) -> np.ndarray:
"""Tonic-relative pitch-class descriptor for one window (D16 step 1).
Raaga lives in the pitch classes *relative to Sa*, so we roll the chromagram so
the tonic pitch class -> bin 0, then summarize as a normalized 12-bin pitch-class
histogram (the raaga fingerprint) + per-bin spread. Timbre features (MFCC,
spectral shape) are deliberately EXCLUDED: they encode recording/instrument
identity and make the model memorize tracks instead of learning raaga.
"""
import librosa
if y.size == 0:
raise ValueError("empty audio")
chroma = librosa.feature.chroma_cqt(y=y, sr=sr)
chroma = np.roll(chroma, -tonic_pc, axis=0) # tonic -> bin 0
hist = chroma.mean(axis=1)
hist = hist / (hist.sum() + 1e-8) # pitch-class distribution
return np.concatenate([hist, chroma.std(axis=1)]).astype(np.float32) # 24-dim
def window_vectors(
y: np.ndarray,
sr: int = SAMPLE_RATE,
window_s: float = CLIP_SECONDS,
hop_s: float = CLIP_SECONDS,
max_windows: int | None = None,
tonic_hz: float | None = None,
) -> list[np.ndarray]:
"""Slide a `window_s` window over a (long) clip -> one frame vector per window.
Saraga tracks are full concert recordings, so one vector per track would be a
single over-smoothed sample. Windowing (D7: 10 s) turns each recording into many
training examples. Tail segments shorter than MIN_CLIP_SECONDS are dropped;
hop_s == window_s gives non-overlapping windows (the training default).
"""
win = int(round(window_s * sr))
hop = max(1, int(round(hop_s * sr)))
min_len = int(round(MIN_CLIP_SECONDS * sr))
if len(y) < min_len:
return []
if max_windows: # only look at the span we'll use (fast tonic estimate on long uploads)
y = y[: win + hop * (max_windows - 1)]
# Prefer a precise tonic (Saraga ctonic) when given; else the drone-argmax heuristic.
tonic_pc = tonic_pc_from_hz(tonic_hz) if tonic_hz else estimate_tonic_pc(y, sr)
out: list[np.ndarray] = []
start = 0
while start < len(y):
seg = y[start : start + win]
if len(seg) < min_len:
break
out.append(frame_vector(seg, sr, tonic_pc))
if max_windows and len(out) >= max_windows:
break
start += hop
return out
def pitch_windows(
times,
f0_hz,
tonic_hz: float,
window_s: float = CLIP_SECONDS,
hop_s: float = CLIP_SECONDS,
max_windows: int | None = None,
n_bins: int = 120,
min_voiced: float = MIN_VOICED_FRAC,
) -> list[np.ndarray]:
"""Slide a window over a predominant-melody pitch track -> one tonic-normalized PCD
per window (the pitch-track analog of window_vectors, D7). Windows that are voiced less
than `min_voiced` of the time are dropped (junk gate, D6/D8 — percussion/speech/silence);
a trailing window >= MIN_CLIP_SECONDS is kept. Works on any (pitch, tonic) source.
"""
times = np.asarray(times, dtype=float)
f0 = np.asarray(f0_hz, dtype=float)
if times.size == 0 or times[-1] < MIN_CLIP_SECONDS:
return []
end = float(times[-1])
out: list[np.ndarray] = []
start = 0.0
while start < end:
if min(start + window_s, end) - start < MIN_CLIP_SECONDS:
break
seg = f0[(times >= start) & (times < start + window_s)]
if _voiced_fraction(seg) >= min_voiced:
hist = pitch_class_histogram(seg, tonic_hz, n_bins)
if hist.sum() > 0:
out.append(hist)
if max_windows and len(out) >= max_windows:
break
start += hop_s
return out
def extract_windows(
path: str,
sr: int = SAMPLE_RATE,
max_windows: int | None = None,
tonic_hz: float | None = None,
) -> list[np.ndarray]:
"""Load a (long) recording and return one frame vector per 10 s window.
Pass tonic_hz (Saraga ctonic) for a precise tonic; otherwise it's estimated."""
duration = None if max_windows is None else max_windows * CLIP_SECONDS + 1.0
y = load_audio(path, sr, duration=duration)
return window_vectors(y, sr, max_windows=max_windows, tonic_hz=tonic_hz)
|