Spaces:
Running
Running
File size: 9,950 Bytes
4e92959 760789c 4e92959 c874dc6 4e92959 1a05ceb 4e92959 760789c 4e92959 c874dc6 4e92959 760789c 1a05ceb 760789c 4e92959 760789c 4e92959 7a2e27d 4e92959 7a2e27d 4e92959 1a05ceb 760789c c874dc6 4e92959 f1e28fc 4e92959 7a2e27d 760789c 4e92959 760789c c874dc6 f1e28fc 4e92959 760789c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 | """One upload → the whole answer. No UI, no Gradio: importable and runnable
headless, which is how the end-to-end verification runs before deploy.
Order matters for RAM on a small box: the audio is loaded once at 48 kHz for
CLAP, the beat analysis reloads it at 22.05 kHz (librosa's own cache-free
path), and the CLAP tile cache is dropped as soon as the vectors exist.
**The split that makes saved analyses work.** `listen()` is everything that
needs the file: the CLAP vectors and the two scores that are properties of
the record alone (hook, downbeat alignment). `score()` is everything that
needs the corpus: which markets, which windows, which tags. `analyse_track`
is the two called in order.
Nothing in `score()` reads audio, so a track measured in August can be
re-measured against November's charts from its stored vectors. It is also the
only path to a verdict, so a re-score and a fresh upload cannot drift apart —
there is one implementation, not two.
"""
from __future__ import annotations
import numpy as np
import boundaries
import clap_embed
import hooks as hookmod
import livematch
import tags as tagmod
import worldmap
WINDOW_S = 30.0 # the official clip length a distributor asks for
HOP_S = clap_embed.GRID_S
def listen(path: str, embedder) -> dict:
"""Everything that needs the audio file, and nothing that needs a corpus.
The return value is what a saved analysis stores: the track vector, the
per-window vectors, and the two window scores that are read against the
record itself rather than against the world (hook, downbeat alignment).
Roughly 75 KB of float32 for a three-minute track.
"""
audio = clap_embed.load_audio(path)
duration = len(audio) / clap_embed.CLAP_SR
track_vec = embedder.embed_track(audio)
starts, wvecs = embedder.embed_windows(audio, WINDOW_S, HOP_S)
# Read the section map off the tiles the sweep just cached, BEFORE the
# cache is dropped — it is the cheapest signal in the app and it is gone
# a line later.
tile_starts, tile_vecs = embedder.tile_matrix()
embedder.reset_cache()
beat = boundaries.analyse(path)
wscores = boundaries.window_scores(beat, starts, WINDOW_S)
hook = hookmod.window_hooks(starts, beat.get("curves") or {},
tile_vecs, tile_starts, WINDOW_S)
# With a hook reading the ranking uses the downbeat term ALONE, because
# the hook score already carries repetition and energy. Passing the
# combined `boundary` term as well would count them twice under two names.
usable = bool(hook.get("method") not in (None, "", "none") and starts)
return {
"duration_s": round(duration, 1),
"track_vec": track_vec,
"window_starts": starts,
"window_vecs": wvecs,
"align": wscores["align"] if usable else wscores["boundary"],
# Where the phrase finishes. `end_align` nudges the ordering toward
# windows whose last bar line lands just before the 30-second mark;
# `natural_end` is that bar line, and is where the played clip is cut.
"end_align": wscores["end_align"],
"natural_end": wscores["natural_end"],
"on_downbeat": wscores["on_downbeat"],
"nearest_downbeat": wscores["nearest_downbeat"],
"hook": hook["hook"] if usable else None,
"hook_label": hook["label"] if usable else None,
"hook_parts": ({k: hook[k] for k in ("repeats", "voice", "lift")}
if usable else None),
"hook_method": hook.get("method"),
"voice_from": (beat.get("curves") or {}).get("voice_from"),
"hook_used": usable,
"beat": {"ok": beat.get("ok"), "tempo": beat.get("tempo"),
"error": beat.get("error")},
}
def score(heard: dict, corpus: livematch.Corpus, vocab=None,
top_markets: int = 8, focus_markets: int = 3,
snippet_top: int = 4) -> dict:
"""Everything that needs the corpus, and nothing that needs the audio.
`heard` is a `listen()` return value, either fresh or read back from a
saved analysis. This is the only place a verdict is decided, so a track
re-scored against a later week goes through exactly the code a fresh
upload does.
"""
track_vec = np.asarray(heard["track_vec"], dtype=np.float32)
starts = [float(s) for s in heard["window_starts"]]
wvecs = np.asarray(heard["window_vecs"], dtype=np.float32)
duration = heard["duration_s"]
scan = livematch.rank_regions(corpus, track_vec, top=5)
# Markets the map cannot draw carry their ISO code where a name belongs.
# Fixed here, at the one point every reading of a track passes through,
# so what is written into the archive is a name and not a code.
worldmap.name_regions(scan["regions"])
ranked_isos = sorted(scan["regions"], key=lambda k: -scan["regions"][k]["best"])
shown = ranked_isos[:top_markets]
focus = ranked_isos[:focus_markets]
per_market = {}
if starts and focus:
for iso in focus:
idx = np.asarray(corpus.regions[iso]["idx"], dtype=np.int64)
sims = wvecs @ corpus.emb[idx].T # (windows, pool)
best = sims.argmax(axis=1)
per_market[iso] = {
"best_sim": sims.max(axis=1),
"best_sound": [int(idx[b]) for b in best],
}
if per_market:
stack = np.stack([per_market[i]["best_sim"] for i in focus])
win_best_market = [focus[int(j)] for j in stack.argmax(axis=0)]
affinity = stack.max(axis=0).tolist()
else:
win_best_market = [None] * len(starts)
affinity = [0.0] * len(starts)
usable = bool(heard.get("hook_used"))
# One shortlist, strongest hook first, with the best trend fit appended
# if the hook order missed it. The two side-by-side sets it replaces put
# the same window on screen twice and made the reader understand how the
# list was built before the highlight meant anything.
picks = livematch.rank_shortlist(
affinity, starts, top=snippet_top,
hook=heard["hook"] if usable else None,
hook_label=heard["hook_label"] if usable else None,
end_align=heard.get("end_align"))
by_start = {round(s, 2): i for i, s in enumerate(starts)}
for n, p in enumerate(picks):
i = by_start[p["start_s"]]
iso = win_best_market[i]
p["clip_i"] = n
p["end_s"] = round(p["start_s"] + WINDOW_S, 2)
p["market"] = iso
p["market_name"] = (worldmap.market_name(
iso, corpus.regions[iso].get("name")) if iso else None)
p["on_downbeat"] = heard["on_downbeat"][i]
p["nearest_downbeat"] = heard["nearest_downbeat"][i]
# Where the last bar line inside the window falls. The DELIVERED clip
# is still exactly 30 seconds, because that is the field a
# distributor asks a label to fill; this is where the app trims the
# clip it plays, so what the team hears finishes its phrase.
nat = (heard.get("natural_end") or [None] * len(starts))[i]
p["natural_end_s"] = (round(float(nat), 2)
if nat is not None and nat == nat else None)
if usable and heard.get("hook_parts"):
p["hook_parts"] = {k: heard["hook_parts"][k][i]
for k in ("repeats", "voice", "lift")}
if iso:
s = corpus.sounds[per_market[iso]["best_sound"][i]]
p["nearest"] = {"artist": s["artist"], "title": s["title"],
"deezer_id": s.get("deezer_id")}
else:
p["nearest"] = None
# -- tags -------------------------------------------------------------
# nearest neighbours by index, straight from the focus pools
neighbour_idx: list[int] = []
if focus:
pool = livematch.pool_index(corpus, focus)
sims = corpus.emb[pool] @ track_vec
neighbour_idx = [int(pool[i]) for i in np.argsort(-sims)[:10]]
trend = tagmod.trend_tags(scan["regions"], shown, corpus.sounds, neighbour_idx)
model_tags = vocab.top(track_vec) if vocab is not None else None
fields = tagmod.copy_fields(model_tags, trend)
return {
"duration_s": round(duration, 1),
"week": scan["week"],
"shown_markets": shown,
"focus_markets": focus,
"regions": scan["regions"],
"track_vec": track_vec,
"snippets": picks, # union; clip_i indexes into this
# `snippet_sets` is gone. One list now; a saved analysis from before
# the change still carries the key and `render` still reads it, so an
# older saved report reopens as it was written.
"window_count": len(starts),
"beat": dict(heard["beat"]),
"hook": {"method": heard.get("hook_method"),
"voice_from": heard.get("voice_from"),
"used": bool(usable)},
"tags": {"model": model_tags, "trend": trend, "fields": fields},
"corpus": {"week": corpus.week, "sounds": len(corpus.sounds),
"regions": len(corpus.regions),
"generated_at": corpus.meta.get("generated_at"),
"sources": corpus.meta.get("sources", []),
"genre_labelled": corpus.meta.get("genre_labelled_sounds", 0)},
}
def analyse_track(path: str, embedder, corpus: livematch.Corpus,
vocab=None, top_markets: int = 8, focus_markets: int = 3,
snippet_top: int = 4) -> dict:
"""The original one-call entry point, kept byte-identical in behaviour.
`listen` then `score`. Callers that want to save the analysis for later
want the `listen` half too, so they call the two themselves.
"""
return score(listen(path, embedder), corpus, vocab=vocab,
top_markets=top_markets, focus_markets=focus_markets,
snippet_top=snippet_top)
|