"""Second Ear — a realtime music production assistant.
Three surfaces over one analysis engine:
Live a rolling window of whatever the browser is hearing, metered
against genre targets, with findings that fire as they happen.
Bounce measurement-grade pass on a rendered file, plus the semantic
layer, the written critique, and an Ableton action plan.
Bridge how to wire the plan into a real Live set through MCP.
Every analysis endpoint is also an MCP tool, so an agent that already has
Ableton MCP connected can use this Space as its ears and its own Ableton
connection as its hands.
"""
from __future__ import annotations
import json
import gradio as gr
import numpy as np
import soundfile as sf
from ear import ableton, dsp, knowledge, llm, render, semantic
LIVE_WINDOW_S = 8.0 # what the meters describe
CAPTURE_S = 30.0 # how far back "analyse what just happened" reaches
TEMPO_EVERY = 8 # ticks between tempo/key refresh (they need context)
semantic.SEMANTIC.start()
# --------------------------------------------------------------------------
# shared helpers
# --------------------------------------------------------------------------
def _read(path: str) -> tuple[int, np.ndarray]:
data, sr = sf.read(path, dtype="float32", always_2d=True)
return sr, data
def _blank_state() -> dict:
return {"buf": np.zeros((0, 2), dtype=np.float32), "tick": 0,
"rhythm": {"bpm": 0.0, "confidence": 0.0, "onset_rate": 0.0},
"key": {"key": "—", "confidence": 0.0}}
# --------------------------------------------------------------------------
# live loop
# --------------------------------------------------------------------------
def live_tick(chunk, state, genre, source):
"""Fold one streamed chunk into the rolling window and re-meter."""
state = state or _blank_state()
if chunk is None:
return render.idle("waiting for signal…"), render.idle("no findings yet"), state
sr, data = chunk
incoming = dsp.to_float_stereo(sr, data)
if incoming.shape[0] == 0:
return gr.skip(), gr.skip(), state
buf = np.concatenate([state["buf"], incoming], axis=0)
keep = int(CAPTURE_S * dsp.SR)
if buf.shape[0] > keep:
buf = buf[-keep:]
state["buf"] = buf
state["tick"] += 1
window = buf[-int(LIVE_WINDOW_S * dsp.SR):]
if float(np.max(np.abs(window))) < 1e-4:
return render.idle("signal is silent — check the input device"), gr.skip(), state
rep = dsp.analyze(dsp.SR, window, fast=True)
if rep is None:
return gr.skip(), gr.skip(), state
# Tempo and key need a longer view than the meter window, and cost more,
# so they refresh on their own slower clock and are carried between ticks.
if state["tick"] % TEMPO_EVERY == 1 and buf.shape[0] > 12 * dsp.SR:
mono = buf[-int(16 * dsp.SR):].mean(axis=1)
flux, fps = dsp.onset_envelope(mono)
state["rhythm"] = dsp.tempo_from_onsets(flux, fps)
freqs, power = dsp.spectrum(mono)
state["key"] = dsp.key_estimate(freqs, power)
rep.rhythm, rep.key = state["rhythm"], state["key"]
diags = knowledge.diagnose(rep, genre, source)
verdict, tone = knowledge.headline_verdict(rep, diags)
held = min(buf.shape[0] / dsp.SR, CAPTURE_S)
return (
render.meters(rep, genre, verdict, tone,
extra=f"{LIVE_WINDOW_S:.0f}s window · {held:.0f}s held"),
render.cards(diags, limit=4),
state,
)
def analyse_capture(state, genre, source, intent, use_llm):
"""Run the full pass on whatever the live loop has been holding."""
state = state or _blank_state()
buf = state.get("buf")
if buf is None or buf.shape[0] < dsp.SR:
return (render.idle("nothing captured yet — start listening first"),
"Not enough audio held to analyse.", "{}")
return _full_pass(dsp.SR, buf, genre, source, intent, use_llm)
# --------------------------------------------------------------------------
# full pass (shared by the bounce tab and the API)
# --------------------------------------------------------------------------
def _full_pass(sr, data, genre, source, intent, use_llm):
rep = dsp.analyze(sr, data)
if rep is None:
return render.idle("clip too short"), "Clip too short to analyse.", "{}"
diags = knowledge.diagnose(rep, genre, source)
verdict, tone = knowledge.headline_verdict(rep, diags)
mono = dsp.to_float_stereo(sr, data).mean(axis=1)
tags = semantic.SEMANTIC.describe(mono)
tag_line = semantic.tags_line(tags)
plan = ableton.build_plan(diags, genre=genre, bpm=rep.rhythm.get("bpm", 0.0))
parts = [render.meters(rep, genre, verdict, tone,
extra=f"{rep.duration:.1f}s · {genre}")]
if tag_line:
parts.append(
f'{render.STYLE}
'
f'
sounds like
'
f'
{tag_line}
'
)
parts.append(f'{render.cards(diags)}
')
written = [f"### {verdict}", ""]
if tag_line:
written.append(f"*Sounds like: {tag_line}*\n")
if use_llm:
note = llm.critique(rep.to_dict(), [d.to_dict() for d in diags],
tags, genre, source, intent)
written += [note, "", "---", ""] if note else [f"*{llm.available()[1]}*", ""]
written += ["## Ableton action plan", "", ableton.plan_to_markdown(plan)]
export = {"report": rep.to_dict(),
"findings": [d.to_dict() for d in diags],
"sounds_like": {g: [t for t, _ in v] for g, v in tags.items()},
"ableton_plan": plan}
return "".join(parts), "\n".join(written), json.dumps(export, indent=2)
def analyse_file(audio_path: str, genre: str = "Dubstep / Riddim",
source: str = "Full mix / master", intent: str = "",
use_llm: bool = True):
"""Analyse a rendered audio file and return a full production report.
Args:
audio_path: path to the audio file to listen to (wav, mp3, flac, aiff).
genre: which target window to judge against, e.g. "Dubstep / Riddim".
source: what the audio is — "Full mix / master", "Drum bus", "Bass / 808",
"Lead / synth", "Vocal" or "Pad / atmosphere".
intent: optional free text describing what you were going for.
use_llm: include the written engineer's critique (needs HF_TOKEN on the Space).
"""
if not audio_path:
return render.idle("load a file first"), "No audio supplied.", "{}"
sr, data = _read(audio_path)
return _full_pass(sr, data, genre, source, intent, use_llm)
def measure(audio_path: str) -> str:
"""Measure an audio file and return the raw metrics as JSON.
Loudness (LUFS-I/S, LRA), true peak, crest factor, seven-band balance,
band ratios, stereo correlation and width, tempo and key. No opinions.
Args:
audio_path: path to the audio file to measure.
"""
if not audio_path:
return json.dumps({"error": "no audio supplied"})
sr, data = _read(audio_path)
rep = dsp.analyze(sr, data)
if rep is None:
return json.dumps({"error": "clip too short"})
return json.dumps(rep.to_dict(), indent=2)
def health() -> str:
"""Report which optional layers are live: the CLAP semantic ear and the
written critique. Returns JSON."""
return json.dumps({
"semantic": semantic.SEMANTIC.status(),
"semantic_ready": semantic.SEMANTIC.ready,
"critique": llm.available()[1],
}, indent=2)
def _status_line() -> str:
return (f"Semantic ear: {semantic.SEMANTIC.status()} · "
f"written critique: {llm.available()[1]}")
def ableton_plan(audio_path: str, genre: str = "Dubstep / Riddim",
source: str = "Full mix / master",
track_index: str = "$MASTER") -> str:
"""Return an Ableton Live action plan for an audio file, as MCP call JSON.
The plan is a sequence of Ableton MCP tool calls (load_instrument_or_effect,
get_device_parameters, set_device_parameter, …) that an agent with a local
Ableton MCP server connected can execute directly against a running set.
Args:
audio_path: path to the audio file to analyse.
genre: target window to judge against.
source: what the audio is (master, drum bus, bass, vocal, …).
track_index: which Live track the plan targets. "$MASTER" for the master.
"""
if not audio_path:
return json.dumps({"error": "no audio supplied"})
sr, data = _read(audio_path)
rep = dsp.analyze(sr, data)
if rep is None:
return json.dumps({"error": "clip too short"})
diags = knowledge.diagnose(rep, genre, source)
plan = ableton.build_plan(diags, track_index=track_index, genre=genre,
bpm=rep.rhythm.get("bpm", 0.0))
return json.dumps(plan, indent=2)
# --------------------------------------------------------------------------
# UI
# --------------------------------------------------------------------------
CSS = """
#col-container{max-width:1180px;margin:0 auto;}
.dark .gradio-container{color:var(--body-text-color);}
#se-title h1{font-size:30px;letter-spacing:-0.02em;margin-bottom:2px;}
"""
GENRE_CHOICES = list(knowledge.GENRES.keys())
BRIDGE = """
## Wiring it into Live
This Space listens and decides. It does not touch your set — nothing hosted on
someone else's machine can, and a tool that pretended otherwise would be lying
to you. The split is deliberate:
| | |
|---|---|
| **Second Ear** (this Space) | ears + judgement — measures, diagnoses, writes the plan |
| **Ableton MCP** (your machine) | hands — executes the plan against the live set |
### 1. Point your agent at both
Every endpoint here is exposed as an MCP tool. Add this Space alongside your
existing Ableton MCP server:
```json
{
"mcpServers": {
"second-ear": {
"command": "npx",
"args": ["mcp-remote", "https://apolithosstudios-second-ear.hf.space/gradio_api/mcp/sse"]
},
"ableton": { "command": "...your existing Ableton MCP entry..." }
}
}
```
Tools you get: `analyse_file`, `measure`, `ableton_plan`.
### 2. Ask for the loop
> "Bounce the drop, run it through second-ear as Dubstep / Riddim, then execute
> the plan on my master."
The agent calls `ableton_plan`, gets back Ableton MCP calls, and runs them.
Every parameter write is preceded by a `get_device_parameters` probe, because
Live's parameter names move between versions — the plan resolves names at
execution time instead of guessing.
### 3. Feeding it live audio
The Live tab listens to whatever the browser's input device is. To point it at
your master bus instead of the room:
1. Install a loopback driver — **BlackHole** (free) or **Loopback**.
2. In Live, set the output (or a dedicated send) to that device.
3. Pick it as the input when the browser asks for microphone permission.
One honest caveat: browsers apply echo cancellation, noise suppression and auto
gain to captured audio by default. That is fine for *direction* — balance
drifting, sub running hot, the drop losing punch — and it is not fine for
absolute numbers. **For measurement-grade LUFS and true peak, bounce a file and
use the Bounce tab.** The Live tab is the ear on your shoulder; the Bounce tab
is the meter.
"""
with gr.Blocks(title="Second Ear") as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(
"# Second Ear\n"
"A realtime production assistant that listens to what you're making, "
"tells you what's wrong in engineer's language, and hands your agent "
"an Ableton plan to fix it.",
elem_id="se-title",
)
with gr.Row():
genre = gr.Dropdown(GENRE_CHOICES, value=GENRE_CHOICES[0],
label="Target sound", scale=2)
source = gr.Dropdown(knowledge.SOURCES, value=knowledge.SOURCES[0],
label="Listening to", scale=2)
with gr.Tabs():
# ---------------------------------------------------------- live
with gr.Tab("Live"):
gr.Markdown(
"Route your master through a loopback device, start the input, "
"and leave it running. The meters show a rolling 8-second "
"window against the target you picked; findings fire as they "
"happen. The ghost block behind each bar is where that band "
"should sit for this genre."
)
live_in = gr.Audio(sources=["microphone"], streaming=True,
type="numpy", label="Studio input")
live_meters = gr.HTML(render.idle("waiting for signal…"))
live_notes = gr.HTML(render.idle("no findings yet"))
with gr.Row():
capture_btn = gr.Button("Analyse the last 30 seconds",
variant="primary", scale=2)
live_llm = gr.Checkbox(value=True, label="Written critique",
scale=1)
live_intent = gr.Textbox(
label="What were you going for? (optional)",
placeholder="heavier drop, needs to hit on a club rig",
lines=1,
)
cap_report = gr.HTML()
cap_text = gr.Markdown()
with gr.Accordion("Raw export (JSON)", open=False):
cap_json = gr.Code(language="json")
state = gr.State(_blank_state())
live_in.stream(
live_tick,
inputs=[live_in, state, genre, source],
outputs=[live_meters, live_notes, state],
stream_every=0.5,
show_progress="hidden",
concurrency_limit=None,
)
capture_btn.click(
analyse_capture,
inputs=[state, genre, source, live_intent, live_llm],
outputs=[cap_report, cap_text, cap_json],
api_name=False, # gr.State can't cross the MCP boundary
)
# -------------------------------------------------------- bounce
with gr.Tab("Bounce"):
gr.Markdown(
"Measurement-grade pass on a rendered file — real LUFS, real "
"true peak, the semantic layer, the written critique, and the "
"Ableton plan."
)
with gr.Row():
file_in = gr.Audio(sources=["upload", "microphone"],
type="filepath", label="Bounce")
with gr.Column():
file_intent = gr.Textbox(
label="What were you going for? (optional)",
placeholder="dark and heavy, has to survive a club system",
lines=2,
)
file_llm = gr.Checkbox(value=True, label="Written critique")
run_btn = gr.Button("Listen", variant="primary")
file_report = gr.HTML()
file_text = gr.Markdown()
with gr.Accordion("Raw export (JSON)", open=False):
file_json = gr.Code(language="json")
run_btn.click(
analyse_file,
inputs=[file_in, genre, source, file_intent, file_llm],
outputs=[file_report, file_text, file_json],
api_name="analyse_file",
)
with gr.Accordion("Numbers only (no opinions)", open=False):
meas_btn = gr.Button("Measure")
meas_out = gr.Code(language="json", label="Metrics")
meas_btn.click(measure, inputs=[file_in], outputs=meas_out,
api_name="measure")
# -------------------------------------------------------- bridge
with gr.Tab("Ableton bridge"):
gr.Markdown(BRIDGE)
with gr.Row():
plan_audio = gr.Audio(type="filepath", label="Bounce")
with gr.Column():
plan_track = gr.Textbox("$MASTER", label="Target track index")
plan_btn = gr.Button("Build the plan", variant="primary")
plan_out = gr.Code(language="json", label="Ableton MCP calls")
plan_btn.click(
ableton_plan,
inputs=[plan_audio, genre, source, plan_track],
outputs=plan_out,
api_name="ableton_plan",
)
# Rendered per page load, not at import — the CLAP ear finishes warming
# up well after the Blocks tree is built.
status = gr.Markdown(_status_line())
demo.load(_status_line, outputs=status, api_name=False)
gr.Button("health", visible=False).click(
health, outputs=gr.Textbox(visible=False), api_name="health")
if __name__ == "__main__":
# Gradio 6 moved theme and css off the Blocks constructor onto launch().
demo.queue(default_concurrency_limit=4).launch(
theme=gr.themes.Citrus(),
css=CSS,
mcp_server=True,
)