second-ear / ear /llm.py
sevahu97's picture
Upload folder using huggingface_hub
5459c43 verified
Raw
History Blame Contribute Delete
3.82 kB
"""Optional narration layer.
The rule engine already produces the diagnosis and the move. This turns that
into the way a good engineer would actually say it in the room, and catches
interactions between findings that per-rule logic cannot see.
Entirely optional: with no token configured the Space runs on the rule engine
alone and says so, rather than degrading silently.
"""
from __future__ import annotations
import json
import os
DEFAULT_MODEL = os.environ.get("EAR_LLM_MODEL", "Qwen/Qwen3-235B-A22B-Instruct-2507")
SYSTEM = """You are a senior sound designer and mix engineer sitting in on a session.
You have been handed measurements and a rule-based diagnosis of a piece of audio.
How you talk:
- Like an engineer in the room, not a manual. Short sentences. No hedging.
- Never repeat a number without saying what it means for the listener.
- Name the move, the device, and the value. "Cut 3 dB at 250" beats "consider EQ".
- If the measurements disagree with each other, say which one you trust and why.
- If nothing is wrong, say so in one line and talk about what to do next instead
of inventing problems.
- Never mention that you were given JSON or that you are an AI.
Structure your answer as exactly three short sections:
**What I'm hearing** β€” two or three sentences, the sound not the stats.
**The one thing to fix first** β€” a single highest-leverage move, with the reason.
**Then** β€” at most three more moves as a tight bulleted list.
"""
def _token() -> str | None:
return os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN")
def available() -> tuple[bool, str]:
if not _token():
return False, "no HF_TOKEN set β€” running on the rule engine only"
return True, f"ready ({DEFAULT_MODEL})"
def critique(
report: dict,
diagnoses: list[dict],
semantic_tags: dict,
genre: str,
source: str,
intent: str = "",
) -> str:
ok, _ = available()
if not ok:
return ""
from huggingface_hub import InferenceClient
payload = {
"genre": genre,
"listening_to": source,
"producer_intent": intent or "(not stated)",
"measurements": {
"lufs_integrated": round(report.get("lufs_i", -120), 1),
"true_peak_dbtp": round(report.get("true_peak", -120), 2),
"crest_db": round(report.get("crest", 0), 1),
"loudness_range_lu": round(report.get("lra", 0), 1),
"band_balance_db": {k: round(v, 1) for k, v in report.get("bands", {}).items()},
"ratios_db": {k: round(v, 1) for k, v in report.get("ratios", {}).items()},
"stereo": {k: round(v, 2) for k, v in report.get("stereo", {}).items()},
"tempo_bpm": round(report.get("rhythm", {}).get("bpm", 0), 1),
"key": report.get("key", {}).get("key", "β€”"),
},
"rule_findings": [
{"severity": d["severity"], "headline": d["headline"],
"evidence": d["evidence"], "suggested_move": d["move"]}
for d in diagnoses[:6]
],
"sounds_like": {g: [t for t, _ in v] for g, v in (semantic_tags or {}).items()},
}
try:
client = InferenceClient(api_key=_token(), provider="auto")
resp = client.chat_completion(
model=DEFAULT_MODEL,
messages=[
{"role": "system", "content": SYSTEM},
{"role": "user", "content": json.dumps(payload, ensure_ascii=False)},
],
max_tokens=700,
temperature=0.6,
)
return (resp.choices[0].message.content or "").strip()
except Exception as exc: # noqa: BLE001 - narration is optional by design
return (f"_Narration unavailable ({type(exc).__name__}: {exc}). "
f"The analysis above stands on its own._")