""" Agent 5 (Acoustics) - shared ASR/TTS/model-routing facade. The legacy role of this agent was phonetic profiling. It now also owns the shared acoustic contract used by PureKITchat, PurePolyglot, and PACYCx: ASR, TTS policy metadata, model listings, cache hints, and fallback policy. """ import asyncio import inspect import json import os from pathlib import Path from src.asr_language import resolve_asr_language_code class AgentAcoustic: def __init__(self, llm_manager=None, input_agent=None): self.llm_manager = llm_manager self.input_agent = input_agent self.cache_dir = Path( os.environ.get("ACOUSTIC_MODEL_CACHE_DIR") or os.environ.get("HF_HOME") or os.environ.get("TRANSFORMERS_CACHE") or (Path.home() / ".cache" / "huggingface") ) print("Agent 5 (Acoustic) Online: Shared ASR/TTS routing engine ready.") def attach_input_agent(self, input_agent): self.input_agent = input_agent def model_catalog(self): return { "asr": [ { "id": "auto", "label": "Auto (Smart route)", "description": "Smart default: tries the best available route, validates script/language, and falls back when needed.", "edge_ready": True, }, { "id": "whisper-large-v3-turbo", "label": "Whisper large-v3 turbo (Multilingual)", "description": "Fast multilingual HF Pro/ZeroGPU route for Korean, English, Arabic, Chinese, French, Spanish, and similar high-coverage languages.", "edge_ready": False, }, { "id": "facebook/mms-1b-all", "label": "MMS 1B (Low-resource)", "description": "Low-resource fallback for underrepresented languages and wrong-language/script retries, including Tagalog.", "edge_ready": False, }, { "id": "whisper-small", "label": "Whisper small (Fallback)", "description": "Reliable CPU fallback when HF GPU/ZeroGPU routes are unavailable or slow.", "edge_ready": True, }, { "id": "base", "label": "Whisper base (Edge balanced)", "description": "Balanced local CPU route for short game utterances.", "edge_ready": True, }, { "id": "tiny", "label": "Whisper tiny (Edge fastest)", "description": "Fast local route for English-only games and quick dialect interactions.", "edge_ready": True, }, { "id": "distil-whisper/distil-large-v3", "label": "Distil-Whisper (English fast)", "description": "Fast manual option for English-only and English-dialect interactions.", "edge_ready": False, }, { "id": "qwen/qwen3-asr-1.7b", "label": "Qwen3-ASR 1.7B (Multilingual)", "description": "Manual multilingual ASR experiment for broad-language stress tests when dependencies are available.", "edge_ready": False, }, ], "tts": [ { "id": "browser-native", "label": "Browser native TTS", "description": "Current shared policy fallback: frontends use local speechSynthesis voices while the backend owns language/voice routing metadata.", "edge_ready": True, } ], "fallback_policy": [ "Auto prefers Whisper large-v3 turbo on GPU/ZeroGPU for high-coverage languages.", "Underrepresented languages use Whisper first, then MMS validation when available.", "If HF GPU/ZeroGPU is unavailable or slow, the route falls back to local Whisper small/base/tiny.", "Manual game-level speech-model choices are honored before Auto routing.", ], } def cache_status(self): known_entries = [] for name in ("models--openai--whisper-large-v3-turbo", "models--facebook--mms-1b-all", "models--distil-whisper--distil-large-v3"): known_entries.append({ "id": name.replace("models--", "").replace("--", "/"), "cached": (self.cache_dir / "hub" / name).exists() or (self.cache_dir / name).exists(), }) return { "cache_dir": str(self.cache_dir), "cache_exists": self.cache_dir.exists(), "known_entries": known_entries, } def models(self): return { "ok": True, "service": "pure-acoustic-agent", "models": self.model_catalog(), "cache": self.cache_status(), } def transcribe(self, audio_path, language="", dialect="", speech_model="auto", audio_sanitation="on"): if not self.input_agent: return { "ok": False, "text": "", "error": "Input agent is not attached to the acoustic agent.", } if not audio_path: return {"ok": False, "text": "", "error": "No audio file provided."} hint = json.dumps({"language": language or "", "dialect": dialect or ""}) asr_code, source_language, source_dialect = resolve_asr_language_code(hint) speech_model_choice = str(speech_model or "auto").strip() or "auto" sanitation_choice = str(audio_sanitation or "on").strip().lower() or "on" print( "Acoustic ASR request: " f"language={source_language or 'auto'} dialect={source_dialect or 'auto'} " f"code={asr_code or 'auto'} speech_model={speech_model_choice} " f"clean_audio={sanitation_choice}" ) result = self.input_agent.transcribe( audio_path, language=asr_code, model_choice=speech_model_choice, dialect_hint=f"{source_language or ''} {source_dialect or ''} {language or ''} {dialect or ''}", sanitize_audio=sanitation_choice, ) item = result[0] if isinstance(result, list) and result else {} text = str(item.get("text", "") if isinstance(item, dict) else item).strip() return { "ok": bool(text), "text": text, "speaker": item.get("speaker", "Speaker 1") if isinstance(item, dict) else "Speaker 1", "model": item.get("model", "") if isinstance(item, dict) else "", "requested_model": speech_model_choice, "language": source_language or language or "", "dialect": source_dialect or dialect or "", "asr_code": asr_code or "", "audio_sanitation": "on" if sanitation_choice not in {"off", "false", "0", "no", "raw", "none"} else "off", "cache": self.cache_status(), } def tts(self, text, language="", dialect="", voice="browser-native"): profile = { "language": language or "", "dialect": dialect or "", "voice": voice or "browser-native", "engine": "browser-native", "status": "policy-only", "text": text or "", "audio_url": "", "message": "Use frontend speechSynthesis for now; server-side TTS can be plugged into this route without changing frontend contracts.", } return {"ok": bool(text), **profile} def _safe_generate(self, prompt): if not self.llm_manager: raise RuntimeError("LLM Manager not connected.") generator = getattr(self.llm_manager, "generate_smart", None) or getattr(self.llm_manager, "generate_fast", None) if not generator: raise RuntimeError("LLM Manager has no generation method.") response = generator(prompt) if inspect.isawaitable(response): loop = asyncio.new_event_loop() try: asyncio.set_event_loop(loop) response = loop.run_until_complete(response) finally: loop.close() asyncio.set_event_loop(None) return response def generate_phonetic_profile(self, text, dialect): """ Uses the AI engine to generate an IPA representation and intonation rules for the given text based on the specific dialect. """ if not self.llm_manager: return json.dumps({ "ipa": "/unavailable/", "intonation": "LLM Manager not connected. Cannot perform acoustic analysis." }) prompt = f""" You are an expert socio-linguist and phonetician. The user has spoken the following text in the following dialect/language: Text: "{text}" Dialect: "{dialect}" Please provide the following: 1. "ipa": The most accurate International Phonetic Alphabet (IPA) transcription of how this exact phrase would be pronounced in this specific dialect. 2. "intonation": A brief, 1-2 sentence description of the stress, rhythm, and intonation patterns typical for this phrase in this dialect. Output ONLY valid JSON in this exact format, with no markdown formatting or backticks: {{ "ipa": "/həˈloʊ/", "intonation": "Stress falls on the second syllable with a rising pitch at the end." }} """ try: response_obj = self._safe_generate(prompt) response_text = response_obj.text if hasattr(response_obj, "text") else str(response_obj) clean_res = response_text.replace("```json", "").replace("```", "").strip() parsed = json.loads(clean_res) return json.dumps({ "ipa": parsed.get("ipa", "Unknown"), "intonation": parsed.get("intonation", "Unknown") }) except Exception as e: print(f"Acoustic Agent Error: {e}") return json.dumps({ "ipa": f"/error parsing {dialect} phonetics/", "intonation": "Error occurred during generation." })