Spaces:
Running
Running
| """ | |
| Agent 5 (Acoustics) - shared ASR/TTS/model-routing facade. | |
| The legacy role of this agent was phonetic profiling. It now also owns the | |
| shared acoustic contract used by PureKITchat, PurePolyglot, and PACYCx: | |
| ASR, TTS policy metadata, model listings, cache hints, and fallback policy. | |
| """ | |
| import asyncio | |
| import inspect | |
| import json | |
| import os | |
| from pathlib import Path | |
| from src.asr_language import resolve_asr_language_code | |
| class AgentAcoustic: | |
| def __init__(self, llm_manager=None, input_agent=None): | |
| self.llm_manager = llm_manager | |
| self.input_agent = input_agent | |
| self.cache_dir = Path( | |
| os.environ.get("ACOUSTIC_MODEL_CACHE_DIR") | |
| or os.environ.get("HF_HOME") | |
| or os.environ.get("TRANSFORMERS_CACHE") | |
| or (Path.home() / ".cache" / "huggingface") | |
| ) | |
| print("Agent 5 (Acoustic) Online: Shared ASR/TTS routing engine ready.") | |
| def attach_input_agent(self, input_agent): | |
| self.input_agent = input_agent | |
| def model_catalog(self): | |
| return { | |
| "asr": [ | |
| { | |
| "id": "auto", | |
| "label": "Auto (Smart route)", | |
| "description": "Smart default: tries the best available route, validates script/language, and falls back when needed.", | |
| "edge_ready": True, | |
| }, | |
| { | |
| "id": "whisper-large-v3-turbo", | |
| "label": "Whisper large-v3 turbo (Multilingual)", | |
| "description": "Fast multilingual HF Pro/ZeroGPU route for Korean, English, Arabic, Chinese, French, Spanish, and similar high-coverage languages.", | |
| "edge_ready": False, | |
| }, | |
| { | |
| "id": "facebook/mms-1b-all", | |
| "label": "MMS 1B (Low-resource)", | |
| "description": "Low-resource fallback for underrepresented languages and wrong-language/script retries, including Tagalog.", | |
| "edge_ready": False, | |
| }, | |
| { | |
| "id": "whisper-small", | |
| "label": "Whisper small (Fallback)", | |
| "description": "Reliable CPU fallback when HF GPU/ZeroGPU routes are unavailable or slow.", | |
| "edge_ready": True, | |
| }, | |
| { | |
| "id": "base", | |
| "label": "Whisper base (Edge balanced)", | |
| "description": "Balanced local CPU route for short game utterances.", | |
| "edge_ready": True, | |
| }, | |
| { | |
| "id": "tiny", | |
| "label": "Whisper tiny (Edge fastest)", | |
| "description": "Fast local route for English-only games and quick dialect interactions.", | |
| "edge_ready": True, | |
| }, | |
| { | |
| "id": "distil-whisper/distil-large-v3", | |
| "label": "Distil-Whisper (English fast)", | |
| "description": "Fast manual option for English-only and English-dialect interactions.", | |
| "edge_ready": False, | |
| }, | |
| { | |
| "id": "qwen/qwen3-asr-1.7b", | |
| "label": "Qwen3-ASR 1.7B (Multilingual)", | |
| "description": "Manual multilingual ASR experiment for broad-language stress tests when dependencies are available.", | |
| "edge_ready": False, | |
| }, | |
| ], | |
| "tts": [ | |
| { | |
| "id": "browser-native", | |
| "label": "Browser native TTS", | |
| "description": "Current shared policy fallback: frontends use local speechSynthesis voices while the backend owns language/voice routing metadata.", | |
| "edge_ready": True, | |
| } | |
| ], | |
| "fallback_policy": [ | |
| "Auto prefers Whisper large-v3 turbo on GPU/ZeroGPU for high-coverage languages.", | |
| "Underrepresented languages use Whisper first, then MMS validation when available.", | |
| "If HF GPU/ZeroGPU is unavailable or slow, the route falls back to local Whisper small/base/tiny.", | |
| "Manual game-level speech-model choices are honored before Auto routing.", | |
| ], | |
| } | |
| def cache_status(self): | |
| known_entries = [] | |
| for name in ("models--openai--whisper-large-v3-turbo", "models--facebook--mms-1b-all", "models--distil-whisper--distil-large-v3"): | |
| known_entries.append({ | |
| "id": name.replace("models--", "").replace("--", "/"), | |
| "cached": (self.cache_dir / "hub" / name).exists() or (self.cache_dir / name).exists(), | |
| }) | |
| return { | |
| "cache_dir": str(self.cache_dir), | |
| "cache_exists": self.cache_dir.exists(), | |
| "known_entries": known_entries, | |
| } | |
| def models(self): | |
| return { | |
| "ok": True, | |
| "service": "pure-acoustic-agent", | |
| "models": self.model_catalog(), | |
| "cache": self.cache_status(), | |
| } | |
| def transcribe(self, audio_path, language="", dialect="", speech_model="auto", audio_sanitation="on"): | |
| if not self.input_agent: | |
| return { | |
| "ok": False, | |
| "text": "", | |
| "error": "Input agent is not attached to the acoustic agent.", | |
| } | |
| if not audio_path: | |
| return {"ok": False, "text": "", "error": "No audio file provided."} | |
| hint = json.dumps({"language": language or "", "dialect": dialect or ""}) | |
| asr_code, source_language, source_dialect = resolve_asr_language_code(hint) | |
| speech_model_choice = str(speech_model or "auto").strip() or "auto" | |
| sanitation_choice = str(audio_sanitation or "on").strip().lower() or "on" | |
| print( | |
| "Acoustic ASR request: " | |
| f"language={source_language or 'auto'} dialect={source_dialect or 'auto'} " | |
| f"code={asr_code or 'auto'} speech_model={speech_model_choice} " | |
| f"clean_audio={sanitation_choice}" | |
| ) | |
| result = self.input_agent.transcribe( | |
| audio_path, | |
| language=asr_code, | |
| model_choice=speech_model_choice, | |
| dialect_hint=f"{source_language or ''} {source_dialect or ''} {language or ''} {dialect or ''}", | |
| sanitize_audio=sanitation_choice, | |
| ) | |
| item = result[0] if isinstance(result, list) and result else {} | |
| text = str(item.get("text", "") if isinstance(item, dict) else item).strip() | |
| return { | |
| "ok": bool(text), | |
| "text": text, | |
| "speaker": item.get("speaker", "Speaker 1") if isinstance(item, dict) else "Speaker 1", | |
| "model": item.get("model", "") if isinstance(item, dict) else "", | |
| "requested_model": speech_model_choice, | |
| "language": source_language or language or "", | |
| "dialect": source_dialect or dialect or "", | |
| "asr_code": asr_code or "", | |
| "audio_sanitation": "on" if sanitation_choice not in {"off", "false", "0", "no", "raw", "none"} else "off", | |
| "cache": self.cache_status(), | |
| } | |
| def tts(self, text, language="", dialect="", voice="browser-native"): | |
| profile = { | |
| "language": language or "", | |
| "dialect": dialect or "", | |
| "voice": voice or "browser-native", | |
| "engine": "browser-native", | |
| "status": "policy-only", | |
| "text": text or "", | |
| "audio_url": "", | |
| "message": "Use frontend speechSynthesis for now; server-side TTS can be plugged into this route without changing frontend contracts.", | |
| } | |
| return {"ok": bool(text), **profile} | |
| def _safe_generate(self, prompt): | |
| if not self.llm_manager: | |
| raise RuntimeError("LLM Manager not connected.") | |
| generator = getattr(self.llm_manager, "generate_smart", None) or getattr(self.llm_manager, "generate_fast", None) | |
| if not generator: | |
| raise RuntimeError("LLM Manager has no generation method.") | |
| response = generator(prompt) | |
| if inspect.isawaitable(response): | |
| loop = asyncio.new_event_loop() | |
| try: | |
| asyncio.set_event_loop(loop) | |
| response = loop.run_until_complete(response) | |
| finally: | |
| loop.close() | |
| asyncio.set_event_loop(None) | |
| return response | |
| def generate_phonetic_profile(self, text, dialect): | |
| """ | |
| Uses the AI engine to generate an IPA representation and intonation rules | |
| for the given text based on the specific dialect. | |
| """ | |
| if not self.llm_manager: | |
| return json.dumps({ | |
| "ipa": "/unavailable/", | |
| "intonation": "LLM Manager not connected. Cannot perform acoustic analysis." | |
| }) | |
| prompt = f""" | |
| You are an expert socio-linguist and phonetician. | |
| The user has spoken the following text in the following dialect/language: | |
| Text: "{text}" | |
| Dialect: "{dialect}" | |
| Please provide the following: | |
| 1. "ipa": The most accurate International Phonetic Alphabet (IPA) transcription of how this exact phrase would be pronounced in this specific dialect. | |
| 2. "intonation": A brief, 1-2 sentence description of the stress, rhythm, and intonation patterns typical for this phrase in this dialect. | |
| Output ONLY valid JSON in this exact format, with no markdown formatting or backticks: | |
| {{ | |
| "ipa": "/həˈloʊ/", | |
| "intonation": "Stress falls on the second syllable with a rising pitch at the end." | |
| }} | |
| """ | |
| try: | |
| response_obj = self._safe_generate(prompt) | |
| response_text = response_obj.text if hasattr(response_obj, "text") else str(response_obj) | |
| clean_res = response_text.replace("```json", "").replace("```", "").strip() | |
| parsed = json.loads(clean_res) | |
| return json.dumps({ | |
| "ipa": parsed.get("ipa", "Unknown"), | |
| "intonation": parsed.get("intonation", "Unknown") | |
| }) | |
| except Exception as e: | |
| print(f"Acoustic Agent Error: {e}") | |
| return json.dumps({ | |
| "ipa": f"/error parsing {dialect} phonetics/", | |
| "intonation": "Error occurred during generation." | |
| }) | |