Spaces:
Sleeping
Sleeping
usertea
EchoScript : two fully independent pipelines, chosen explicitly, never both from one click : Transcript & Translations vs. Phonetic Transcription (IPA)
9b67bb3 | """Phonetic transcription: Audio -> IPA phones, independent of language. | |
| This is deliberately NOT derived from the Transcript. Every other service | |
| in this codebase follows "Audio -> Transcript -> everything else" (see | |
| models/transcript.py's module docstring) -- phonetics is the one | |
| intentional exception, because the whole point is a representation of | |
| the raw acoustics that owes nothing to any language's orthography or to | |
| Whisper's language-modeled guess at "what words were probably said". | |
| Whisper's transcript is produced by a model that's biased toward | |
| producing valid words in some language -- it fills in gaps using | |
| linguistic context. What's implemented here is a *phone recognizer*: | |
| a model that outputs the IPA symbols for the sounds it hears, using a | |
| universal (language-agnostic) phone inventory, with no dictionary, no | |
| grammar, and no language identity involved at all. Two speakers of | |
| different languages making the same mouth sounds get the same IPA | |
| output from this service; they would NOT get the same Whisper transcript. | |
| Backend: Allosaurus (https://github.com/xinjli/allosaurus), a universal | |
| phone recognizer trained across ~2000 languages specifically to avoid | |
| being biased toward any single language's phoneme set. Its default | |
| inference mode (lang_id="ipa") is exactly this: no target-language | |
| assumption at all. | |
| Note: only .wav is accepted by Allosaurus directly. mp3/m4a/flac inputs | |
| are transcoded to a temporary wav first (see _ensure_wav below) -- | |
| this is a format conversion, not a re-interpretation of content, so it | |
| doesn't violate the "independent of language" property. | |
| """ | |
| from __future__ import annotations | |
| import subprocess | |
| import tempfile | |
| from functools import lru_cache | |
| from pathlib import Path | |
| from typing import Optional | |
| class PhoneticsError(RuntimeError): | |
| """Raised when phonetic transcription can't be produced for this audio.""" | |
| def _get_recognizer(): | |
| """Load (and cache) the Allosaurus universal phone recognizer. | |
| Cached at module level deliberately: this is a large model with no | |
| per-request state (unlike TranslationService's API key), so it's safe | |
| and desirable to load it once and reuse it across every call in the | |
| process, the same way TranscriptionService's Whisper model is reused. | |
| """ | |
| try: | |
| from allosaurus.app import read_recognizer | |
| except ImportError as exc: | |
| raise PhoneticsError( | |
| "The 'allosaurus' package is not installed. Add it to " | |
| "requirements.txt and reinstall to enable phonetic transcription." | |
| ) from exc | |
| try: | |
| return read_recognizer() | |
| except Exception as exc: | |
| raise PhoneticsError( | |
| f"Failed to load the Allosaurus phone recognizer: {exc}" | |
| ) from exc | |
| def _ensure_wav(audio_path: str) -> tuple[str, Optional[Path]]: | |
| """Return a path Allosaurus can read, converting to wav if necessary. | |
| Returns (wav_path, temp_dir_to_clean_up_or_None). Allosaurus only | |
| accepts .wav files; this is a lossless-in-content format conversion | |
| (resample/remux), not a transcription step, so it has no bearing on | |
| the language-independence of the result. | |
| """ | |
| if audio_path.lower().endswith(".wav"): | |
| return audio_path, None | |
| tmp_dir = Path(tempfile.mkdtemp(prefix="echoscript_phon_")) | |
| wav_path = tmp_dir / "audio.wav" | |
| result = subprocess.run( | |
| ["ffmpeg", "-y", "-i", str(audio_path), "-ar", "16000", "-ac", "1", str(wav_path)], | |
| capture_output=True, | |
| text=True, | |
| ) | |
| if result.returncode != 0: | |
| raise PhoneticsError(f"ffmpeg failed to prepare audio for phonetic analysis: {result.stderr.strip()}") | |
| return str(wav_path), tmp_dir | |
| def transcribe_phonetics(audio_path: str) -> str: | |
| """Return the IPA phone sequence for this audio file, start to finish. | |
| Operates directly on the (already time-windowed, if applicable) audio | |
| file -- never on a Transcript. Uses Allosaurus's universal 'ipa' mode, | |
| which makes no assumption about what language is being spoken. | |
| """ | |
| wav_path, cleanup_dir = _ensure_wav(audio_path) | |
| try: | |
| recognizer = _get_recognizer() | |
| try: | |
| phones = recognizer.recognize(wav_path, lang_id="ipa") | |
| except Exception as exc: | |
| raise PhoneticsError(f"Allosaurus failed to process the audio: {exc}") from exc | |
| return phones.strip() | |
| finally: | |
| if cleanup_dir is not None: | |
| import shutil | |
| shutil.rmtree(cleanup_dir, ignore_errors=True) | |