| FROM nvidia/cuda:12.9.0-runtime-ubuntu24.04 | |
| # Avoid interactive prompts during package installation | |
| ENV DEBIAN_FRONTEND=noninteractive | |
| # Install Python and system dependencies. | |
| # | |
| # NO ffmpeg, and NO torchcodec below: this pipeline only ever decodes .wav, which | |
| # libsndfile1 (via soundfile/librosa) handles on its own. ffmpeg was here solely to satisfy | |
| # torchcodec, which in turn was only needed to decode the dataset's Audio column — and the | |
| # `datasets<4.0` pin below removes that need. Dropping ffmpeg saves ~630 MB of apt packages and | |
| # most of this layer's build time (it was ~60 min of a recent Space build). | |
| # | |
| # THESE THREE GO TOGETHER — do not change one alone: | |
| # * unpin `datasets` (>=4.0) and its Audio column decodes via torchcodec, so torchcodec and | |
| # ffmpeg both have to come back; | |
| # * re-add torchcodec WITHOUT ffmpeg and `transformers.audio_utils.load_audio` gets worse, not | |
| # better: its default `backend="auto"` prefers torchcodec whenever it is importable, so it | |
| # would select a decoder whose FFmpeg shared libs are missing instead of falling back to | |
| # librosa. Stages 2 (transcribe) and 3 (SIM) both call it. | |
| RUN apt-get update && apt-get install -y --no-install-recommends \ | |
| python3 \ | |
| python3-pip \ | |
| python3-dev \ | |
| git \ | |
| libsndfile1 \ | |
| && rm -rf /var/lib/apt/lists/* | |
| # Set Python alias (Ubuntu 24.04 ships Python 3.12) | |
| RUN ln -sf /usr/bin/python3 /usr/bin/python | |
| # Allow pip to install packages system-wide in the container (PEP 668) | |
| ENV PIP_BREAK_SYSTEM_PACKAGES=1 | |
| # Set working directory | |
| WORKDIR /app | |
| # Install PyTorch ecosystem (cu128 wheels for CUDA 12.8+/12.9 compat) | |
| # torchvision is required by Qwen3-Omni's AutoProcessor, which wraps a video processor | |
| # (Qwen2VLVideoProcessor) even when only text/audio inputs are used. | |
| RUN pip install --no-cache-dir \ | |
| torch==2.8.0 \ | |
| torchaudio==2.8.0 \ | |
| torchvision==0.23.0 \ | |
| --index-url https://download.pytorch.org/whl/cu128 | |
| # Install common requirements (torch already installed above, pip will skip it) | |
| # regex + kaldialign are required by the Open ASR Leaderboard `normalizer/` used for WER scoring. | |
| # zhconv folds traditional Chinese to simplified for the CER scoring path (Seed-TTS zh splits). | |
| # `datasets` is pinned <4.0 to match the other backend images (chatterbox, xtts, pocket-tts): | |
| # 4.0 moved the Audio column's decoder to torchcodec, which pins exact torch minors and needs | |
| # FFmpeg shared libs. 3.x decodes with soundfile/librosa instead, which is all this pipeline needs | |
| # (.wav only) — and it keeps `prompt_audio["array"]`/`["sampling_rate"]` in run_eval.py reading a | |
| # plain dict rather than relying on the AudioDecoder compat shim 4.x/5.x ship. | |
| RUN pip install --no-cache-dir \ | |
| git+https://github.com/huggingface/transformers.git \ | |
| evaluate \ | |
| "datasets<4.0" \ | |
| librosa \ | |
| jiwer \ | |
| num2words \ | |
| regex \ | |
| kaldialign \ | |
| zhconv \ | |
| sentencepiece \ | |
| protobuf \ | |
| peft \ | |
| g2p-en \ | |
| soundfile | |
| # Download NLTK data required by g2p-en (used by FastSpeech2ConformerTokenizer). | |
| RUN python3 -c "import nltk; nltk.download('averaged_perceptron_tagger_eng'); nltk.download('cmudict')" | |
| # Copy the full repository | |
| COPY . /app | |
| # Default entrypoint | |
| ENTRYPOINT ["bash"] | |
| # Keep-alive CMD so the Space runtime stays healthy. HF Jobs and `docker run` | |
| # override this with their own command (e.g. `run_cohere.sh`). | |
| EXPOSE 7860 | |
| CMD ["-c", "python3 -m http.server 7860"] |