# Pocket TTS gets its own image: `pocket-tts` pins numpy>=2 and pulls a server stack # (fastapi/uvicorn/typer) plus sentencepiece, and it is the only backend here that needs no # torchaudio at all — its audio I/O is stdlib `wave` + soundfile. FROM nvidia/cuda:12.9.0-runtime-ubuntu24.04 # Avoid interactive prompts during package installation ENV DEBIAN_FRONTEND=noninteractive # System deps: libsndfile1 for soundfile I/O (writing generated wavs + decoding the dataset's # prompt_audio column); ffmpeg for librosa/soundfile fallback decoders; git for pip VCS installs. RUN apt-get update && apt-get install -y --no-install-recommends \ python3 \ python3-pip \ python3-dev \ git \ ffmpeg \ libsndfile1 \ && rm -rf /var/lib/apt/lists/* # Set Python alias (Ubuntu 24.04 ships Python 3.12; pocket-tts requires >=3.10,<3.15) RUN ln -sf /usr/bin/python3 /usr/bin/python # Allow pip to install packages system-wide in the container (PEP 668) ENV PIP_BREAK_SYSTEM_PACKAGES=1 WORKDIR /app # Upgrade pip so it prefers prebuilt manylinux wheels. --ignore-installed is required on # Ubuntu 24.04: the Debian-installed pip has no RECORD file and cannot be uninstalled. RUN pip install --no-cache-dir --upgrade --ignore-installed pip setuptools wheel # Install PyTorch first (cu128 wheels for CUDA 12.8+/12.9 compat) so `pip install pocket-tts` # does not pull a CPU wheel for its `torch>=2.5.0` requirement. NOTE: no torchaudio — pocket_tts # reads wavs with the stdlib `wave` module and other formats with soundfile. RUN pip install --no-cache-dir \ torch==2.8.0 \ --index-url https://download.pytorch.org/whl/cu128 # pocket-tts, version-pinned for reproducibility: the effective sampling temperature is a # per-version detail (2.1.0 hard-defaults temp=0.7 in load_model's signature; upstream main moves # it to a per-config `default_temperature` of 0.3 for English), and the package type-checks its own # arguments at runtime with beartype, so a signature change is a hard failure rather than a # silent one. The `[audio]` extra adds soundfile, needed for non-wav voice prompts. RUN pip install --no-cache-dir "pocket-tts[audio]==2.1.0" RUN pip install --no-cache-dir "datasets<4.0" tqdm soundfile librosa # NOTE: the weights are deliberately NOT baked into this image, unlike the other backends. # kyutai/pocket-tts is a GATED repo (auto-approve), and this image is published as a PUBLIC Space # for HF Jobs to pull — baking gated weights into it would redistribute them outside the gate. # They are fetched at runtime instead, so the job needs HF_TOKEN from an account that has accepted # the terms at https://huggingface.co/kyutai/pocket-tts (submit_jobs.sh passes --secrets HF_TOKEN). # Downloaded per run: the gated `languages//model.safetensors`, plus the tokenizer and any # predefined-voice states, which live in the ungated kyutai/pocket-tts-without-voice-cloning repo. # Copy the full repository COPY . /app # Default entrypoint ENTRYPOINT ["bash"] # Keep-alive CMD so the Space runtime stays healthy. HF Jobs and `docker run` # override this with their own command (e.g. run_eval.sh). EXPOSE 7860 CMD ["-c", "python3 -m http.server 7860"]