FROM nvidia/cuda:12.9.0-runtime-ubuntu24.04 # Avoid interactive prompts during package installation ENV DEBIAN_FRONTEND=noninteractive # Make apt resilient to slow/stalled mirrors: fail a stuck download fast and retry a few # times instead of hanging for hours on a single package. RUN printf 'Acquire::Retries "5";\nAcquire::http::Timeout "30";\nAcquire::https::Timeout "30";\n' \ > /etc/apt/apt.conf.d/99-retries-timeout # Some build hosts can't reach security.ubuntu.com (its mirror IPs time out on :80), which # breaks apt on the krb5/security-pocket packages. archive.ubuntu.com carries the same # packages and is reachable, so redirect the security pocket to it. Handles both the # deb822 (.sources, default on 24.04) and legacy sources.list layouts. RUN sed -i 's|http://security.ubuntu.com/ubuntu|http://archive.ubuntu.com/ubuntu|g' \ /etc/apt/sources.list.d/ubuntu.sources 2>/dev/null || true; \ sed -i 's|http://security.ubuntu.com/ubuntu|http://archive.ubuntu.com/ubuntu|g' \ /etc/apt/sources.list 2>/dev/null || true # System deps: git(-lfs) to clone the repo + weights; portaudio19-dev + build-essential to # build pyaudio (a fish-speech dep); ffmpeg/libsndfile1 for audio I/O. RUN apt-get update && apt-get install -y --no-install-recommends \ python3 \ python3-pip \ python3-dev \ git \ git-lfs \ ffmpeg \ libsndfile1 \ portaudio19-dev \ build-essential \ && rm -rf /var/lib/apt/lists/* \ && git lfs install # Set Python alias (Ubuntu 24.04 ships Python 3.12) RUN ln -sf /usr/bin/python3 /usr/bin/python ENV PIP_BREAK_SYSTEM_PACKAGES=1 WORKDIR /app RUN apt-get update && apt-get install -y --no-install-recommends cargo \ && rm -rf /var/lib/apt/lists/* # Upgrade pip so it fetches prebuilt manylinux wheels (e.g. tokenizers) instead of trying to # build them from source — the base image's system pip is old enough to miss them, which would # otherwise require a Rust toolchain to compile `tokenizers`. RUN pip install --no-cache-dir --upgrade --ignore-installed pip setuptools wheel # Install PyTorch (cu128 wheels). fish-speech hard-pins torch==2.8.0 / torchaudio==2.8.0. RUN pip install --no-cache-dir \ torch==2.8.0 \ torchaudio==2.8.0 \ --index-url https://download.pytorch.org/whl/cu128 # Clone fish-speech (= S2 codebase) and install it editable. Provides the inference module # run_eval.py imports and the vendored configs / .project-root it resolves at runtime. RUN git clone https://github.com/fishaudio/fish-speech.git /opt/fish-speech \ && cd /opt/fish-speech && pip install --no-cache-dir -e . # datasets + tqdm for the eval loop (soundfile comes in via fish-speech deps). RUN pip install --no-cache-dir datasets tqdm # Download the S2-Pro weights (~11 GB: safetensors + codec.pth) into the checkpoint dir. # S2-Pro is ungated, so this needs no HF token. Baked in so the eval runs offline. RUN python3 -c "from huggingface_hub import snapshot_download; snapshot_download('fishaudio/s2-pro', local_dir='/opt/fish-speech/checkpoints/s2-pro')" # Copy the full repository COPY . /app # Default entrypoint ENTRYPOINT ["bash"] # Keep-alive CMD so the Space runtime stays healthy; `docker run` overrides it. EXPOSE 7860 CMD ["-c", "python3 -m http.server 7860"]