self-improving-agent / Dockerfile
Zeetay
perf: use CPU-only torch in Docker image (8.9GB -> 2.3GB)
ef5d512
Raw
History Blame Contribute Delete
1.25 kB
# Hugging Face Spaces (Docker SDK) image for the FastAPI backend.
# Spaces routes traffic to port 7860 by default.
FROM python:3.11-slim
# System libraries needed at runtime by torch (OpenMP via libgomp) and by
# chromadb/onnxruntime; ca-certificates for HTTPS model downloads.
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
libgomp1 \
ca-certificates \
&& rm -rf /var/lib/apt/lists/*
# Use the PyTorch backend only; never import TensorFlow/Keras in transformers.
ENV USE_TF=0 \
USE_TORCH=1 \
PYTHONUNBUFFERED=1 \
PIP_NO_CACHE_DIR=1
WORKDIR /app
# Install Python deps first so this layer caches across code changes.
# Install the CPU-only torch wheel up front: Spaces has no GPU, so this avoids
# pulling ~6 GB of unused NVIDIA CUDA libraries. sentence-transformers then sees
# torch is already satisfied and skips the default CUDA build.
COPY requirements.txt ./
RUN pip install --upgrade pip \
&& pip install torch --index-url https://download.pytorch.org/whl/cpu \
&& pip install -r requirements.txt
# Copy the rest of the project (the .dockerignore keeps secrets/db/web out).
COPY . .
EXPOSE 7860
CMD ["uvicorn", "api.main:app", "--host", "0.0.0.0", "--port", "7860"]