# ============================================================================ # ACTIVE: llama-server (Qwen3-0.6B Q8_0), OpenAI-compatible, port 7860 # ============================================================================ FROM ghcr.io/ggml-org/llama.cpp:server # pre-build: bake the official Q8_0 GGUF into the image (610 MB) RUN mkdir -p /models && \ curl -fL --retry 3 -o /models/Qwen3-0.6B-Q8_0.gguf \ "https://huggingface.co/Qwen/Qwen3-0.6B-GGUF/resolve/main/Qwen3-0.6B-Q8_0.gguf" && \ ls -lh /models # key lives in .env (llama-server reads the LLAMA_API_KEY env var natively) COPY .env /app/.env EXPOSE 7860 HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=6 \ CMD curl -fsS http://127.0.0.1:7860/health || exit 1 # source .env unless the key is already set (Space secret / --env-file win) ENTRYPOINT ["/bin/sh", "-c", "if [ -z \"$LLAMA_API_KEY\" ]; then set -a; . /app/.env; set +a; fi; exec /app/llama-server \"$@\"", "--"] # -c is TOTAL across slots, so 32768 / 2 slots = 16384 ctx per request. # Raise -c to 65536 if you want a full 32k per concurrent request. CMD ["-m", "/models/Qwen3-0.6B-Q8_0.gguf", \ "--host", "0.0.0.0", \ "--port", "7860", \ "-c", "32768", \ "-np", "2", \ "-t", "2", \ "-tb", "4", \ "-ctk", "q8_0", \ "-ctv", "q8_0", \ "--jinja", \ "--alias", "Qwen3-0.6B"] # ============================================================================ # COMMENTED OUT: docqa — Document Data Extraction API (Python + FastAPI) # # Kept for later, not built. To restore it, delete the FROM/ENTRYPOINT/CMD above # and uncomment the block below. Its own README is in docqa/. # ============================================================================ # # FROM python:3.11-slim-bookworm # # # poppler-utils -> pdftotext/pdftoppm for the PDF text layer and rasterising # # scans; tesseract-ocr -> OCR for images and scanned pages. # RUN apt-get update && apt-get install -y --no-install-recommends \ # poppler-utils \ # tesseract-ocr \ # tesseract-ocr-eng \ # curl \ # libgomp1 \ # && rm -rf /var/lib/apt/lists/* # # WORKDIR /app # # COPY requirements.txt ./ # RUN pip install --no-cache-dir -r requirements.txt # # COPY docqa ./docqa # ENV PYTHONPATH=/app/docqa # # # The model is pulled from the Hub at first start and cached under # # /root/.cache/huggingface. Pre-fetching during build means a cold Space does # # not serve requests until the weights are on disk. # ARG MODEL_ID=impira/layoutlm-invoices # RUN python -c "\ # from huggingface_hub import snapshot_download; \ # print(snapshot_download('${MODEL_ID}', \ # allow_patterns=['*.json','*.txt','*.bin','*.safetensors','*.model'])); \ # " || echo 'prefetch failed; model will download at first start' # # ENV DOCX_MODEL_ID=impira/layoutlm-invoices \ # DOCX_DEVICE=cpu \ # DOCX_TORCH_THREADS=4 \ # DOCX_BATCH_SIZE=8 \ # DOCX_MAX_PAGES=3 \ # DOCX_MAX_UPLOAD_MB=25 \ # DOCX_CONFIDENCE_THRESHOLD=0.5 \ # DOCX_REQUEST_TIMEOUT_S=60 \ # DOCX_LOG_JSON=false \ # HF_HOME=/app/.cache/huggingface \ # PORT=7860 # # EXPOSE 7860 # # # Warm the model first, then serve. The model loads before uvicorn binds, so a # # passing health check means inference will actually work rather than the # # process merely being alive. # HEALTHCHECK --interval=30s --timeout=10s --start-period=600s --retries=6 \ # CMD curl -fsS http://127.0.0.1:7860/health || exit 1 # # CMD ["sh", "-c", "python -c \"from docxextract.engine import get_engine; from docxextract.config import get_settings; s=get_settings(); get_engine(s).warmup(); print('model warm')\" && exec uvicorn docxextract.api:app --host 0.0.0.0 --port ${PORT} --workers 1 --timeout-keep-alive 65"]