File size: 7,602 Bytes
8c1b9fe
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de23710
8c1b9fe
 
 
f78dd4f
 
 
 
 
8c1b9fe
 
f78dd4f
b67cacd
 
 
 
 
 
 
 
32f283e
 
 
b67cacd
32f283e
 
b67cacd
5a0feef
 
b67cacd
b4c429e
a406137
 
 
 
 
 
95a521a
 
 
3d0f1e8
b4c429e
39e5349
 
 
 
 
 
de23710
 
39e5349
de23710
 
 
39e5349
8c1b9fe
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
555a43b
 
 
 
 
 
8c1b9fe
 
 
 
 
 
b4c429e
3d0f1e8
 
b4c429e
b67cacd
5a0feef
b67cacd
 
32f283e
 
8c1b9fe
 
 
 
 
 
 
cc2baaf
8c1b9fe
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
# Single-container image for a Hugging Face Space (Docker SDK).
#
# Spaces expose exactly one public port, but Auralynq's default topology is
# 7 containers behind a Caddy proxy (see containers/ and compose.yml). This
# image instead runs the API (FastAPI/uvicorn, loopback-only) and the web UI
# (Next.js standalone) as two processes in one container; Next.js itself
# proxies /api/* to the API process (see web/next.config.js's rewrites()),
# so no separate reverse proxy is needed.
#
# Build context must be the repository root:
#   podman build -f deploy/huggingface/Dockerfile -t auralynq-space .

FROM docker.io/library/node:20-slim AS web-builder
WORKDIR /web
COPY web/package.json web/package-lock.json* ./
RUN npm install --no-audit --no-fund
COPY web/ ./
# next.config.js's rewrites() resolves at `next build` time (baked into
# .next/routes-manifest.json), so AURALYNQ_INTERNAL_API_URL must be set here,
# not just in the runtime stage below — otherwise the /api proxy is silently
# empty and every API request 404s through the web server.
ENV NEXT_PUBLIC_API_BASE=/api \
    NEXT_TELEMETRY_DISABLED=1 \
    AURALYNQ_INTERNAL_API_URL=http://127.0.0.1:8000
RUN npm run build

FROM docker.io/library/python:3.11-slim AS runtime
ENV PYTHONUNBUFFERED=1 \
    PYTHONDONTWRITEBYTECODE=1 \
    PIP_NO_CACHE_DIR=1 \
    PIP_DISABLE_PIP_VERSION_CHECK=1
WORKDIR /app

# poppler-utils: PDF page rendering for visual grounding. nodejs/npm: runtime
# only for the pre-built Next.js standalone server (no build tools needed
# here — the build already happened in the web-builder stage).
RUN apt-get update \
    && apt-get install -y --no-install-recommends curl poppler-utils nodejs npm ffmpeg libsndfile1 espeak-ng zstd \
    && rm -rf /var/lib/apt/lists/*

# ---- API (light extras only — no torch/GPU stack; see pyproject.toml) ----
# `ingest` = PDF/DOCX parsing + page rendering for visual grounding. `llm` =
# the lightweight openai/anthropic/cohere HTTP SDKs (no torch, no langchain) —
# openai is required for the `huggingface` provider (HF Inference Providers via
# the OpenAI-compatible router). `eval` is deliberately excluded (ragas drags in
# langchain/langgraph and ~doubles image size; only `make eval`/`bench` need it).
COPY pyproject.toml README.md ./
COPY auralynq ./auralynq
RUN pip install -e ".[ingest,llm]"

# ---- Voice ASR (free, torch-free): faster-whisper (CTranslate2) + soundfile. ----
# The full `voice` extra pulls silero-vad (→torch, ~800MB) + librosa; we skip
# those and disable VAD/diarization (VAD is only in the live-mic path, not the
# uploaded-file path the web voice query uses). The Whisper model is baked into
# an image layer so voice works instantly + offline at runtime (no per-cold-start
# download on the free CPU tier).
ENV HF_HOME=/app/hfcache
# faster-whisper (ASR, small) + Kokoro (TTS, spoken replies). Install the
# CPU-only torch wheel first so kokoro doesn't drag in the ~2GB CUDA build.
# Prebake both models into image layers → voice is instant + offline at runtime.
RUN pip install "faster-whisper>=1.0" "soundfile>=0.12" \
    && pip install --index-url https://download.pytorch.org/whl/cpu torch \
    && pip install "kokoro>=0.9" \
    && mkdir -p /app/hfcache \
    && python -c "from faster_whisper import WhisperModel; WhisperModel('base.en', device='cpu', compute_type='int8')" \
    && python -c "from kokoro import KPipeline; p=KPipeline(lang_code='a'); [c for _,_,c in p('warm up', voice='af_heart')]; [c for _,_,c in p('warm up', voice='am_michael')]"

# ---- Local generative LLM (free, in-container): llama-cpp-python + GGUF. ----
# Install a PREBUILT AVX2/FMA/F16C wheel (compiled in a matching python:3.11-slim
# and committed under wheels/). HF's build node can't finish a source compile in
# its time window, so we bring our own optimized binary — no toolchain in the
# image, no build timeout. Qwen2.5-3B-Instruct Q4_K_M (~2.1 GB) follows the
# "cite every claim with [n]" instruction far more reliably than 1.5B. Model
# baked in for offline runtime.
COPY wheels/llama_cpp_python-0.3.34-py3-none-linux_x86_64.whl /tmp/wheels/
RUN pip install --no-cache-dir /tmp/wheels/llama_cpp_python-0.3.34-py3-none-linux_x86_64.whl \
    && rm -rf /tmp/wheels \
    && python -c "from huggingface_hub import hf_hub_download; hf_hub_download('Qwen/Qwen2.5-3B-Instruct-GGUF', 'qwen2.5-3b-instruct-q4_k_m.gguf')"

# ---- Ollama (GPU path): self-contained CUDA runtime, auto-detects the T4. ----
# The provider is `auto`: ollama (GPU) is preferred, with the CPU GGUF above as
# a fallback if the GPU/ollama is ever unavailable. Model baked into an image
# layer at OLLAMA_MODELS so runtime is offline. `ollama serve` is started by
# entrypoint.sh.
ENV OLLAMA_MODELS=/app/ollama_models
RUN curl -fsSL https://github.com/ollama/ollama/releases/download/v0.32.1/ollama-linux-amd64.tar.zst -o /tmp/ollama.tar.zst \
    && tar --use-compress-program=unzstd -xf /tmp/ollama.tar.zst -C /usr && rm /tmp/ollama.tar.zst \
    && mkdir -p /app/ollama_models \
    && bash -c 'OLLAMA_HOST=127.0.0.1:11434 /usr/bin/ollama serve & SRV=$!; \
        for i in $(seq 1 30); do curl -fsS http://127.0.0.1:11434/api/tags >/dev/null 2>&1 && break; sleep 1; done; \
        /usr/bin/ollama pull qwen2.5:3b; kill "$SRV"'

COPY scripts ./scripts
COPY examples ./examples

# ---- Web (pre-built Next.js standalone output) ----
COPY --from=web-builder /web/.next/standalone ./web
COPY --from=web-builder /web/.next/static ./web/.next/static
COPY --from=web-builder /web/public ./web/public

COPY deploy/huggingface/entrypoint.sh /app/entrypoint.sh
RUN chmod +x /app/entrypoint.sh \
    && useradd -m -u 10001 auralynq \
    && mkdir -p /data/auralynq \
    && chown -R auralynq:auralynq /app /data

LABEL org.opencontainers.image.title="auralynq-space" \
      org.opencontainers.image.description="Auralynq single-container image for Hugging Face Spaces" \
      org.opencontainers.image.source="https://github.com/MHHamdan/Auralynq" \
      org.opencontainers.image.licenses="Apache-2.0"

USER auralynq

# Safe-by-default posture for a public Space — see env.example for the full
# list and docs/getting-started/huggingface-space.md for the design notes.
# Every one of these can be overridden via Space Variables/Secrets.
# HF_HUB_OFFLINE: ASR/TTS models are baked into the image above — never let a
# runtime model load block on (rate-limited, unauthenticated) hub HEAD calls.
# A stalled load during a relaunch is exactly what "workload not healthy after
# 30 min" looks like on the free tier.
ENV HF_HUB_OFFLINE=1 \
    AURALYNQ_HF_SPACE=true \
    AURALYNQ_DEMO_MODE=true \
    AURALYNQ_PUBLIC_DEMO=true \
    AURALYNQ_ALLOW_UPLOADS=false \
    AURALYNQ_DATA_DIR=/data/auralynq \
    AURALYNQ_VECTOR__BACKEND=memory \
    AURALYNQ_EMBEDDING__PROVIDER=hash \
    AURALYNQ_LLM__PROVIDER=slm \
    AURALYNQ_LLM__SLM_REPO=Qwen/Qwen2.5-3B-Instruct-GGUF \
    AURALYNQ_LLM__SLM_FILENAME=qwen2.5-3b-instruct-q4_k_m.gguf \
    AURALYNQ_LLM__MAX_TOKENS=512 \
    AURALYNQ_VOICE__ASR_PROVIDER=faster_whisper \
    AURALYNQ_VOICE__ASR_MODEL=base.en \
    AURALYNQ_VOICE__VAD=false \
    AURALYNQ_VOICE__DIARIZE=false \
    AURALYNQ_VOICE__TTS_PROVIDER=kokoro \
    AURALYNQ_VOICE__TTS_VOICE=af_heart \
    AURALYNQ_VISUAL__ENABLED=true \
    AURALYNQ_SERVE__API_KEY= \
    AURALYNQ_INTERNAL_API_URL=http://127.0.0.1:8000 \
    NEXT_PUBLIC_API_BASE=/api \
    PORT=7860

EXPOSE 7860
HEALTHCHECK --interval=15s --timeout=10s --retries=6 --start-period=240s \
    CMD curl -fsS "http://127.0.0.1:${PORT:-7860}/api/health" || exit 1

CMD ["/app/entrypoint.sh"]