tommytracx commited on
Commit
5928492
·
verified ·
1 Parent(s): e15b3f8

Restore Q4_K_M after fidelity failure

Browse files

Exact source: ttracx/thoxroute@21f69dda9b31838ffa69de7c61de9565d7fc6957

Files changed (3) hide show
  1. Dockerfile +2 -2
  2. README.md +1 -2
  3. app.py +1 -2
Dockerfile CHANGED
@@ -22,7 +22,7 @@ ENV HF_HOME=/home/mambauser/.cache/huggingface \
22
  PORT=7860 \
23
  THOX_FAST_MODEL_REPO=Qwen/Qwen2.5-0.5B-Instruct-GGUF \
24
  THOX_FAST_MODEL_REVISION=9217f5db79a29953eb74d5343926648285ec7e67 \
25
- THOX_FAST_MODEL_FILE=qwen2.5-0.5b-instruct-q4_0.gguf \
26
  THOX_FAST_POOL_SIZE=2 \
27
  THOX_FAST_QUEUE_TIMEOUT_S=6 \
28
  THOX_FAST_MAX_OUTPUT_TOKENS=16 \
@@ -31,7 +31,7 @@ ENV HF_HOME=/home/mambauser/.cache/huggingface \
31
 
32
  # Bake the immutable public interactive model into the image. A RUNNING Space
33
  # therefore cannot report ready while still waiting on a first-request download.
34
- RUN python -c "from huggingface_hub import hf_hub_download; hf_hub_download('Qwen/Qwen2.5-0.5B-Instruct-GGUF', 'qwen2.5-0.5b-instruct-q4_0.gguf', revision='9217f5db79a29953eb74d5343926648285ec7e67')"
35
 
36
  WORKDIR /app
37
  COPY --chown=$MAMBA_USER:$MAMBA_USER app.py /app/app.py
 
22
  PORT=7860 \
23
  THOX_FAST_MODEL_REPO=Qwen/Qwen2.5-0.5B-Instruct-GGUF \
24
  THOX_FAST_MODEL_REVISION=9217f5db79a29953eb74d5343926648285ec7e67 \
25
+ THOX_FAST_MODEL_FILE=qwen2.5-0.5b-instruct-q4_k_m.gguf \
26
  THOX_FAST_POOL_SIZE=2 \
27
  THOX_FAST_QUEUE_TIMEOUT_S=6 \
28
  THOX_FAST_MAX_OUTPUT_TOKENS=16 \
 
31
 
32
  # Bake the immutable public interactive model into the image. A RUNNING Space
33
  # therefore cannot report ready while still waiting on a first-request download.
34
+ RUN python -c "from huggingface_hub import hf_hub_download; hf_hub_download('Qwen/Qwen2.5-0.5B-Instruct-GGUF', 'qwen2.5-0.5b-instruct-q4_k_m.gguf', revision='9217f5db79a29953eb74d5343926648285ec7e67')"
35
 
36
  WORKDIR /app
37
  COPY --chown=$MAMBA_USER:$MAMBA_USER app.py /app/app.py
README.md CHANGED
@@ -25,5 +25,4 @@ work; two-second SSE comments keep the stream live during synchronous prompt
25
  evaluation. The interactive provider itself caps every request at 16 output
26
  tokens, so malformed or older callers cannot leave minutes of abandoned work.
27
  The interactive model is pinned to immutable model revision
28
- `9217f5db79a29953eb74d5343926648285ec7e67`; its simpler Q4_0 artifact is
29
- baked into the image to minimize sustained CPU prompt-evaluation latency.
 
25
  evaluation. The interactive provider itself caps every request at 16 output
26
  tokens, so malformed or older callers cannot leave minutes of abandoned work.
27
  The interactive model is pinned to immutable model revision
28
+ `9217f5db79a29953eb74d5343926648285ec7e67` and baked into the image.
 
app.py CHANGED
@@ -32,7 +32,7 @@ FAST_MODEL_REVISION = os.environ.get(
32
  "THOX_FAST_MODEL_REVISION", "9217f5db79a29953eb74d5343926648285ec7e67"
33
  )
34
  FAST_MODEL_FILE = os.environ.get(
35
- "THOX_FAST_MODEL_FILE", "qwen2.5-0.5b-instruct-q4_0.gguf"
36
  )
37
  CODER_MODEL_ID = "thox-rust-coder"
38
  CODER_MODEL_REPO = os.environ.get(
@@ -244,7 +244,6 @@ def healthz() -> dict[str, Any]:
244
  "interactive_model": FAST_MODEL_ID,
245
  "interactive_model_repo": FAST_MODEL_REPO,
246
  "interactive_model_revision": FAST_MODEL_REVISION,
247
- "interactive_model_file": FAST_MODEL_FILE,
248
  "interactive_pool_size": FAST_POOL_SIZE,
249
  "interactive_available": active.interactive_available,
250
  "specialist_model": CODER_MODEL_ID,
 
32
  "THOX_FAST_MODEL_REVISION", "9217f5db79a29953eb74d5343926648285ec7e67"
33
  )
34
  FAST_MODEL_FILE = os.environ.get(
35
+ "THOX_FAST_MODEL_FILE", "qwen2.5-0.5b-instruct-q4_k_m.gguf"
36
  )
37
  CODER_MODEL_ID = "thox-rust-coder"
38
  CODER_MODEL_REPO = os.environ.get(
 
244
  "interactive_model": FAST_MODEL_ID,
245
  "interactive_model_repo": FAST_MODEL_REPO,
246
  "interactive_model_revision": FAST_MODEL_REVISION,
 
247
  "interactive_pool_size": FAST_POOL_SIZE,
248
  "interactive_available": active.interactive_available,
249
  "specialist_model": CODER_MODEL_ID,