GLM-5.2-Vision-FP8 / truss /data /start_server.sh
harrypart's picture
truss deployment example
57e82be verified
Raw
History Blame Contribute Delete
3.68 kB
#!/bin/bash
# GLM-5.2-Vision on SGLang, Truss entrypoint.
#
# The model_cache repo is the FULLY ASSEMBLED checkpoint, so there is no assembly
# step: register the glm5v plugin, apply the SGLang arch registration, serve.
set -uo pipefail
echo "== glm-5.2-vision start: $(date) =="
# Set by the config's environment_variables; each variant mounts its own volume_folder.
CKPT=${GLM5V_CKPT:-/app/model_cache/glm5v}
# --- 0. block until the weights are actually on disk -------------------------
# This is a 465-756GB transfer and individual range requests DO fail transiently
# (HTTP/connection resets). truss-transfer-cli resumes, so retrying makes progress
# even when each attempt ends in an error. Never launch on a partial checkpoint:
# a missing shard surfaces much later as a confusing weight-loading crash.
expected_shards() {
python3 - "$CKPT/model.safetensors.index.json" <<'PY' 2>/dev/null || echo 0
import json, sys
idx = json.load(open(sys.argv[1]))
print(len({v for v in idx["weight_map"].values() if v.startswith("model-")}))
PY
}
for attempt in $(seq 1 40); do
truss-transfer-cli && echo "truss-transfer-cli OK (attempt $attempt)"
want=$(expected_shards)
have=$(ls "$CKPT"/model-*.safetensors 2>/dev/null | wc -l)
if [ "$want" -gt 0 ] && [ "$have" -ge "$want" ]; then
echo "checkpoint complete: $have/$want shards after $attempt attempt(s)"
break
fi
echo "attempt $attempt: $have/${want:-?} shards present; retrying in 15s..."
sleep 15
done
want=$(expected_shards); have=$(ls "$CKPT"/model-*.safetensors 2>/dev/null | wc -l)
if [ "$want" -gt 0 ] && [ "$have" -lt "$want" ]; then
echo "FATAL: checkpoint incomplete ($have/$want shards) — refusing to start"; exit 1
fi
echo "checkpoint entries: $(ls "$CKPT" | wc -l); shards: $have"
# --- 1. install the engine plugin, straight from the checkpoint --------------
# The plugin ships inside the model repo, so it arrives with the weights and there
# is nothing to vendor into this Truss. Tiny pure-python wheel, no dependencies.
pip install --no-deps -q "$CKPT/plugins" && echo "installed glm5v-serve from the checkpoint"
# These three env vars are SGLang's shipped external-package hooks — no in-tree edits.
export SGLANG_EXTERNAL_MODEL_PACKAGE=sglang_glm5v
export SGLANG_EXTERNAL_MM_PROCESSOR_PACKAGE=sglang_glm5v
export SGLANG_EXTERNAL_MM_MODEL_ARCH=Glm5vForConditionalGeneration
# --- 2. register the arch on SGLang's DSA/MLA paths --------------------------
# The one thing the external hooks above do not cover: SGLang gates its DSA
# sparse-attention path on a hardcoded architecture list.
python3 -m sglang_glm5v.patch || echo "(arch patch: continuing)"
# --- 3. serve ----------------------------------------------------------------
# --quantization is empty for the FP8 build (auto-detected from config.json) and
# `modelopt_fp4` for the NVFP4 build; NVFP4 additionally needs the two disable
# flags (see the release README for why).
QUANT_ARGS=""
if [ "${GLM5V_QUANTIZATION:-}" = "modelopt_fp4" ]; then
QUANT_ARGS="--quantization modelopt_fp4 --disable-shared-experts-fusion --disable-flashinfer-autotune"
fi
exec python3 -m sglang.launch_server \
--model-path "$CKPT" \
--trust-remote-code \
--tp-size "${GLM5V_TP_SIZE:-8}" \
$QUANT_ARGS \
--attention-backend "${GLM5V_ATTN_BACKEND:-dsa}" \
--mm-attention-backend sdpa \
--kv-cache-dtype fp8_e4m3 \
--page-size 64 \
--mem-fraction-static "${GLM5V_MEM_FRACTION:-0.85}" \
--context-length "${GLM5V_MAX_MODEL_LEN:-1048576}" \
--reasoning-parser "${GLM5V_REASONING_PARSER:-glm45}" \
--tool-call-parser "${GLM5V_TOOL_PARSER:-glm47}" \
--served-model-name glm-5.2-vision \
--host 0.0.0.0 --port 8000