Buckets:
| # GLM-5.2-Vision on SGLang, Truss entrypoint. | |
| # | |
| # The model_cache repo is the FULLY ASSEMBLED checkpoint, so there is no assembly | |
| # step: register the glm5v plugin, apply the SGLang arch registration, serve. | |
| set -uo pipefail | |
| echo "== glm-5.2-vision start: $(date) ==" | |
| # Set by the config's environment_variables; each variant mounts its own volume_folder. | |
| CKPT=${GLM5V_CKPT:-/app/model_cache/glm5v} | |
| # --- 0. block until the weights are actually on disk ------------------------- | |
| # This is a 465-756GB transfer and individual range requests DO fail transiently | |
| # (HTTP/connection resets). truss-transfer-cli resumes, so retrying makes progress | |
| # even when each attempt ends in an error. Never launch on a partial checkpoint: | |
| # a missing shard surfaces much later as a confusing weight-loading crash. | |
| expected_shards() { | |
| python3 - "$CKPT/model.safetensors.index.json" <<'PY' 2>/dev/null || echo 0 | |
| import json, sys | |
| idx = json.load(open(sys.argv[1])) | |
| print(len({v for v in idx["weight_map"].values() if v.startswith("model-")})) | |
| PY | |
| } | |
| for attempt in $(seq 1 40); do | |
| truss-transfer-cli && echo "truss-transfer-cli OK (attempt $attempt)" | |
| want=$(expected_shards) | |
| have=$(ls "$CKPT"/model-*.safetensors 2>/dev/null | wc -l) | |
| if [ "$want" -gt 0 ] && [ "$have" -ge "$want" ]; then | |
| echo "checkpoint complete: $have/$want shards after $attempt attempt(s)" | |
| break | |
| fi | |
| echo "attempt $attempt: $have/${want:-?} shards present; retrying in 15s..." | |
| sleep 15 | |
| done | |
| want=$(expected_shards); have=$(ls "$CKPT"/model-*.safetensors 2>/dev/null | wc -l) | |
| if [ "$want" -gt 0 ] && [ "$have" -lt "$want" ]; then | |
| echo "FATAL: checkpoint incomplete ($have/$want shards) — refusing to start"; exit 1 | |
| fi | |
| echo "checkpoint entries: $(ls "$CKPT" | wc -l); shards: $have" | |
| # --- 1. install the engine plugin, straight from the checkpoint -------------- | |
| # The plugin ships inside the model repo, so it arrives with the weights and there | |
| # is nothing to vendor into this Truss. Tiny pure-python wheel, no dependencies. | |
| pip install --no-deps -q "$CKPT/plugins" && echo "installed glm5v-serve from the checkpoint" | |
| # These three env vars are SGLang's shipped external-package hooks — no in-tree edits. | |
| export SGLANG_EXTERNAL_MODEL_PACKAGE=sglang_glm5v | |
| export SGLANG_EXTERNAL_MM_PROCESSOR_PACKAGE=sglang_glm5v | |
| export SGLANG_EXTERNAL_MM_MODEL_ARCH=Glm5vForConditionalGeneration | |
| # --- 2. register the arch on SGLang's DSA/MLA paths -------------------------- | |
| # The one thing the external hooks above do not cover: SGLang gates its DSA | |
| # sparse-attention path on a hardcoded architecture list. | |
| python3 -m sglang_glm5v.patch || echo "(arch patch: continuing)" | |
| # --- 3. serve ---------------------------------------------------------------- | |
| # --quantization is empty for the FP8 build (auto-detected from config.json) and | |
| # `modelopt_fp4` for the NVFP4 build; NVFP4 additionally needs the two disable | |
| # flags (see the release README for why). | |
| QUANT_ARGS="" | |
| if [ "${GLM5V_QUANTIZATION:-}" = "modelopt_fp4" ]; then | |
| QUANT_ARGS="--quantization modelopt_fp4 --disable-shared-experts-fusion --disable-flashinfer-autotune" | |
| fi | |
| exec python3 -m sglang.launch_server \ | |
| --model-path "$CKPT" \ | |
| --trust-remote-code \ | |
| --tp-size "${GLM5V_TP_SIZE:-8}" \ | |
| $QUANT_ARGS \ | |
| --attention-backend "${GLM5V_ATTN_BACKEND:-dsa}" \ | |
| --mm-attention-backend sdpa \ | |
| --kv-cache-dtype fp8_e4m3 \ | |
| --page-size 64 \ | |
| --mem-fraction-static "${GLM5V_MEM_FRACTION:-0.85}" \ | |
| --context-length "${GLM5V_MAX_MODEL_LEN:-1048576}" \ | |
| --reasoning-parser "${GLM5V_REASONING_PARSER:-glm45}" \ | |
| --tool-call-parser "${GLM5V_TOOL_PARSER:-glm47}" \ | |
| --served-model-name glm-5.2-vision \ | |
| --host 0.0.0.0 --port 8000 | |
Xet Storage Details
- Size:
- 3.68 kB
- Xet hash:
- d6d9b9dccdcd6f4bac715c1a8862c0e122eab79da52f7d08ff24ad94a755d586
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.