Download serve.sh from Maincode/Matilda-K3: direct link, hf CLI and curl.
- Browser
- Download file 4.38 kB
-
https://huggingface.co/Maincode/Matilda-K3/resolve/main/serve.sh
- Command line
-
hf download hf://Maincode/Matilda-K3/serve.sh
-
curl -L -o serve.sh https://huggingface.co/Maincode/Matilda-K3/resolve/main/serve.sh
4.38 kB
| # Start the Matilda-K3 server (OpenAI-compatible API) on one 8-GPU node. | |
| # | |
| # Usage: | |
| # MODEL_DIR=/path/to/Matilda-K3 ./serve.sh # start in the background | |
| # MODEL_DIR=/path/to/Matilda-K3 ./serve.sh --wait # start and block until ready | |
| # ./serve.sh --stop # stop and remove the container | |
| # | |
| # Environment: | |
| # MODEL_DIR directory holding this repository's files (default: script directory) | |
| # IMAGE runtime image (default: docker.io/maincodehq/matilda-vllm:kimi-k3) | |
| # NAME container name (default: matilda-k3) | |
| # PORT API port (default: 8000) | |
| # TP number of GPUs, tensor parallel (default: 8) | |
| # CACHE_DIR kernel build cache, keep it between starts (default: ~/matilda-cache) | |
| # KERNEL_CFG_DIR kernel tuning cache, keep it between starts (default: ~/matilda-kernel-cfg) | |
| # MATILDA_VERBOSE 1 streams the engine log to the container log (default: 0) | |
| # MAX_MODEL_LEN, MAX_NUM_SEQS, MAX_NUM_BATCHED_TOKENS, GPU_MEMORY_UTILIZATION | |
| # optional engine settings (runtime defaults: 262144, 64, 16384, 0.88) | |
| # RUNTIME podman or docker (default: whichever is installed, podman first) | |
| set -euo pipefail | |
| HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" | |
| IMAGE="${IMAGE:-docker.io/maincodehq/matilda-vllm:kimi-k3}" | |
| NAME="${NAME:-matilda-k3}" | |
| PORT="${PORT:-8000}" | |
| TP="${TP:-8}" | |
| MODEL_DIR="${MODEL_DIR:-$HERE}" | |
| CACHE_DIR="${CACHE_DIR:-$HOME/matilda-cache}" | |
| KERNEL_CFG_DIR="${KERNEL_CFG_DIR:-$HOME/matilda-kernel-cfg}" | |
| if [ -z "${RUNTIME:-}" ]; then | |
| if command -v podman >/dev/null 2>&1; then RUNTIME=podman | |
| elif command -v docker >/dev/null 2>&1; then RUNTIME=docker | |
| else echo "serve.sh: neither podman nor docker found" >&2; exit 2; fi | |
| fi | |
| if [ "${1:-}" = "--stop" ]; then | |
| "$RUNTIME" stop "$NAME" >/dev/null 2>&1 || true | |
| "$RUNTIME" rm "$NAME" >/dev/null 2>&1 || true | |
| echo "stopped $NAME" | |
| exit 0 | |
| fi | |
| [ -f "$MODEL_DIR/config.json" ] || { echo "serve.sh: MODEL_DIR=$MODEL_DIR has no config.json" >&2; exit 2; } | |
| ls "$MODEL_DIR"/model-*.safetensors >/dev/null 2>&1 || { echo "serve.sh: no model shards in $MODEL_DIR" >&2; exit 2; } | |
| [ -f "$MODEL_DIR/matilda-release.json" ] && [ -f "$MODEL_DIR/runtime/adapters.safetensors" ] || { | |
| echo "serve.sh: $MODEL_DIR is missing matilda-release.json or runtime/adapters.safetensors" >&2; exit 2; } | |
| [ -e /dev/kfd ] && [ -d /dev/dri ] || { echo "serve.sh: /dev/kfd or /dev/dri missing (AMD GPU driver not loaded?)" >&2; exit 2; } | |
| MODEL_DIR="$(cd "$MODEL_DIR" && pwd -P)" | |
| if "$RUNTIME" container inspect "$NAME" >/dev/null 2>&1; then | |
| echo "serve.sh: a container named $NAME already exists; run '$0 --stop' first" >&2 | |
| exit 2 | |
| fi | |
| mkdir -p "$CACHE_DIR" "$KERNEL_CFG_DIR" | |
| "$RUNTIME" image inspect "$IMAGE" >/dev/null 2>&1 || "$RUNTIME" pull "$IMAGE" | |
| # podman needs keep-groups for GPU device access when rootless, and the k8s-file | |
| # log driver so that "podman logs" shows the server output. | |
| EXTRA_ARGS=() | |
| [ "$RUNTIME" = podman ] && EXTRA_ARGS=(--group-add keep-groups --log-driver k8s-file) | |
| # Optional engine settings, passed through only when set. | |
| TUNING=() | |
| for v in MAX_MODEL_LEN MAX_NUM_SEQS MAX_NUM_BATCHED_TOKENS GPU_MEMORY_UTILIZATION; do | |
| [ -n "${!v:-}" ] && TUNING+=(-e "$v=${!v}") | |
| done | |
| "$RUNTIME" run -d --name "$NAME" \ | |
| --device=/dev/kfd --device=/dev/dri "${EXTRA_ARGS[@]}" \ | |
| --network=host --ipc=host --security-opt seccomp=unconfined --ulimit memlock=-1 \ | |
| -e PORT="$PORT" -e TP="$TP" -e MATILDA_VERBOSE="${MATILDA_VERBOSE:-0}" "${TUNING[@]}" \ | |
| -v "$MODEL_DIR":/models/Matilda-V3:ro \ | |
| -v "$CACHE_DIR":/root/.cache \ | |
| -v "$KERNEL_CFG_DIR":/tmp/aiter_configs \ | |
| "$IMAGE" >/dev/null | |
| echo "started $NAME from $IMAGE" | |
| echo "logs: $RUNTIME logs -f $NAME" | |
| echo "API: http://localhost:$PORT/v1 (first start compiles GPU kernels: 20 to 40 minutes)" | |
| if [ "${1:-}" = "--wait" ]; then | |
| echo "waiting for http://localhost:$PORT/health ..." | |
| while :; do | |
| if curl -sf -m 3 "http://localhost:$PORT/health" >/dev/null 2>&1; then echo "ready"; exit 0; fi | |
| state="$("$RUNTIME" inspect -f '{{.State.Status}}' "$NAME" 2>/dev/null || echo gone)" | |
| if [ "$state" != running ]; then | |
| echo "serve.sh: container is $state; last log lines:" >&2 | |
| "$RUNTIME" logs --tail 20 "$NAME" >&2 || true | |
| exit 1 | |
| fi | |
| sleep 10 | |
| done | |
| fi | |