#!/usr/bin/env bash # Start the Matilda-K3 server (OpenAI-compatible API) on one 8-GPU node. # # Usage: # MODEL_DIR=/path/to/Matilda-K3 ./serve.sh # start in the background # MODEL_DIR=/path/to/Matilda-K3 ./serve.sh --wait # start and block until ready # ./serve.sh --stop # stop and remove the container # # Environment: # MODEL_DIR directory holding this repository's files (default: script directory) # IMAGE runtime image (default: docker.io/maincodehq/matilda-vllm:kimi-k3) # NAME container name (default: matilda-k3) # PORT API port (default: 8000) # TP number of GPUs, tensor parallel (default: 8) # CACHE_DIR kernel build cache, keep it between starts (default: ~/matilda-cache) # KERNEL_CFG_DIR kernel tuning cache, keep it between starts (default: ~/matilda-kernel-cfg) # MATILDA_VERBOSE 1 streams the engine log to the container log (default: 0) # MAX_MODEL_LEN, MAX_NUM_SEQS, MAX_NUM_BATCHED_TOKENS, GPU_MEMORY_UTILIZATION # optional engine settings (runtime defaults: 262144, 64, 16384, 0.88) # RUNTIME podman or docker (default: whichever is installed, podman first) set -euo pipefail HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" IMAGE="${IMAGE:-docker.io/maincodehq/matilda-vllm:kimi-k3}" NAME="${NAME:-matilda-k3}" PORT="${PORT:-8000}" TP="${TP:-8}" MODEL_DIR="${MODEL_DIR:-$HERE}" CACHE_DIR="${CACHE_DIR:-$HOME/matilda-cache}" KERNEL_CFG_DIR="${KERNEL_CFG_DIR:-$HOME/matilda-kernel-cfg}" if [ -z "${RUNTIME:-}" ]; then if command -v podman >/dev/null 2>&1; then RUNTIME=podman elif command -v docker >/dev/null 2>&1; then RUNTIME=docker else echo "serve.sh: neither podman nor docker found" >&2; exit 2; fi fi if [ "${1:-}" = "--stop" ]; then "$RUNTIME" stop "$NAME" >/dev/null 2>&1 || true "$RUNTIME" rm "$NAME" >/dev/null 2>&1 || true echo "stopped $NAME" exit 0 fi [ -f "$MODEL_DIR/config.json" ] || { echo "serve.sh: MODEL_DIR=$MODEL_DIR has no config.json" >&2; exit 2; } ls "$MODEL_DIR"/model-*.safetensors >/dev/null 2>&1 || { echo "serve.sh: no model shards in $MODEL_DIR" >&2; exit 2; } [ -f "$MODEL_DIR/matilda-release.json" ] && [ -f "$MODEL_DIR/runtime/adapters.safetensors" ] || { echo "serve.sh: $MODEL_DIR is missing matilda-release.json or runtime/adapters.safetensors" >&2; exit 2; } [ -e /dev/kfd ] && [ -d /dev/dri ] || { echo "serve.sh: /dev/kfd or /dev/dri missing (AMD GPU driver not loaded?)" >&2; exit 2; } MODEL_DIR="$(cd "$MODEL_DIR" && pwd -P)" if "$RUNTIME" container inspect "$NAME" >/dev/null 2>&1; then echo "serve.sh: a container named $NAME already exists; run '$0 --stop' first" >&2 exit 2 fi mkdir -p "$CACHE_DIR" "$KERNEL_CFG_DIR" "$RUNTIME" image inspect "$IMAGE" >/dev/null 2>&1 || "$RUNTIME" pull "$IMAGE" # podman needs keep-groups for GPU device access when rootless, and the k8s-file # log driver so that "podman logs" shows the server output. EXTRA_ARGS=() [ "$RUNTIME" = podman ] && EXTRA_ARGS=(--group-add keep-groups --log-driver k8s-file) # Optional engine settings, passed through only when set. TUNING=() for v in MAX_MODEL_LEN MAX_NUM_SEQS MAX_NUM_BATCHED_TOKENS GPU_MEMORY_UTILIZATION; do [ -n "${!v:-}" ] && TUNING+=(-e "$v=${!v}") done "$RUNTIME" run -d --name "$NAME" \ --device=/dev/kfd --device=/dev/dri "${EXTRA_ARGS[@]}" \ --network=host --ipc=host --security-opt seccomp=unconfined --ulimit memlock=-1 \ -e PORT="$PORT" -e TP="$TP" -e MATILDA_VERBOSE="${MATILDA_VERBOSE:-0}" "${TUNING[@]}" \ -v "$MODEL_DIR":/models/Matilda-V3:ro \ -v "$CACHE_DIR":/root/.cache \ -v "$KERNEL_CFG_DIR":/tmp/aiter_configs \ "$IMAGE" >/dev/null echo "started $NAME from $IMAGE" echo "logs: $RUNTIME logs -f $NAME" echo "API: http://localhost:$PORT/v1 (first start compiles GPU kernels: 20 to 40 minutes)" if [ "${1:-}" = "--wait" ]; then echo "waiting for http://localhost:$PORT/health ..." while :; do if curl -sf -m 3 "http://localhost:$PORT/health" >/dev/null 2>&1; then echo "ready"; exit 0; fi state="$("$RUNTIME" inspect -f '{{.State.Status}}' "$NAME" 2>/dev/null || echo gone)" if [ "$state" != running ]; then echo "serve.sh: container is $state; last log lines:" >&2 "$RUNTIME" logs --tail 20 "$NAME" >&2 || true exit 1 fi sleep 10 done fi