File size: 4,378 Bytes
1256ff0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 | #!/usr/bin/env bash
# Start the Matilda-K3 server (OpenAI-compatible API) on one 8-GPU node.
#
# Usage:
# MODEL_DIR=/path/to/Matilda-K3 ./serve.sh # start in the background
# MODEL_DIR=/path/to/Matilda-K3 ./serve.sh --wait # start and block until ready
# ./serve.sh --stop # stop and remove the container
#
# Environment:
# MODEL_DIR directory holding this repository's files (default: script directory)
# IMAGE runtime image (default: docker.io/maincodehq/matilda-vllm:kimi-k3)
# NAME container name (default: matilda-k3)
# PORT API port (default: 8000)
# TP number of GPUs, tensor parallel (default: 8)
# CACHE_DIR kernel build cache, keep it between starts (default: ~/matilda-cache)
# KERNEL_CFG_DIR kernel tuning cache, keep it between starts (default: ~/matilda-kernel-cfg)
# MATILDA_VERBOSE 1 streams the engine log to the container log (default: 0)
# MAX_MODEL_LEN, MAX_NUM_SEQS, MAX_NUM_BATCHED_TOKENS, GPU_MEMORY_UTILIZATION
# optional engine settings (runtime defaults: 262144, 64, 16384, 0.88)
# RUNTIME podman or docker (default: whichever is installed, podman first)
set -euo pipefail
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
IMAGE="${IMAGE:-docker.io/maincodehq/matilda-vllm:kimi-k3}"
NAME="${NAME:-matilda-k3}"
PORT="${PORT:-8000}"
TP="${TP:-8}"
MODEL_DIR="${MODEL_DIR:-$HERE}"
CACHE_DIR="${CACHE_DIR:-$HOME/matilda-cache}"
KERNEL_CFG_DIR="${KERNEL_CFG_DIR:-$HOME/matilda-kernel-cfg}"
if [ -z "${RUNTIME:-}" ]; then
if command -v podman >/dev/null 2>&1; then RUNTIME=podman
elif command -v docker >/dev/null 2>&1; then RUNTIME=docker
else echo "serve.sh: neither podman nor docker found" >&2; exit 2; fi
fi
if [ "${1:-}" = "--stop" ]; then
"$RUNTIME" stop "$NAME" >/dev/null 2>&1 || true
"$RUNTIME" rm "$NAME" >/dev/null 2>&1 || true
echo "stopped $NAME"
exit 0
fi
[ -f "$MODEL_DIR/config.json" ] || { echo "serve.sh: MODEL_DIR=$MODEL_DIR has no config.json" >&2; exit 2; }
ls "$MODEL_DIR"/model-*.safetensors >/dev/null 2>&1 || { echo "serve.sh: no model shards in $MODEL_DIR" >&2; exit 2; }
[ -f "$MODEL_DIR/matilda-release.json" ] && [ -f "$MODEL_DIR/runtime/adapters.safetensors" ] || {
echo "serve.sh: $MODEL_DIR is missing matilda-release.json or runtime/adapters.safetensors" >&2; exit 2; }
[ -e /dev/kfd ] && [ -d /dev/dri ] || { echo "serve.sh: /dev/kfd or /dev/dri missing (AMD GPU driver not loaded?)" >&2; exit 2; }
MODEL_DIR="$(cd "$MODEL_DIR" && pwd -P)"
if "$RUNTIME" container inspect "$NAME" >/dev/null 2>&1; then
echo "serve.sh: a container named $NAME already exists; run '$0 --stop' first" >&2
exit 2
fi
mkdir -p "$CACHE_DIR" "$KERNEL_CFG_DIR"
"$RUNTIME" image inspect "$IMAGE" >/dev/null 2>&1 || "$RUNTIME" pull "$IMAGE"
# podman needs keep-groups for GPU device access when rootless, and the k8s-file
# log driver so that "podman logs" shows the server output.
EXTRA_ARGS=()
[ "$RUNTIME" = podman ] && EXTRA_ARGS=(--group-add keep-groups --log-driver k8s-file)
# Optional engine settings, passed through only when set.
TUNING=()
for v in MAX_MODEL_LEN MAX_NUM_SEQS MAX_NUM_BATCHED_TOKENS GPU_MEMORY_UTILIZATION; do
[ -n "${!v:-}" ] && TUNING+=(-e "$v=${!v}")
done
"$RUNTIME" run -d --name "$NAME" \
--device=/dev/kfd --device=/dev/dri "${EXTRA_ARGS[@]}" \
--network=host --ipc=host --security-opt seccomp=unconfined --ulimit memlock=-1 \
-e PORT="$PORT" -e TP="$TP" -e MATILDA_VERBOSE="${MATILDA_VERBOSE:-0}" "${TUNING[@]}" \
-v "$MODEL_DIR":/models/Matilda-V3:ro \
-v "$CACHE_DIR":/root/.cache \
-v "$KERNEL_CFG_DIR":/tmp/aiter_configs \
"$IMAGE" >/dev/null
echo "started $NAME from $IMAGE"
echo "logs: $RUNTIME logs -f $NAME"
echo "API: http://localhost:$PORT/v1 (first start compiles GPU kernels: 20 to 40 minutes)"
if [ "${1:-}" = "--wait" ]; then
echo "waiting for http://localhost:$PORT/health ..."
while :; do
if curl -sf -m 3 "http://localhost:$PORT/health" >/dev/null 2>&1; then echo "ready"; exit 0; fi
state="$("$RUNTIME" inspect -f '{{.State.Status}}' "$NAME" 2>/dev/null || echo gone)"
if [ "$state" != running ]; then
echo "serve.sh: container is $state; last log lines:" >&2
"$RUNTIME" logs --tail 20 "$NAME" >&2 || true
exit 1
fi
sleep 10
done
fi
|