File size: 4,378 Bytes
1256ff0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
#!/usr/bin/env bash
# Start the Matilda-K3 server (OpenAI-compatible API) on one 8-GPU node.
#
# Usage:
#   MODEL_DIR=/path/to/Matilda-K3 ./serve.sh          # start in the background
#   MODEL_DIR=/path/to/Matilda-K3 ./serve.sh --wait   # start and block until ready
#   ./serve.sh --stop                                       # stop and remove the container
#
# Environment:
#   MODEL_DIR        directory holding this repository's files (default: script directory)
#   IMAGE            runtime image (default: docker.io/maincodehq/matilda-vllm:kimi-k3)
#   NAME             container name (default: matilda-k3)
#   PORT             API port (default: 8000)
#   TP               number of GPUs, tensor parallel (default: 8)
#   CACHE_DIR        kernel build cache, keep it between starts (default: ~/matilda-cache)
#   KERNEL_CFG_DIR   kernel tuning cache, keep it between starts (default: ~/matilda-kernel-cfg)
#   MATILDA_VERBOSE  1 streams the engine log to the container log (default: 0)
#   MAX_MODEL_LEN, MAX_NUM_SEQS, MAX_NUM_BATCHED_TOKENS, GPU_MEMORY_UTILIZATION
#                    optional engine settings (runtime defaults: 262144, 64, 16384, 0.88)
#   RUNTIME          podman or docker (default: whichever is installed, podman first)
set -euo pipefail

HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
IMAGE="${IMAGE:-docker.io/maincodehq/matilda-vllm:kimi-k3}"
NAME="${NAME:-matilda-k3}"
PORT="${PORT:-8000}"
TP="${TP:-8}"
MODEL_DIR="${MODEL_DIR:-$HERE}"
CACHE_DIR="${CACHE_DIR:-$HOME/matilda-cache}"
KERNEL_CFG_DIR="${KERNEL_CFG_DIR:-$HOME/matilda-kernel-cfg}"

if [ -z "${RUNTIME:-}" ]; then
  if command -v podman >/dev/null 2>&1; then RUNTIME=podman
  elif command -v docker >/dev/null 2>&1; then RUNTIME=docker
  else echo "serve.sh: neither podman nor docker found" >&2; exit 2; fi
fi

if [ "${1:-}" = "--stop" ]; then
  "$RUNTIME" stop "$NAME" >/dev/null 2>&1 || true
  "$RUNTIME" rm "$NAME" >/dev/null 2>&1 || true
  echo "stopped $NAME"
  exit 0
fi

[ -f "$MODEL_DIR/config.json" ] || { echo "serve.sh: MODEL_DIR=$MODEL_DIR has no config.json" >&2; exit 2; }
ls "$MODEL_DIR"/model-*.safetensors >/dev/null 2>&1 || { echo "serve.sh: no model shards in $MODEL_DIR" >&2; exit 2; }
[ -f "$MODEL_DIR/matilda-release.json" ] && [ -f "$MODEL_DIR/runtime/adapters.safetensors" ] || {
  echo "serve.sh: $MODEL_DIR is missing matilda-release.json or runtime/adapters.safetensors" >&2; exit 2; }
[ -e /dev/kfd ] && [ -d /dev/dri ] || { echo "serve.sh: /dev/kfd or /dev/dri missing (AMD GPU driver not loaded?)" >&2; exit 2; }
MODEL_DIR="$(cd "$MODEL_DIR" && pwd -P)"

if "$RUNTIME" container inspect "$NAME" >/dev/null 2>&1; then
  echo "serve.sh: a container named $NAME already exists; run '$0 --stop' first" >&2
  exit 2
fi

mkdir -p "$CACHE_DIR" "$KERNEL_CFG_DIR"
"$RUNTIME" image inspect "$IMAGE" >/dev/null 2>&1 || "$RUNTIME" pull "$IMAGE"

# podman needs keep-groups for GPU device access when rootless, and the k8s-file
# log driver so that "podman logs" shows the server output.
EXTRA_ARGS=()
[ "$RUNTIME" = podman ] && EXTRA_ARGS=(--group-add keep-groups --log-driver k8s-file)

# Optional engine settings, passed through only when set.
TUNING=()
for v in MAX_MODEL_LEN MAX_NUM_SEQS MAX_NUM_BATCHED_TOKENS GPU_MEMORY_UTILIZATION; do
  [ -n "${!v:-}" ] && TUNING+=(-e "$v=${!v}")
done

"$RUNTIME" run -d --name "$NAME" \
  --device=/dev/kfd --device=/dev/dri "${EXTRA_ARGS[@]}" \
  --network=host --ipc=host --security-opt seccomp=unconfined --ulimit memlock=-1 \
  -e PORT="$PORT" -e TP="$TP" -e MATILDA_VERBOSE="${MATILDA_VERBOSE:-0}" "${TUNING[@]}" \
  -v "$MODEL_DIR":/models/Matilda-V3:ro \
  -v "$CACHE_DIR":/root/.cache \
  -v "$KERNEL_CFG_DIR":/tmp/aiter_configs \
  "$IMAGE" >/dev/null

echo "started $NAME from $IMAGE"
echo "logs:  $RUNTIME logs -f $NAME"
echo "API:   http://localhost:$PORT/v1   (first start compiles GPU kernels: 20 to 40 minutes)"

if [ "${1:-}" = "--wait" ]; then
  echo "waiting for http://localhost:$PORT/health ..."
  while :; do
    if curl -sf -m 3 "http://localhost:$PORT/health" >/dev/null 2>&1; then echo "ready"; exit 0; fi
    state="$("$RUNTIME" inspect -f '{{.State.Status}}' "$NAME" 2>/dev/null || echo gone)"
    if [ "$state" != running ]; then
      echo "serve.sh: container is $state; last log lines:" >&2
      "$RUNTIME" logs --tail 20 "$NAME" >&2 || true
      exit 1
    fi
    sleep 10
  done
fi