File size: 2,337 Bytes
ddf8c5b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
#!/usr/bin/env bash
# k3-test/setup.sh — one-time box setup: packages, hf tooling, llama.cpp build.
# Idempotent; safe to re-run.
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "$SCRIPT_DIR/config.env"
set -euo pipefail

SUDO=""; [[ ${EUID:-$(id -u)} -ne 0 ]] && SUDO="sudo -n"

echo "==> [1/4] APT packages"
$SUDO apt-get update -qq
DEBIAN_FRONTEND=noninteractive $SUDO apt-get install -y -qq \
  build-essential cmake git git-lfs curl wget aria2 \
  fio stress-ng sysbench sysstat dmidecode numactl pciutils \
  python3 python3-pip python3-venv nvme-cli htop tmux

# cuBLAS dev headers matching the installed toolkit (llama.cpp CMake needs CUDA::cublas;
# vast.ai's own llama.cpp image ships only the runtime libs — discovered 2026-08-25)
CUDA_REL=$(nvcc --version 2>/dev/null | grep -oP 'release \K[0-9]+\.[0-9]+' || true)
if [[ -n "$CUDA_REL" ]]; then
  DEBIAN_FRONTEND=noninteractive $SUDO apt-get install -y -qq \
    "libcublas-${CUDA_REL/./-}" "libcublas-dev-${CUDA_REL/./-}" || true
fi

echo "==> [2/4] huggingface_hub + hf_transfer"
python3 -m pip install -U --quiet --break-system-packages "huggingface_hub[hf_transfer]" 2>/dev/null \
  || python3 -m pip install -U --quiet "huggingface_hub[hf_transfer]"
need() { command -v "$1" >/dev/null 2>&1 || { echo "FATAL: missing $1"; exit 1; }; }
need hf || { pip3 install -U --quiet "huggingface_hub[hf_transfer]"; need hf; }

echo "==> [3/4] llama.cpp @ ${LLAMA_REF:0:10}"
if [[ ! -d "$LLAMA_DIR/.git" ]]; then
  git clone "$LLAMA_GIT" "$LLAMA_DIR"
fi
git -C "$LLAMA_DIR" fetch --quiet origin "$LLAMA_REF" || true
git -C "$LLAMA_DIR" checkout --quiet "$LLAMA_REF"

echo "==> [4/4] build (CUDA)"
if nvidia-smi -L >/dev/null 2>&1; then CUDA_ON=ON; else CUDA_ON=OFF; echo "WARN: no GPUs, CPU-only build"; fi
cmake -S "$LLAMA_DIR" -B "$LLAMA_DIR/build" \
  -DGGML_CUDA=$CUDA_ON -DBUILD_SHARED_LIBS=OFF -DCMAKE_BUILD_TYPE=Release
cmake --build "$LLAMA_DIR/build" --config Release -j"$(nproc)" \
  --target llama-cli llama-server llama-bench llama-gguf-split llama-perplexity

"$LLAMA_DIR/build/bin/llama-cli" --version | head -3
"$SCRIPT_DIR/prompts/gen_prompts.sh" >/dev/null 2>&1 || true
echo "==> setup OK.  Binaries in $LLAMA_DIR/build/bin"
echo "NOTE: llama-cli --help | grep -iE 'cpu-moe|lookup|numa|load-mode|override-tensor'  # flags this plan relies on"