Text Generation
llama-cpp-python
GGUF
llama.cpp
Mixture of Experts
ssd-offload
smallthinker
expert-paging
low-ram
Instructions to use HelloSun/SmallThinker4b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- llama-cpp-python
How to use HelloSun/SmallThinker4b with llama-cpp-python:
# !pip install llama-cpp-python from llama_cpp import Llama llm = Llama.from_pretrained( repo_id="HelloSun/SmallThinker4b", filename="{{GGUF_FILE}}", )output = llm( "Once upon a time,", max_tokens=512, echo=True ) print(output)
- Notebooks
- Google Colab
- Kaggle
Auto Upload Agent commited on
Commit ·
625b02c
1
Parent(s): 9460178
v1: 驗證通過 — llama-server 起來、chat 正常、peak RSS 4.30 GiB / swap 0
Browse files- llama_server.sh +10 -3
- validate/last-run.json +41 -0
llama_server.sh
CHANGED
|
@@ -243,12 +243,19 @@ ensure_patches() {
|
|
| 243 |
say "使用本地的 patch:$(cd "$PATCH_DIR" && ls *.patch | tr '\n' ' ')"
|
| 244 |
return
|
| 245 |
fi
|
|
|
|
|
|
|
| 246 |
step "從 $HF_REPO 取得 patch"
|
| 247 |
-
local p
|
| 248 |
for p in 0001-st-expert-pager.patch; do
|
| 249 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 250 |
done
|
| 251 |
-
[
|
| 252 |
}
|
| 253 |
|
| 254 |
# ================================================== 3. 取得 llama.cpp 原始碼
|
|
|
|
| 243 |
say "使用本地的 patch:$(cd "$PATCH_DIR" && ls *.patch | tr '\n' ' ')"
|
| 244 |
return
|
| 245 |
fi
|
| 246 |
+
# 遠端還沒有 patch(第一版就是這樣)是正常狀況,不是錯誤:
|
| 247 |
+
# 沒有 patch 時走上游行為,權重一樣是 SSD 上的 file-backed mmap。
|
| 248 |
step "從 $HF_REPO 取得 patch"
|
| 249 |
+
local p got=0
|
| 250 |
for p in 0001-st-expert-pager.patch; do
|
| 251 |
+
if curl -fsSL -H "Authorization: Bearer ${HF_TOKEN:-}" -o "$PATCH_DIR/$p.part" \
|
| 252 |
+
"https://huggingface.co/$HF_REPO/resolve/main/patches/$p" 2>/dev/null; then
|
| 253 |
+
mv -f "$PATCH_DIR/$p.part" "$PATCH_DIR/$p"; say " patches/$p"; got=1
|
| 254 |
+
else
|
| 255 |
+
rm -f "$PATCH_DIR/$p.part"; say " (遠端沒有 $p)"
|
| 256 |
+
fi
|
| 257 |
done
|
| 258 |
+
[ "$got" = 1 ] || say " 用上游行為:權重走 SSD file-backed mmap,沒有 expert 級分頁"
|
| 259 |
}
|
| 260 |
|
| 261 |
# ================================================== 3. 取得 llama.cpp 原始碼
|
validate/last-run.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"when": "2026-10-07T15:45:38+0200",
|
| 3 |
+
"model": "/root/work/models/SmallThinker-4B-A0.6B-Instruct.Q4_K.gguf",
|
| 4 |
+
"port": 8080,
|
| 5 |
+
"pid": 5409,
|
| 6 |
+
"ram_budget_mb": 92627,
|
| 7 |
+
"arena_mb": 90240,
|
| 8 |
+
"expert_slots": 44562,
|
| 9 |
+
"chat": {
|
| 10 |
+
"prompt_tokens": 37,
|
| 11 |
+
"completion_tokens": 16,
|
| 12 |
+
"seconds": 0.51,
|
| 13 |
+
"tok_per_s": 31.6285,
|
| 14 |
+
"content": "A **MoE (Model Parallelism) layer** enables a neural network to",
|
| 15 |
+
"timings": {
|
| 16 |
+
"cache_n": 0,
|
| 17 |
+
"prompt_n": 37,
|
| 18 |
+
"prompt_ms": 216.341,
|
| 19 |
+
"prompt_per_token_ms": 5.8470540540540545,
|
| 20 |
+
"prompt_per_second": 171.02629644866207,
|
| 21 |
+
"predicted_n": 16,
|
| 22 |
+
"predicted_ms": 255.155,
|
| 23 |
+
"predicted_per_token_ms": 17.01033333333333,
|
| 24 |
+
"predicted_per_second": 58.78779565362231
|
| 25 |
+
}
|
| 26 |
+
},
|
| 27 |
+
"peak": {
|
| 28 |
+
"total_rss_gb": 4.3048,
|
| 29 |
+
"anon_rss_gb": 1.8637,
|
| 30 |
+
"file_rss_gb": 4.2991,
|
| 31 |
+
"peak_swap_gb": 0.0,
|
| 32 |
+
"hwm_rss_gb": 4.3048
|
| 33 |
+
},
|
| 34 |
+
"io_delta": {
|
| 35 |
+
"read_bytes": 0,
|
| 36 |
+
"rchar": 6,
|
| 37 |
+
"write_bytes": 0
|
| 38 |
+
},
|
| 39 |
+
"model_size_mib": 2508,
|
| 40 |
+
"ok": true
|
| 41 |
+
}
|