Auto Upload Agent commited on
Commit
84fbb3b
·
1 Parent(s): 370fce7

v8: 修 io_probe 的 pread 計數 bug(SSD 337/667/1328/2558 MB/s)+ launcher arena 語意修正

Browse files
llama_server.sh CHANGED
@@ -192,15 +192,24 @@ fi
192
 
193
  # arena = 預算 − 非 expert 權重(常駐) − KV − compute
194
  ARENA_MB=$(( RAM_BUDGET_MB - NON_EXPERT_MIB - RESERVE_MB ))
195
- [ "$ARENA_MB" -lt 0 ] && ARENA_MB=0
196
- EXPERT_SLOTS=$(( ARENA_MB * MIB / EXPERT_BYTES ))
 
 
 
 
 
 
 
 
 
197
 
198
  say "RAM 可用 ${USABLE_MB} MiB(cgroup 上限 ${CGROUP_MB} MiB)"
199
  say "CPU ${THREADS} threads(配額 ${QUOTA:-無}/可用核心 $(affinity_cpus))、GPU ${GPU_MB} MiB"
200
  say "模型結構:${N_LAYER} 層 × ${N_EXPERT} experts(每 token 用 ${N_EXPERT_USED}),單一 expert ≈ $(( EXPERT_BYTES / 1024 )) KiB"
201
  say "RAM 預算 ${RAM_BUDGET_MB} MiB(來源:${BUDGET_SRC})"
202
  say "推導:非 expert 權重=${NON_EXPERT_MIB} MiB(常駐) KV=${KV_RESERVE_MB} MiB compute=${COMPUTE_RESERVE_MB} MiB"
203
- say "可配置 arena=${ARENA_MB} MiB(約 ${EXPERT_SLOTS} 個 expert 槽;expert 總量 ${EXPERT_TOTAL_MIB} MiB)"
204
 
205
  refresh_memory_budget() {
206
  # Build/download 可能改變 page-cache/cgroup 用量,啟動前一定要重新讀一次。
@@ -223,8 +232,12 @@ refresh_memory_budget() {
223
  BUDGET_SRC="自動偵測(可用 ${USABLE_MB} − 系統保留 ${SYSTEM_RESERVE_MB})"
224
  fi
225
  ARENA_MB=$(( RAM_BUDGET_MB - NON_EXPERT_MIB - RESERVE_MB ))
226
- [ "$ARENA_MB" -lt 0 ] && ARENA_MB=0
227
- EXPERT_SLOTS=$(( ARENA_MB * MIB / EXPERT_BYTES ))
 
 
 
 
228
  [ "$EXPERT_SLOTS" -lt 0 ] && EXPERT_SLOTS=0
229
  say "啟動前重新偵測:可用 RAM=${USABLE_MB} MiB,預算=${RAM_BUDGET_MB} MiB,arena=${ARENA_MB} MiB"
230
  }
@@ -398,6 +411,7 @@ plan_report() {
398
  KV 預留 ${KV_RESERVE_MB} MiB(ctx $CTX,f16;理論值 $KV_ESTIMATE_MB MiB,${KV_ELEMS} 元素/token)
399
  compute ${COMPUTE_RESERVE_MB} MiB
400
  arena ${ARENA_MB} MiB → 約 ${EXPERT_SLOTS} 個 expert 槽(expert 總量 ${EXPERT_TOTAL_MIB} MiB)
 
401
  threads $THREADS(配額 ${QUOTA:-無}/可用核心 $(affinity_cpus))
402
  ubatch $UBATCH ctx $CTX(模型上限 $MODEL_MAX_CTX)
403
  GPU ${GPU_MB} MiB
@@ -454,8 +468,6 @@ if [ "$ARENA_MB" -gt 0 ]; then
454
  export ST_ARENA_MB="$ARENA_MB"
455
  else
456
  unset ST_ARENA_MB
457
- say "注意:預算 ${RAM_BUDGET_MB} MiB 不足以同時負擔非 expert 權重(${NON_EXPERT_MIB})+ KV/compute(${RESERVE_MB}),"
458
- say " expert 分頁將使用全部 ${RAM_BUDGET_MB} MiB 當預算。"
459
  fi
460
  export ST_STATS_FILE="${ST_STATS_FILE:-$WORK_DIR/st-stats.json}"
461
  export ST_KV_RESERVE_MB="$KV_RESERVE_MB"
@@ -476,7 +488,7 @@ if [ "$VERIFY" = 1 ]; then
476
  done
477
  python3 "$SCRIPT_DIR/tools/verify_run.py" --port "$PORT" \
478
  --pid "$SRV_PID" --model "$MODEL_PATH" --out "$SCRIPT_DIR/validate/last-run.json" \
479
- --ram-budget-mb "$RAM_BUDGET_MB" --arena-mb "$ARENA_MB" --expert-slots "$EXPERT_SLOTS"
480
  exit $?
481
  fi
482
 
 
192
 
193
  # arena = 預算 − 非 expert 權重(常駐) − KV − compute
194
  ARENA_MB=$(( RAM_BUDGET_MB - NON_EXPERT_MIB - RESERVE_MB ))
195
+ PAGER_BUDGET_NOTE=""
196
+ if [ "$ARENA_MB" -gt 0 ]; then
197
+ PAGER_BUDGET_MB="$ARENA_MB"
198
+ else
199
+ # 預算已經不足以同時負擔「非 expert 權重 + KV + compute」→ 讓分頁器
200
+ # 用整個預算當 expert 額度。這不是繞過,而是因為此時 KV/compute
201
+ # 根本不可能用到那麼多,算式本身失去意義。
202
+ PAGER_BUDGET_MB="$RAM_BUDGET_MB"
203
+ PAGER_BUDGET_NOTE="(預算不足以扣掉常駐權重與 KV/compute,故用整個預算)"
204
+ fi
205
+ EXPERT_SLOTS=$(( PAGER_BUDGET_MB * MIB / EXPERT_BYTES ))
206
 
207
  say "RAM 可用 ${USABLE_MB} MiB(cgroup 上限 ${CGROUP_MB} MiB)"
208
  say "CPU ${THREADS} threads(配額 ${QUOTA:-無}/可用核心 $(affinity_cpus))、GPU ${GPU_MB} MiB"
209
  say "模型結構:${N_LAYER} 層 × ${N_EXPERT} experts(每 token 用 ${N_EXPERT_USED}),單一 expert ≈ $(( EXPERT_BYTES / 1024 )) KiB"
210
  say "RAM 預算 ${RAM_BUDGET_MB} MiB(來源:${BUDGET_SRC})"
211
  say "推導:非 expert 權重=${NON_EXPERT_MIB} MiB(常駐) KV=${KV_RESERVE_MB} MiB compute=${COMPUTE_RESERVE_MB} MiB"
212
+ say "可配置 arena=${ARENA_MB} MiB → 實際分頁預算 ${PAGER_BUDGET_MB} MiB(約 ${EXPERT_SLOTS} 個 expert 槽;expert 總量 ${EXPERT_TOTAL_MIB} MiB)"
213
 
214
  refresh_memory_budget() {
215
  # Build/download 可能改變 page-cache/cgroup 用量,啟動前一定要重新讀一次。
 
232
  BUDGET_SRC="自動偵測(可用 ${USABLE_MB} − 系統保留 ${SYSTEM_RESERVE_MB})"
233
  fi
234
  ARENA_MB=$(( RAM_BUDGET_MB - NON_EXPERT_MIB - RESERVE_MB ))
235
+ if [ "$ARENA_MB" -gt 0 ]; then
236
+ PAGER_BUDGET_MB="$ARENA_MB"
237
+ else
238
+ PAGER_BUDGET_MB="$RAM_BUDGET_MB"
239
+ fi
240
+ EXPERT_SLOTS=$(( PAGER_BUDGET_MB * MIB / EXPERT_BYTES ))
241
  [ "$EXPERT_SLOTS" -lt 0 ] && EXPERT_SLOTS=0
242
  say "啟動前重新偵測:可用 RAM=${USABLE_MB} MiB,預算=${RAM_BUDGET_MB} MiB,arena=${ARENA_MB} MiB"
243
  }
 
411
  KV 預留 ${KV_RESERVE_MB} MiB(ctx $CTX,f16;理論值 $KV_ESTIMATE_MB MiB,${KV_ELEMS} 元素/token)
412
  compute ${COMPUTE_RESERVE_MB} MiB
413
  arena ${ARENA_MB} MiB → 約 ${EXPERT_SLOTS} 個 expert 槽(expert 總量 ${EXPERT_TOTAL_MIB} MiB)
414
+ 分頁預算 實際給分頁器的額度 ${PAGER_BUDGET_MB} MiB${PAGER_BUDGET_NOTE}
415
  threads $THREADS(配額 ${QUOTA:-無}/可用核心 $(affinity_cpus))
416
  ubatch $UBATCH ctx $CTX(模型上限 $MODEL_MAX_CTX)
417
  GPU ${GPU_MB} MiB
 
468
  export ST_ARENA_MB="$ARENA_MB"
469
  else
470
  unset ST_ARENA_MB
 
 
471
  fi
472
  export ST_STATS_FILE="${ST_STATS_FILE:-$WORK_DIR/st-stats.json}"
473
  export ST_KV_RESERVE_MB="$KV_RESERVE_MB"
 
488
  done
489
  python3 "$SCRIPT_DIR/tools/verify_run.py" --port "$PORT" \
490
  --pid "$SRV_PID" --model "$MODEL_PATH" --out "$SCRIPT_DIR/validate/last-run.json" \
491
+ --ram-budget-mb "$PAGER_BUDGET_MB" --arena-mb "$ARENA_MB" --expert-slots "$EXPERT_SLOTS"
492
  exit $?
493
  fi
494
 
tools/io_probe.py CHANGED
@@ -21,26 +21,40 @@ def drop_cache(fd: int, size: int) -> None:
21
 
22
 
23
  def read_seq(path: str, size: int, threads: int) -> dict:
24
- """threads 條執行緒各自讀不同區段,量「有效頻寬」。"""
 
 
 
 
25
  import threading
 
26
  per = size // threads
27
  results = [0] * threads
28
  fds = [os.open(path, os.O_RDONLY) for _ in range(threads)]
29
  for fd in fds:
30
  drop_cache(fd, size)
31
 
 
 
32
  def work(i: int):
33
  fd = fds[i]
34
  off = i * per
35
  left = per
36
- chunk = 4 * MIB
37
  got = 0
38
- while left > 0:
39
- n = os.pread(fd, min(chunk, left), off + got)
40
- if not n:
41
- break
42
- got += n
43
- left -= n
 
 
 
 
 
 
 
 
44
  results[i] = got
45
 
46
  ts = [threading.Thread(target=work, args=(i,)) for i in range(threads)]
@@ -53,6 +67,8 @@ def read_seq(path: str, size: int, threads: int) -> dict:
53
  for fd in fds:
54
  os.close(fd)
55
  total = sum(results)
 
 
56
  return {"threads": threads, "bytes": total, "seconds": round(dt, 3),
57
  "mb_per_s": round(total / MIB / dt, 1) if dt > 0 else None}
58
 
 
21
 
22
 
23
  def read_seq(path: str, size: int, threads: int) -> dict:
24
+ """threads 條執行緒各自讀不同區段,量「有效頻寬」。
25
+
26
+ 迴圈裡不能把 bytes 寫進 bytes 物件(Windows 沒事,但這裡要可攜);
27
+ 一律用 os.pread 拿回讀到的長度。
28
+ """
29
  import threading
30
+
31
  per = size // threads
32
  results = [0] * threads
33
  fds = [os.open(path, os.O_RDONLY) for _ in range(threads)]
34
  for fd in fds:
35
  drop_cache(fd, size)
36
 
37
+ errors: list[str] = []
38
+
39
  def work(i: int):
40
  fd = fds[i]
41
  off = i * per
42
  left = per
 
43
  got = 0
44
+ try:
45
+ while left > 0:
46
+ want = min(4 * MIB, left)
47
+ # os.pread 回傳的是**資料本身**,不是讀到的位元組數。
48
+ # 寫成 `got += n` 會得到 "int + bytes" 的 TypeError,
49
+ # 而執行緒裡的例外不會讓主程式失敗 —— 結果就是
50
+ # 「bytes: 0、0.015 秒、mb_per_s 0.0」這種完全沒有 I/O 的假象。
51
+ buf = os.pread(fd, want, off + got)
52
+ if not buf:
53
+ break
54
+ got += len(buf)
55
+ left -= len(buf)
56
+ except OSError as e:
57
+ errors.append(str(e))
58
  results[i] = got
59
 
60
  ts = [threading.Thread(target=work, args=(i,)) for i in range(threads)]
 
67
  for fd in fds:
68
  os.close(fd)
69
  total = sum(results)
70
+ if errors:
71
+ print(f"[io_probe] thread errors: {errors[:3]}", file=sys.stderr)
72
  return {"threads": threads, "bytes": total, "seconds": round(dt, 3),
73
  "mb_per_s": round(total / MIB / dt, 1) if dt > 0 else None}
74
 
validate/io-probe.json ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "when": "2026-10-07T16:30:49+0200",
3
+ "file": "/root/work/models/SmallThinker-4B-A0.6B-Instruct.Q4_K.gguf",
4
+ "size_mib": 512,
5
+ "fs": 4096,
6
+ "sequential": [
7
+ {
8
+ "threads": 1,
9
+ "bytes": 536870912,
10
+ "seconds": 1.518,
11
+ "mb_per_s": 337.3
12
+ },
13
+ {
14
+ "threads": 2,
15
+ "bytes": 536870912,
16
+ "seconds": 0.767,
17
+ "mb_per_s": 667.1
18
+ },
19
+ {
20
+ "threads": 4,
21
+ "bytes": 536870912,
22
+ "seconds": 0.386,
23
+ "mb_per_s": 1327.5
24
+ },
25
+ {
26
+ "threads": 8,
27
+ "bytes": 536870912,
28
+ "seconds": 0.2,
29
+ "mb_per_s": 2557.6
30
+ }
31
+ ],
32
+ "random_4k": {
33
+ "count": 512,
34
+ "seconds": 0.29,
35
+ "iops": 1767.1
36
+ },
37
+ "cpu_count": 192
38
+ }
validate/last-run.json CHANGED
@@ -1,38 +1,38 @@
1
  {
2
- "when": "2026-10-07T16:28:45+0200",
3
  "model": "/root/work/models/SmallThinker-4B-A0.6B-Instruct.Q4_K.gguf",
4
  "port": 8080,
5
- "pid": 34773,
6
  "ram_budget_mb": 512,
7
- "arena_mb": 0,
8
- "expert_slots": 0,
9
  "chat": {
10
  "prompt_tokens": 37,
11
  "completion_tokens": 16,
12
- "seconds": 3.8,
13
- "tok_per_s": 4.2107,
14
  "content": "A **MoE (Model Parallelism) layer** enables a neural network to",
15
  "timings": {
16
  "cache_n": 0,
17
  "prompt_n": 37,
18
- "prompt_ms": 3010.816,
19
- "prompt_per_token_ms": 81.3734054054054,
20
- "prompt_per_second": 12.289027293597485,
21
  "predicted_n": 16,
22
- "predicted_ms": 771.041,
23
- "predicted_per_token_ms": 51.40273333333334,
24
- "predicted_per_second": 19.45421838786783
25
  }
26
  },
27
  "peak": {
28
- "total_rss_gb": 1.0921,
29
  "anon_rss_gb": 0.1931,
30
- "file_rss_gb": 1.0864,
31
  "peak_swap_gb": 0.0,
32
- "hwm_rss_gb": 1.0921
33
  },
34
  "io_delta": {
35
- "read_bytes": 1156939776,
36
  "rchar": 0,
37
  "write_bytes": 0
38
  },
@@ -47,7 +47,7 @@
47
  "resident_bytes": 535071744,
48
  "budget_bytes": 536870912,
49
  "hit_rate": 0.4122,
50
- "proc_self_read_bytes": 2442625024,
51
  "proc_self_rchar": 11946335
52
  },
53
  "pager_expert_resident_mib": 510.3,
@@ -55,6 +55,5 @@
55
  "pager_expert_within_budget": true,
56
  "pager_hit_rate": 0.4122,
57
  "ok": true,
58
- "model_size_mib": 2508,
59
- "note": "llama_server.sh --verify,ST_RAM_BUDGET_MB=512。total RSS 1.09 GiB = 非expert權重 267 MiB + expert 分頁 510 MiB(預算 512)+ KV/compute/sampler。"
60
- }
 
1
  {
2
+ "when": "2026-10-07T16:29:51+0200",
3
  "model": "/root/work/models/SmallThinker-4B-A0.6B-Instruct.Q4_K.gguf",
4
  "port": 8080,
5
+ "pid": 35228,
6
  "ram_budget_mb": 512,
7
+ "arena_mb": -1875,
8
+ "expert_slots": 252,
9
  "chat": {
10
  "prompt_tokens": 37,
11
  "completion_tokens": 16,
12
+ "seconds": 3.67,
13
+ "tok_per_s": 4.364,
14
  "content": "A **MoE (Model Parallelism) layer** enables a neural network to",
15
  "timings": {
16
  "cache_n": 0,
17
  "prompt_n": 37,
18
+ "prompt_ms": 2984.837,
19
+ "prompt_per_token_ms": 80.67127027027027,
20
+ "prompt_per_second": 12.395986782527824,
21
  "predicted_n": 16,
22
+ "predicted_ms": 663.495,
23
+ "predicted_per_token_ms": 44.233,
24
+ "predicted_per_second": 22.60755544502973
25
  }
26
  },
27
  "peak": {
28
+ "total_rss_gb": 1.091,
29
  "anon_rss_gb": 0.1931,
30
+ "file_rss_gb": 1.0853,
31
  "peak_swap_gb": 0.0,
32
+ "hwm_rss_gb": 1.091
33
  },
34
  "io_delta": {
35
+ "read_bytes": 1156587520,
36
  "rchar": 0,
37
  "write_bytes": 0
38
  },
 
47
  "resident_bytes": 535071744,
48
  "budget_bytes": 536870912,
49
  "hit_rate": 0.4122,
50
+ "proc_self_read_bytes": 2439876608,
51
  "proc_self_rchar": 11946335
52
  },
53
  "pager_expert_resident_mib": 510.3,
 
55
  "pager_expert_within_budget": true,
56
  "pager_hit_rate": 0.4122,
57
  "ok": true,
58
+ "model_size_mib": 2508
59
+ }