Auto Upload Agent commited on
Commit
eb9c86c
·
1 Parent(s): 1e54449

v4: 修 prefetch 反轉 bug — peak RSS 2.56→0.59 GiB,輸出逐字相同,all_passed=true

Browse files
Files changed (2) hide show
  1. patches/0001-st-expert-pager.patch +109 -31
  2. validate/ab-test.json +565 -167
patches/0001-st-expert-pager.patch CHANGED
@@ -1,23 +1,12 @@
1
- From 90deb52647021d7b6d0ff0d97a242f959fc6dce1 Mon Sep 17 00:00:00 2001
2
- From: c <a@b>
3
- Date: Wed, 7 Oct 2026 16:15:13 +0200
4
- Subject: [PATCH] st pager: disable MAP_POPULATE when paging, fix /proc/self/io
5
- parser
6
 
7
- ---
8
- .st-patched | 0
9
- .st-unpatched | 0
10
- src/CMakeLists.txt | 1 +
11
- src/llama-graph.cpp | 10 +
12
- src/llama-model.cpp | 41 +++-
13
- src/st_pager.cpp | 535 ++++++++++++++++++++++++++++++++++++++++++++
14
- src/st_pager.h | 80 +++++++
15
- 7 files changed, 666 insertions(+), 1 deletion(-)
16
- create mode 100644 .st-patched
17
- create mode 100644 .st-unpatched
18
- create mode 100644 src/st_pager.cpp
19
- create mode 100644 src/st_pager.h
20
 
 
21
  diff --git a/.st-patched b/.st-patched
22
  new file mode 100644
23
  index 000000000..e69de29bb
@@ -64,7 +53,7 @@ index 1112ad885..e335a9d5f 100644
64
 
65
  if (weight_before_ffn) {
66
  diff --git a/src/llama-model.cpp b/src/llama-model.cpp
67
- index 1f3b80b08..a74118ae5 100644
68
  --- a/src/llama-model.cpp
69
  +++ b/src/llama-model.cpp
70
  @@ -1,5 +1,7 @@
@@ -84,22 +73,50 @@ index 1f3b80b08..a74118ae5 100644
84
  const auto & split_mode = params.split_mode;
85
  const bool use_mlock = params.load_mode == LLAMA_LOAD_MODE_MLOCK || params.load_mode == LLAMA_LOAD_MODE_MMAP_MLOCK;
86
  const auto & tensor_split = params.tensor_split;
87
- @@ -1821,7 +1825,13 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
88
  // per-tensor activation precision policy
89
  prec_policy.load(ml, *this);
90
 
91
  - ml.init_mappings(true, use_mlock ? &pimpl->mlock_mmaps : nullptr);
 
 
 
 
 
 
 
 
 
 
 
92
  + // ST expert pager:開著分頁時**不能** prefetch。
93
  + // llama.cpp 預設 prefetch=true 會用 MAP_POPULATE 把整個 GGUF(2.45 GiB)
94
  + // 在「載入模型」那一瞬间全部 fault 進 page cache —— 那一刻的 RSS 峰值
95
  + // 就已經超過任何分頁預算了,後面的 madvise 再怎麼丟都沒用(峰值已經發生)。
96
  + // 這是量測方式/時機的坑,不是分頁機制本身有問題。
97
  + // 註意 st::set_config_from_env() 必須在這之前呼叫(見 load_tensors 開頭)。
98
- + ml.init_mappings(!st::prefetch_requested(), use_mlock ? &pimpl->mlock_mmaps : nullptr);
99
  pimpl->mappings.reserve(ml.mappings.size());
100
 
101
  // create the backend buffers
102
- @@ -1969,6 +1979,35 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
103
  }
104
  }
105
 
@@ -123,7 +140,20 @@ index 1f3b80b08..a74118ae5 100644
123
  + by_name.emplace(nm, t);
124
  + }
125
  +
126
- + if (getenv("ST_VERBOSE") && !map_base) {
 
 
 
 
 
 
 
 
 
 
 
 
 
127
  + LLAMA_LOG_INFO("%s: [st] 沒有 mapping(mappings=%zu files=%zu),無法分頁\n",
128
  + __func__, pimpl->mappings.size(), ml.files.size());
129
  + }
@@ -137,10 +167,10 @@ index 1f3b80b08..a74118ae5 100644
137
 
138
  diff --git a/src/st_pager.cpp b/src/st_pager.cpp
139
  new file mode 100644
140
- index 000000000..30d339827
141
  --- /dev/null
142
  +++ b/src/st_pager.cpp
143
- @@ -0,0 +1,535 @@
144
  +// ST expert pager —— 熱 expert 留在 RAM,冷 expert 丟回 SSD
145
  +//
146
  +// 設計取捨(為什麼不照抄 sddqwen35a3b_v01 的 pread arena):
@@ -262,7 +292,15 @@ index 000000000..30d339827
262
  +
263
  +bool enabled() { return g_enabled && !g_slots.empty(); }
264
  +
265
- +bool prefetch_requested() { return !g_enabled; }
 
 
 
 
 
 
 
 
266
  +
267
  +// ------------------------------------------------------------------ 統計
268
  +
@@ -485,6 +523,37 @@ index 000000000..30d339827
485
  +
486
  +// ------------------------------------------------------------------ 淘汰
487
  +
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
488
  +// ST_EVICT_MODE 診斷用(實測發現只做其中一步會導致輸出損毀,見 docs):
489
  +// both(預設)| madvise | fadvise | none
490
  +static int evict_mode() {
@@ -572,12 +641,17 @@ index 000000000..30d339827
572
  + }
573
  +
574
  + if (g_verbose && !to_evict.empty()) {
 
575
  + fprintf(stderr,
576
- + "[st] sweep #%llu:丟 %zu 個 expert(%.1f MiB)→ resident %.1f MiB / budget %.1f MiB\n",
 
577
  + (unsigned long long) g_st.sweeps, to_evict.size(),
578
  + freed / (1024.0 * 1024.0),
579
  + (resident - freed) / (1024.0 * 1024.0),
580
- + g_budget / (1024.0 * 1024.0));
 
 
 
581
  + }
582
  +}
583
  +
@@ -679,10 +753,10 @@ index 000000000..30d339827
679
 
680
  diff --git a/src/st_pager.h b/src/st_pager.h
681
  new file mode 100644
682
- index 000000000..43852a968
683
  --- /dev/null
684
  +++ b/src/st_pager.h
685
- @@ -0,0 +1,80 @@
686
  +// ST expert pager —— 熱 expert 留在 RAM,冷 expert 丟回 SSD
687
  +//
688
  +// 對外介面。llama.cpp 端(llama-model.cpp / llama-graph.cpp)只include 這個檔案。
@@ -755,7 +829,11 @@ index 000000000..43852a968
755
  + int layer, int n_expert, int n_expert_used);
756
  +
757
  +bool enabled();
758
- +bool prefetch_requested(); // 分頁開著時必須關掉 MAP_POPULATE
 
 
 
 
759
  +stats get_stats();
760
  +void dump_stats(const char * path);
761
  +
 
1
+ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
2
+ From: HelloSun <agent@huggingface.co>
3
+ Date: Thu, 7 Oct 2026 17:00:00 +0200
4
+ Subject: [PATCH] st: SmallThinker expert 級 SSD 分頁
 
5
 
6
+ 熱的 expert 權重留在 RAM,冷的用 madvise + posix_fadvise 丟回 SSD。
7
+ 計算路徑完全不動(仍是 llama.cpp 自己的 mul_mat_id),所以輸出與上游逐字相同。
 
 
 
 
 
 
 
 
 
 
 
8
 
9
+ ---
10
  diff --git a/.st-patched b/.st-patched
11
  new file mode 100644
12
  index 000000000..e69de29bb
 
53
 
54
  if (weight_before_ffn) {
55
  diff --git a/src/llama-model.cpp b/src/llama-model.cpp
56
+ index 1f3b80b08..662491537 100644
57
  --- a/src/llama-model.cpp
58
  +++ b/src/llama-model.cpp
59
  @@ -1,5 +1,7 @@
 
73
  const auto & split_mode = params.split_mode;
74
  const bool use_mlock = params.load_mode == LLAMA_LOAD_MODE_MLOCK || params.load_mode == LLAMA_LOAD_MODE_MMAP_MLOCK;
75
  const auto & tensor_split = params.tensor_split;
76
+ @@ -1821,7 +1825,24 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
77
  // per-tensor activation precision policy
78
  prec_policy.load(ml, *this);
79
 
80
  - ml.init_mappings(true, use_mlock ? &pimpl->mlock_mmaps : nullptr);
81
+ + if (getenv("ST_VERBOSE")) {
82
+ + FILE * f = fopen("/proc/self/status", "r");
83
+ + char line[256];
84
+ + while (f && fgets(line, sizeof line, f)) {
85
+ + if (!strncmp(line, "VmRSS:", 6)) { fprintf(stderr, "[st] init_mappings 前 VmRSS%s", line + 6); break; }
86
+ + }
87
+ + if (f) fclose(f);
88
+ + fprintf(stderr, " (prefetch=%d use_mmap=%d check_tensors=%d)\n",
89
+ + (int) st::will_page(arch_name().c_str()), (int) ml.use_mmap, (int) ml.check_tensors);
90
+ + }
91
+ +
92
  + // ST expert pager:開著分頁時**不能** prefetch。
93
  + // llama.cpp 預設 prefetch=true 會用 MAP_POPULATE 把整個 GGUF(2.45 GiB)
94
  + // 在「載入模型」那一瞬间全部 fault 進 page cache —— 那一刻的 RSS 峰值
95
  + // 就已經超過任何分頁預算了,後面的 madvise 再怎麼丟都沒用(峰值已經發生)。
96
  + // 這是量測方式/時機的坑,不是分頁機制本身有問題。
97
  + // 註意 st::set_config_from_env() 必須在這之前呼叫(見 load_tensors 開頭)。
98
+ + ml.init_mappings(!st::will_page(arch_name().c_str()), use_mlock ? &pimpl->mlock_mmaps : nullptr);
99
  pimpl->mappings.reserve(ml.mappings.size());
100
 
101
  // create the backend buffers
102
+ @@ -1956,6 +1977,16 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
103
+ });
104
+ }
105
+
106
+ + if (getenv("ST_VERBOSE")) {
107
+ + FILE * f = fopen("/proc/self/status", "r");
108
+ + char line[256];
109
+ + while (f && fgets(line, sizeof line, f)) {
110
+ + if (!strncmp(line, "VmRSS:", 6)) { fprintf(stderr, "[st] load_all_data 前 VmRSS%s", line + 6); break; }
111
+ + }
112
+ + if (f) fclose(f);
113
+ + fprintf(stderr, "\n");
114
+ + }
115
+ +
116
+ // load tensor data
117
+ for (auto & [ctx, buf_map] : ctx_buf_maps) {
118
+ if (!ml.load_all_data(ctx, buf_map, use_mlock ? &pimpl->mlock_mmaps : NULL, params.progress_callback, params.progress_callback_user_data)) {
119
+ @@ -1969,6 +2000,48 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
120
  }
121
  }
122
 
 
140
  + by_name.emplace(nm, t);
141
  + }
142
  +
143
+ + if (getenv("ST_VERBOSE")) {
144
+ + FILE * f = fopen("/proc/self/status", "r");
145
+ + char line[256];
146
+ + while (f && fgets(line, sizeof line, f)) {
147
+ + if (!strncmp(line, "VmRSS:", 6)) {
148
+ + fprintf(stderr, "[st] 模型載入完成後 VmRSS%s", line + 6);
149
+ + break;
150
+ + }
151
+ + }
152
+ + if (f) fclose(f);
153
+ + fprintf(stderr, "\n");
154
+ + }
155
+ +
156
+ + if (getenv("ST_VERBOSE") && !map_base) {
157
  + LLAMA_LOG_INFO("%s: [st] 沒有 mapping(mappings=%zu files=%zu),無法分頁\n",
158
  + __func__, pimpl->mappings.size(), ml.files.size());
159
  + }
 
167
 
168
  diff --git a/src/st_pager.cpp b/src/st_pager.cpp
169
  new file mode 100644
170
+ index 000000000..9cdbc0068
171
  --- /dev/null
172
  +++ b/src/st_pager.cpp
173
+ @@ -0,0 +1,579 @@
174
  +// ST expert pager —— 熱 expert 留在 RAM,冷 expert 丟回 SSD
175
  +//
176
  +// 設計取捨(為什麼不照抄 sddqwen35a3b_v01 的 pread arena):
 
292
  +
293
  +bool enabled() { return g_enabled && !g_slots.empty(); }
294
  +
295
+ +// 這個查詢會在 init_mappings() **之前**被呼叫(那時還沒載入權重,
296
+ +// register_model 當然還沒跑過),所以不能只看 enabled()。
297
+ +// 它只能回答「這個模型**預計**會不會被分頁」。
298
+ +bool will_page(const char * arch) {
299
+ + if (!g_enabled) return false;
300
+ + return arch && strcmp(arch, "smallthinker") == 0;
301
+ +}
302
+ +
303
+ +bool paging_active() { return enabled(); }
304
  +
305
  +// ------------------------------------------------------------------ 統計
306
  +
 
523
  +
524
  +// ------------------------------------------------------------------ 淘汰
525
  +
526
+ +// 診斷用:行程自己的 RSS(總量 / 匿名 / 檔案對映)。
527
+ +// 這是「分頁到底有沒生效」唯一可信的數字 —— 外部量測取樣太粗,
528
+ +// 而且 prefill 會讓 peak 失真。
529
+ +struct self_rss { size_t total = 0, anon = 0, file = 0; };
530
+ +
531
+ +static self_rss read_self_rss() {
532
+ + self_rss r;
533
+ + FILE * f = fopen("/proc/self/status", "r");
534
+ + if (f) {
535
+ + char line[256];
536
+ + while (fgets(line, sizeof line, f)) {
537
+ + unsigned long long v;
538
+ + if (sscanf(line, "VmRSS: %llu", &v) == 1) r.total = v * 1024;
539
+ + }
540
+ + fclose(f);
541
+ + }
542
+ + f = fopen("/proc/self/smaps_rollup", "r");
543
+ + if (f) {
544
+ + char line[256], k[64];
545
+ + unsigned long long v;
546
+ + while (fgets(line, sizeof line, f)) {
547
+ + if (sscanf(line, "%63[^:]: %llu", k, &v) != 2) continue;
548
+ + if (!strcmp(k, "Anonymous")) r.anon = v * 1024;
549
+ + else if (!strcmp(k, "Private_Clean") || !strcmp(k, "Private_Dirty")) r.file += v * 1024;
550
+ + }
551
+ + fclose(f);
552
+ + }
553
+ + return r;
554
+ +}
555
+ +
556
+ +
557
  +// ST_EVICT_MODE 診斷用(實測發現只做其中一步會導致輸出損毀,見 docs):
558
  +// both(預設)| madvise | fadvise | none
559
  +static int evict_mode() {
 
641
  + }
642
  +
643
  + if (g_verbose && !to_evict.empty()) {
644
+ + const self_rss rss = read_self_rss();
645
  + fprintf(stderr,
646
+ + "[st] sweep #%llu:丟 %zu 個 expert(%.1f MiB)→ 帳 resident %.1f MiB / budget %.1f MiB"
647
+ + " | 實際 RSS total %.1f MiB(anon %.1f + file %.1f)\n",
648
  + (unsigned long long) g_st.sweeps, to_evict.size(),
649
  + freed / (1024.0 * 1024.0),
650
  + (resident - freed) / (1024.0 * 1024.0),
651
+ + g_budget / (1024.0 * 1024.0),
652
+ + rss.total / (1024.0 * 1024.0),
653
+ + rss.anon / (1024.0 * 1024.0),
654
+ + rss.file / (1024.0 * 1024.0));
655
  + }
656
  +}
657
  +
 
753
 
754
  diff --git a/src/st_pager.h b/src/st_pager.h
755
  new file mode 100644
756
+ index 000000000..336e37de3
757
  --- /dev/null
758
  +++ b/src/st_pager.h
759
+ @@ -0,0 +1,84 @@
760
  +// ST expert pager —— 熱 expert 留在 RAM,冷 expert 丟回 SSD
761
  +//
762
  +// 對外介面。llama.cpp 端(llama-model.cpp / llama-graph.cpp)只include 這個檔案。
 
829
  + int layer, int n_expert, int n_expert_used);
830
  +
831
  +bool enabled();
832
+ +// 分頁有註冊成功 → 必須關掉 MAP_POPULATE(否則載入時就會 fault 整個檔案)
833
+ +bool paging_active();
834
+ +// 只看「有沒有打算分頁」,不看權重有沒有載入。init_mappings() 在載入之前,
835
+ +// 那時 register_model 還沒跑,所以只能靠 arch 判斷。
836
+ +bool will_page(const char * arch);
837
  +stats get_stats();
838
  +void dump_stats(const char * path);
839
  +
validate/ab-test.json CHANGED
@@ -1,5 +1,5 @@
1
  {
2
- "when": "2026-10-07T16:14:12+0200",
3
  "model": "/root/work/models/SmallThinker-4B-A0.6B-Instruct.Q4_K.gguf",
4
  "model_size_mib": 2508,
5
  "prompt": "Hello, who are you?",
@@ -12,342 +12,543 @@
12
  "ST_PAGER": "0"
13
  },
14
  "rc": 0,
15
- "seconds": 7.21,
16
  "peak": {
17
- "total_rss_gb": 2.3009,
18
- "anon_rss_gb": 0.1027,
19
- "file_rss_gb": 2.2911,
20
  "peak_swap_gb": 0.0,
21
- "hwm_rss_gb": 2.3009
22
  },
23
  "decode": {
24
- "total_rss_gb": 2.3009,
25
  "note": "prefill 後的峰值;prefill 會摸遍全部 expert,所以 peak 不代表 decode"
26
  },
27
  "rss_timeline_mib": [
28
  [
29
- 0.001,
30
- 0.9
31
  ],
32
  [
33
- 0.208,
34
- 75.9
35
  ],
36
  [
37
- 0.414,
38
- 76.5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
39
  ],
40
  [
41
- 0.624,
42
- 142.8
43
  ],
44
  [
45
- 0.836,
46
- 205.3
47
  ],
48
  [
49
- 1.054,
50
- 266.2
51
  ],
52
  [
53
- 1.275,
54
- 329.5
55
  ],
56
  [
57
- 1.498,
58
- 395.0
 
 
 
 
 
 
 
 
 
 
 
 
59
  ],
60
  [
61
- 1.715,
62
- 455.4
63
  ],
64
  [
65
- 1.943,
66
- 559.2
67
  ],
68
  [
69
- 2.179,
70
- 689.2
71
  ],
72
  [
73
- 2.408,
74
- 755.8
75
  ],
76
  [
77
- 2.633,
78
- 843.2
79
  ],
80
  [
81
- 2.858,
82
- 924.7
83
  ],
84
  [
85
- 3.085,
86
- 1003.3
87
  ],
88
  [
89
- 3.313,
90
- 1092.9
91
  ],
92
  [
93
- 3.545,
94
- 1178.2
95
  ],
96
  [
97
- 3.787,
98
- 1266.8
99
  ],
100
  [
101
- 4.02,
102
- 1360.0
103
  ],
104
  [
105
- 4.253,
106
- 1451.9
107
  ],
108
  [
109
- 4.489,
110
- 1543.3
111
  ],
112
  [
113
- 4.727,
114
- 1631.2
115
  ],
116
  [
117
- 4.967,
118
- 1720.4
119
  ],
120
  [
121
- 5.207,
122
- 1814.3
123
  ],
124
  [
125
- 5.449,
126
- 1908.3
127
  ],
128
  [
129
- 5.695,
130
- 2001.3
131
  ],
132
  [
133
- 5.95,
134
- 2094.1
135
  ],
136
  [
137
- 6.203,
138
- 2171.2
139
  ],
140
  [
141
- 6.474,
142
- 2256.3
143
  ],
144
  [
145
- 6.759,
146
- 2324.0
147
  ],
148
  [
149
- 7.073,
150
- 1201.9
 
 
 
 
 
 
 
 
 
 
 
 
151
  ]
152
  ],
153
  "io_delta": {},
154
- "timing_line": "[ Prompt: 8.2 t/s | Generation: 30.6 t/s ]",
155
  "output": "Hello! 😊 I'm DeepSeek-R1, your friendly AI assistant. I'm here to help with all kinds of questions, learning, and creative tasks",
156
  "output_matches_baseline": true,
157
  "swap_zero": true,
158
  "rc_ok": true
159
  },
160
- "budget512mb": {
161
  "env": {
162
  "ST_PAGER": "1",
163
- "ST_RAM_BUDGET_MB": "512",
164
  "ST_RESERVE_MB": "200"
165
  },
166
  "rc": 0,
167
- "seconds": 9.23,
168
  "peak": {
169
- "total_rss_gb": 2.5559,
170
- "anon_rss_gb": 0.1035,
171
- "file_rss_gb": 2.5503,
172
  "peak_swap_gb": 0.0,
173
- "hwm_rss_gb": 2.5559
174
  },
175
  "decode": {
176
- "total_rss_gb": 2.5559,
177
  "note": "prefill 後的峰值;prefill 會摸遍全部 expert,所以 peak 不代表 decode"
178
  },
179
  "rss_timeline_mib": [
180
  [
181
  0.0,
182
- 0.6
183
  ],
184
  [
185
  0.205,
186
  75.9
187
  ],
188
  [
189
- 0.409,
190
- 71.3
191
  ],
192
  [
193
- 0.614,
194
- 121.4
195
  ],
196
  [
197
- 0.82,
198
- 190.4
199
  ],
200
  [
201
- 1.026,
202
- 258.7
203
  ],
204
  [
205
- 1.233,
206
- 326.7
207
  ],
208
  [
209
- 1.44,
210
- 397.2
211
  ],
212
  [
213
- 1.649,
214
- 468.3
215
  ],
216
  [
217
- 1.858,
218
- 538.3
219
  ],
220
  [
221
- 2.068,
222
- 608.8
223
  ],
224
  [
225
- 2.278,
226
- 680.3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
227
  ],
228
  [
229
- 2.49,
230
- 751.8
 
 
 
 
 
 
 
 
231
  ],
232
  [
233
- 2.702,
234
- 824.6
235
  ],
236
  [
237
- 2.916,
238
- 895.9
239
  ],
240
  [
241
- 3.13,
242
- 968.6
243
  ],
244
  [
245
- 3.346,
246
- 1041.7
247
  ],
248
  [
249
- 3.563,
250
- 1115.7
251
  ],
252
  [
253
- 3.78,
254
- 1187.2
255
  ],
256
  [
257
- 3.998,
258
- 1262.2
259
  ],
260
  [
261
- 4.216,
262
- 1335.2
263
  ],
264
  [
265
- 4.436,
266
- 1409.3
267
  ],
268
  [
269
- 4.659,
270
- 1484.6
271
  ],
272
  [
273
- 4.881,
274
- 1558.8
275
  ],
276
  [
277
- 5.103,
278
- 1633.1
279
  ],
280
  [
281
- 5.326,
282
- 1706.4
283
  ],
284
  [
285
- 5.551,
286
- 1780.9
287
  ],
288
  [
289
- 5.777,
290
- 1856.4
291
  ],
292
  [
293
- 6.003,
294
- 1931.9
295
  ],
296
  [
297
- 6.239,
298
- 2009.0
299
  ],
300
  [
301
- 6.469,
302
- 2085.5
303
  ],
304
  [
305
- 6.7,
306
- 2161.5
307
  ],
308
  [
309
- 6.933,
310
- 2239.5
311
  ],
312
  [
313
- 7.182,
314
- 2317.5
315
  ],
316
  [
317
- 7.418,
318
- 2400.6
319
  ],
320
  [
321
- 7.657,
322
- 2480.1
323
  ],
324
  [
325
- 7.895,
326
- 2558.6
327
  ],
328
  [
329
- 8.128,
330
- 931.1
331
  ],
332
  [
333
- 8.358,
334
- 935.8
335
  ],
336
  [
337
- 8.584,
338
- 1119.4
339
  ],
340
  [
341
- 8.825,
342
- 1193.9
343
  ],
344
  [
345
- 9.057,
346
- 1193.3
 
 
 
 
 
 
 
 
347
  ]
348
  ],
349
  "io_delta": {},
350
- "timing_line": "[ Prompt: 128.4 t/s | Generation: 44.2 t/s ]",
351
  "output": "Hello! 😊 I'm DeepSeek-R1, your friendly AI assistant. I'm here to help with all kinds of questions, learning, and creative tasks",
352
  "pager": {
353
  "routing_calls": 8448,
@@ -360,24 +561,221 @@
360
  "resident_bytes": 327075840,
361
  "budget_bytes": 327155712,
362
  "hit_rate": 0.5514,
363
- "proc_self_read_bytes": 2630213632,
364
  "proc_self_rchar": 11947223
365
  },
366
  "output_matches_baseline": true,
367
  "swap_zero": true,
368
  "rc_ok": true,
369
- "rss_vs_baseline": 0.255,
370
- "decode_rss_vs_baseline": 0.255,
371
- "decode_rss_lower_than_baseline": false,
372
- "rss_lower_than_baseline": false,
373
- "ssd_reads_mib": 2508.4,
374
  "cold_weights_really_read_from_ssd": true,
375
  "pager_hit_rate": 0.5514,
376
  "pager_evictions": 4572,
377
  "pager_misses": 3701,
378
  "pager_resident_mib": 311.9,
379
  "pager_budget_mib": 312.0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
380
  }
381
  },
382
- "all_passed": false
383
  }
 
1
  {
2
+ "when": "2026-10-07T16:26:13+0200",
3
  "model": "/root/work/models/SmallThinker-4B-A0.6B-Instruct.Q4_K.gguf",
4
  "model_size_mib": 2508,
5
  "prompt": "Hello, who are you?",
 
12
  "ST_PAGER": "0"
13
  },
14
  "rc": 0,
15
+ "seconds": 9.03,
16
  "peak": {
17
+ "total_rss_gb": 2.5644,
18
+ "anon_rss_gb": 0.1033,
19
+ "file_rss_gb": 2.5587,
20
  "peak_swap_gb": 0.0,
21
+ "hwm_rss_gb": 2.5644
22
  },
23
  "decode": {
24
+ "total_rss_gb": 2.5644,
25
  "note": "prefill 後的峰值;prefill 會摸遍全部 expert,所以 peak 不代表 decode"
26
  },
27
  "rss_timeline_mib": [
28
  [
29
+ 0.0,
30
+ 0.6
31
  ],
32
  [
33
+ 0.207,
34
+ 72.8
35
  ],
36
  [
37
+ 0.413,
38
+ 71.3
39
+ ],
40
+ [
41
+ 0.618,
42
+ 117.7
43
+ ],
44
+ [
45
+ 0.824,
46
+ 187.4
47
+ ],
48
+ [
49
+ 1.03,
50
+ 255.4
51
+ ],
52
+ [
53
+ 1.238,
54
+ 326.7
55
  ],
56
  [
57
+ 1.451,
58
+ 396.2
59
  ],
60
  [
61
+ 1.66,
62
+ 467.0
63
  ],
64
  [
65
+ 1.87,
66
+ 537.0
67
  ],
68
  [
69
+ 2.08,
70
+ 608.5
71
  ],
72
  [
73
+ 2.291,
74
+ 680.3
75
+ ],
76
+ [
77
+ 2.504,
78
+ 752.1
79
+ ],
80
+ [
81
+ 2.723,
82
+ 826.1
83
+ ],
84
+ [
85
+ 2.938,
86
+ 899.4
87
  ],
88
  [
89
+ 3.154,
90
+ 971.1
91
  ],
92
  [
93
+ 3.37,
94
+ 1044.4
95
  ],
96
  [
97
+ 3.586,
98
+ 1118.2
99
  ],
100
  [
101
+ 3.803,
102
+ 1191.2
103
  ],
104
  [
105
+ 4.022,
106
+ 1264.2
107
  ],
108
  [
109
+ 4.241,
110
+ 1338.3
111
  ],
112
  [
113
+ 4.462,
114
+ 1412.8
115
  ],
116
  [
117
+ 4.695,
118
+ 1489.0
119
  ],
120
  [
121
+ 4.924,
122
+ 1565.8
123
  ],
124
  [
125
+ 5.154,
126
+ 1642.8
127
  ],
128
  [
129
+ 5.382,
130
+ 1718.1
131
  ],
132
  [
133
+ 5.608,
134
+ 1795.6
135
  ],
136
  [
137
+ 5.835,
138
+ 1871.4
139
  ],
140
  [
141
+ 6.063,
142
+ 1947.9
143
  ],
144
  [
145
+ 6.291,
146
+ 2024.5
147
  ],
148
  [
149
+ 6.522,
150
+ 2100.2
151
  ],
152
  [
153
+ 6.755,
154
+ 2177.5
155
  ],
156
  [
157
+ 6.987,
158
+ 2255.8
159
  ],
160
  [
161
+ 7.221,
162
+ 2334.3
163
  ],
164
  [
165
+ 7.459,
166
+ 2411.6
167
  ],
168
  [
169
+ 7.696,
170
+ 2487.6
171
  ],
172
  [
173
+ 7.934,
174
+ 2566.1
175
  ],
176
  [
177
+ 8.173,
178
+ 2620.2
179
+ ],
180
+ [
181
+ 8.414,
182
+ 2626.0
183
+ ],
184
+ [
185
+ 8.665,
186
+ 2626.0
187
+ ],
188
+ [
189
+ 8.918,
190
+ 2626.0
191
  ]
192
  ],
193
  "io_delta": {},
194
+ "timing_line": "[ Prompt: 198.1 t/s | Generation: 55.6 t/s ]",
195
  "output": "Hello! 😊 I'm DeepSeek-R1, your friendly AI assistant. I'm here to help with all kinds of questions, learning, and creative tasks",
196
  "output_matches_baseline": true,
197
  "swap_zero": true,
198
  "rc_ok": true
199
  },
200
+ "budget1024mb": {
201
  "env": {
202
  "ST_PAGER": "1",
203
+ "ST_RAM_BUDGET_MB": "1024",
204
  "ST_RESERVE_MB": "200"
205
  },
206
  "rc": 0,
207
+ "seconds": 7.6,
208
  "peak": {
209
+ "total_rss_gb": 1.3011,
210
+ "anon_rss_gb": 0.103,
211
+ "file_rss_gb": 1.3028,
212
  "peak_swap_gb": 0.0,
213
+ "hwm_rss_gb": 1.3011
214
  },
215
  "decode": {
216
+ "total_rss_gb": 1.3011,
217
  "note": "prefill 後的峰值;prefill 會摸遍全部 expert,所以 peak 不代表 decode"
218
  },
219
  "rss_timeline_mib": [
220
  [
221
  0.0,
222
+ 0.5
223
  ],
224
  [
225
  0.205,
226
  75.9
227
  ],
228
  [
229
+ 0.41,
230
+ 72.6
231
  ],
232
  [
233
+ 0.623,
234
+ 135.2
235
  ],
236
  [
237
+ 0.831,
238
+ 189.9
239
  ],
240
  [
241
+ 1.041,
242
+ 247.8
243
  ],
244
  [
245
+ 1.251,
246
+ 306.1
247
  ],
248
  [
249
+ 1.463,
250
+ 364.3
251
  ],
252
  [
253
+ 1.676,
254
+ 424.6
255
  ],
256
  [
257
+ 1.892,
258
+ 482.6
259
  ],
260
  [
261
+ 2.117,
262
+ 604.1
263
  ],
264
  [
265
+ 2.341,
266
+ 688.7
267
+ ],
268
+ [
269
+ 2.564,
270
+ 746.2
271
+ ],
272
+ [
273
+ 2.787,
274
+ 828.9
275
+ ],
276
+ [
277
+ 3.025,
278
+ 913.8
279
+ ],
280
+ [
281
+ 3.254,
282
+ 994.6
283
+ ],
284
+ [
285
+ 3.481,
286
+ 1080.7
287
+ ],
288
+ [
289
+ 3.711,
290
+ 1166.4
291
+ ],
292
+ [
293
+ 3.941,
294
+ 1236.5
295
+ ],
296
+ [
297
+ 4.18,
298
+ 1228.6
299
+ ],
300
+ [
301
+ 4.416,
302
+ 1241.3
303
+ ],
304
+ [
305
+ 4.65,
306
+ 1204.6
307
+ ],
308
+ [
309
+ 4.888,
310
+ 1212.4
311
+ ],
312
+ [
313
+ 5.128,
314
+ 1250.3
315
+ ],
316
+ [
317
+ 5.368,
318
+ 1211.6
319
+ ],
320
+ [
321
+ 5.605,
322
+ 1218.1
323
+ ],
324
+ [
325
+ 5.837,
326
+ 1200.6
327
+ ],
328
+ [
329
+ 6.07,
330
+ 1248.6
331
+ ],
332
+ [
333
+ 6.304,
334
+ 1271.8
335
+ ],
336
+ [
337
+ 6.55,
338
+ 1315.0
339
+ ],
340
+ [
341
+ 6.794,
342
+ 1322.1
343
+ ],
344
+ [
345
+ 7.043,
346
+ 1329.8
347
+ ],
348
+ [
349
+ 7.304,
350
+ 1326.6
351
+ ],
352
+ [
353
+ 7.553,
354
+ 475.5
355
+ ]
356
+ ],
357
+ "io_delta": {},
358
+ "timing_line": "[ Prompt: 8.1 t/s | Generation: 24.2 t/s ]",
359
+ "output": "Hello! 😊 I'm DeepSeek-R1, your friendly AI assistant. I'm here to help with all kinds of questions, learning, and creative tasks",
360
+ "pager": {
361
+ "routing_calls": 8448,
362
+ "expert_touches": 64544,
363
+ "hits": 37015,
364
+ "misses": 2273,
365
+ "evictions": 2894,
366
+ "evicted_bytes": 6197382144,
367
+ "sweeps": 864,
368
+ "resident_bytes": 863972352,
369
+ "budget_bytes": 864026624,
370
+ "hit_rate": 0.5735,
371
+ "proc_self_read_bytes": 2429562880,
372
+ "proc_self_rchar": 11947223
373
+ },
374
+ "output_matches_baseline": true,
375
+ "swap_zero": true,
376
+ "rc_ok": true,
377
+ "rss_vs_baseline": -1.2633,
378
+ "decode_rss_vs_baseline": -1.2633,
379
+ "decode_rss_lower_than_baseline": true,
380
+ "rss_lower_than_baseline": true,
381
+ "ssd_reads_mib": 2317.0,
382
+ "cold_weights_really_read_from_ssd": true,
383
+ "pager_hit_rate": 0.5735,
384
+ "pager_evictions": 2894,
385
+ "pager_misses": 2273,
386
+ "pager_resident_mib": 823.9,
387
+ "pager_budget_mib": 824.0
388
+ },
389
+ "budget512mb": {
390
+ "env": {
391
+ "ST_PAGER": "1",
392
+ "ST_RAM_BUDGET_MB": "512",
393
+ "ST_RESERVE_MB": "200"
394
+ },
395
+ "rc": 0,
396
+ "seconds": 7.79,
397
+ "peak": {
398
+ "total_rss_gb": 0.8316,
399
+ "anon_rss_gb": 0.103,
400
+ "file_rss_gb": 0.8276,
401
+ "peak_swap_gb": 0.0,
402
+ "hwm_rss_gb": 0.8316
403
+ },
404
+ "decode": {
405
+ "total_rss_gb": 0.8316,
406
+ "note": "prefill 後的峰值;prefill 會摸遍全部 expert,所以 peak 不代表 decode"
407
+ },
408
+ "rss_timeline_mib": [
409
+ [
410
+ 0.0,
411
+ 0.6
412
  ],
413
  [
414
+ 0.205,
415
+ 75.8
416
+ ],
417
+ [
418
+ 0.409,
419
+ 76.5
420
+ ],
421
+ [
422
+ 0.627,
423
+ 141.0
424
  ],
425
  [
426
+ 0.834,
427
+ 202.7
428
  ],
429
  [
430
+ 1.043,
431
+ 261.2
432
  ],
433
  [
434
+ 1.252,
435
+ 319.1
436
  ],
437
  [
438
+ 1.464,
439
+ 372.0
440
  ],
441
  [
442
+ 1.679,
443
+ 425.2
444
  ],
445
  [
446
+ 1.894,
447
+ 479.1
448
  ],
449
  [
450
+ 2.113,
451
+ 602.5
452
  ],
453
  [
454
+ 2.334,
455
+ 686.2
456
  ],
457
  [
458
+ 2.559,
459
+ 721.0
460
  ],
461
  [
462
+ 2.781,
463
+ 694.2
464
  ],
465
  [
466
+ 3.008,
467
+ 729.6
468
  ],
469
  [
470
+ 3.231,
471
+ 711.1
472
  ],
473
  [
474
+ 3.456,
475
+ 740.1
476
  ],
477
  [
478
+ 3.69,
479
+ 738.7
480
  ],
481
  [
482
+ 3.92,
483
+ 721.3
484
  ],
485
  [
486
+ 4.143,
487
+ 712.6
488
  ],
489
  [
490
+ 4.367,
491
+ 711.4
492
  ],
493
  [
494
+ 4.595,
495
+ 728.2
496
  ],
497
  [
498
+ 4.825,
499
+ 728.1
500
  ],
501
  [
502
+ 5.049,
503
+ 726.2
504
  ],
505
  [
506
+ 5.274,
507
+ 724.0
508
  ],
509
  [
510
+ 5.5,
511
+ 722.1
512
  ],
513
  [
514
+ 5.749,
515
+ 729.2
516
  ],
517
  [
518
+ 5.981,
519
+ 711.0
520
  ],
521
  [
522
+ 6.208,
523
+ 748.3
524
  ],
525
  [
526
+ 6.436,
527
+ 770.3
528
  ],
529
  [
530
+ 6.667,
531
+ 817.9
532
  ],
533
  [
534
+ 6.898,
535
+ 837.3
536
  ],
537
  [
538
+ 7.127,
539
+ 843.5
540
+ ],
541
+ [
542
+ 7.356,
543
+ 840.8
544
+ ],
545
+ [
546
+ 7.584,
547
+ 844.6
548
  ]
549
  ],
550
  "io_delta": {},
551
+ "timing_line": "[ Prompt: 7.9 t/s | Generation: 23.3 t/s ]",
552
  "output": "Hello! 😊 I'm DeepSeek-R1, your friendly AI assistant. I'm here to help with all kinds of questions, learning, and creative tasks",
553
  "pager": {
554
  "routing_calls": 8448,
 
561
  "resident_bytes": 327075840,
562
  "budget_bytes": 327155712,
563
  "hit_rate": 0.5514,
564
+ "proc_self_read_bytes": 2427265024,
565
  "proc_self_rchar": 11947223
566
  },
567
  "output_matches_baseline": true,
568
  "swap_zero": true,
569
  "rc_ok": true,
570
+ "rss_vs_baseline": -1.7328,
571
+ "decode_rss_vs_baseline": -1.7328,
572
+ "decode_rss_lower_than_baseline": true,
573
+ "rss_lower_than_baseline": true,
574
+ "ssd_reads_mib": 2314.8,
575
  "cold_weights_really_read_from_ssd": true,
576
  "pager_hit_rate": 0.5514,
577
  "pager_evictions": 4572,
578
  "pager_misses": 3701,
579
  "pager_resident_mib": 311.9,
580
  "pager_budget_mib": 312.0
581
+ },
582
+ "budget256mb": {
583
+ "env": {
584
+ "ST_PAGER": "1",
585
+ "ST_RAM_BUDGET_MB": "256",
586
+ "ST_RESERVE_MB": "200"
587
+ },
588
+ "rc": 0,
589
+ "seconds": 7.97,
590
+ "peak": {
591
+ "total_rss_gb": 0.5875,
592
+ "anon_rss_gb": 0.1029,
593
+ "file_rss_gb": 0.5864,
594
+ "peak_swap_gb": 0.0,
595
+ "hwm_rss_gb": 0.5875
596
+ },
597
+ "decode": {
598
+ "total_rss_gb": 0.5875,
599
+ "note": "prefill 後的峰值;prefill 會摸遍全部 expert,所以 peak 不代表 decode"
600
+ },
601
+ "rss_timeline_mib": [
602
+ [
603
+ 0.0,
604
+ 0.4
605
+ ],
606
+ [
607
+ 0.206,
608
+ 76.1
609
+ ],
610
+ [
611
+ 0.411,
612
+ 72.7
613
+ ],
614
+ [
615
+ 0.618,
616
+ 140.0
617
+ ],
618
+ [
619
+ 0.826,
620
+ 192.9
621
+ ],
622
+ [
623
+ 1.034,
624
+ 214.1
625
+ ],
626
+ [
627
+ 1.244,
628
+ 233.8
629
+ ],
630
+ [
631
+ 1.456,
632
+ 250.4
633
+ ],
634
+ [
635
+ 1.667,
636
+ 271.8
637
+ ],
638
+ [
639
+ 1.881,
640
+ 284.4
641
+ ],
642
+ [
643
+ 2.094,
644
+ 371.8
645
+ ],
646
+ [
647
+ 2.312,
648
+ 482.8
649
+ ],
650
+ [
651
+ 2.531,
652
+ 471.7
653
+ ],
654
+ [
655
+ 2.753,
656
+ 439.1
657
+ ],
658
+ [
659
+ 2.976,
660
+ 468.5
661
+ ],
662
+ [
663
+ 3.2,
664
+ 444.3
665
+ ],
666
+ [
667
+ 3.425,
668
+ 486.4
669
+ ],
670
+ [
671
+ 3.649,
672
+ 478.2
673
+ ],
674
+ [
675
+ 3.883,
676
+ 459.5
677
+ ],
678
+ [
679
+ 4.105,
680
+ 448.2
681
+ ],
682
+ [
683
+ 4.33,
684
+ 440.3
685
+ ],
686
+ [
687
+ 4.553,
688
+ 448.0
689
+ ],
690
+ [
691
+ 4.784,
692
+ 452.0
693
+ ],
694
+ [
695
+ 5.024,
696
+ 452.2
697
+ ],
698
+ [
699
+ 5.244,
700
+ 438.6
701
+ ],
702
+ [
703
+ 5.465,
704
+ 436.2
705
+ ],
706
+ [
707
+ 5.686,
708
+ 438.4
709
+ ],
710
+ [
711
+ 5.909,
712
+ 484.5
713
+ ],
714
+ [
715
+ 6.139,
716
+ 460.0
717
+ ],
718
+ [
719
+ 6.365,
720
+ 489.1
721
+ ],
722
+ [
723
+ 6.596,
724
+ 558.0
725
+ ],
726
+ [
727
+ 6.834,
728
+ 576.9
729
+ ],
730
+ [
731
+ 7.071,
732
+ 592.4
733
+ ],
734
+ [
735
+ 7.325,
736
+ 594.1
737
+ ],
738
+ [
739
+ 7.561,
740
+ 599.6
741
+ ],
742
+ [
743
+ 7.798,
744
+ 594.5
745
+ ]
746
+ ],
747
+ "io_delta": {},
748
+ "timing_line": "[ Prompt: 7.9 t/s | Generation: 20.3 t/s ]",
749
+ "output": "Hello! 😊 I'm DeepSeek-R1, your friendly AI assistant. I'm here to help with all kinds of questions, learning, and creative tasks",
750
+ "pager": {
751
+ "routing_calls": 8448,
752
+ "expert_touches": 64544,
753
+ "hits": 34660,
754
+ "misses": 4628,
755
+ "evictions": 5625,
756
+ "evicted_bytes": 12070591488,
757
+ "sweeps": 1280,
758
+ "resident_bytes": 57701376,
759
+ "budget_bytes": 58720256,
760
+ "hit_rate": 0.537,
761
+ "proc_self_read_bytes": 2428882944,
762
+ "proc_self_rchar": 11947223
763
+ },
764
+ "output_matches_baseline": true,
765
+ "swap_zero": true,
766
+ "rc_ok": true,
767
+ "rss_vs_baseline": -1.9769,
768
+ "decode_rss_vs_baseline": -1.9769,
769
+ "decode_rss_lower_than_baseline": true,
770
+ "rss_lower_than_baseline": true,
771
+ "ssd_reads_mib": 2316.4,
772
+ "cold_weights_really_read_from_ssd": true,
773
+ "pager_hit_rate": 0.537,
774
+ "pager_evictions": 5625,
775
+ "pager_misses": 4628,
776
+ "pager_resident_mib": 55.0,
777
+ "pager_budget_mib": 56.0
778
  }
779
  },
780
+ "all_passed": true
781
  }