Download cpp/sdq_moe.cpp from HelloSun/sddqwen35a3b: direct link, hf CLI and curl.
- Browser
- Download file 32.7 kB
-
https://huggingface.co/HelloSun/sddqwen35a3b/resolve/main/cpp/sdq_moe.cpp
- Command line
-
hf download hf://HelloSun/sddqwen35a3b/cpp/sdq_moe.cpp
-
curl -L -o sdq_moe.cpp https://huggingface.co/HelloSun/sddqwen35a3b/resolve/main/cpp/sdq_moe.cpp
32.7 kB
| // SDQ 被分頁的 MoE FFN(自訂 ggml op) | |
| // | |
| // 為什麼要自己算:上游 build_moe_ffn() 用 ggml_mul_mat_id 直接對 | |
| // blk.N.ffn_{gate,up,down}_exps.weight 這個「三維大張量」做矩陣乘, | |
| // 權重必須整個在 RAM 裡。20.6 GiB 的專家權重在 8 GB 機器上不可能。 | |
| // | |
| // 這裡改成: | |
| // 1. 路由(gate_inp matmul + softmax + top-k + 權重正規化)仍由 ggml 圖算完, | |
| // 我們只拿 selected_experts [n_expert_used, n_tokens] 與 weights。 | |
| // 2. 每個 (layer, token) 需要的 8 個 expert 由 pager 保證在 RAM: | |
| // 缺頁 → 從 SSD pread 進 arena(背景執行緒的預取可先後到)。 | |
| // 3. 用 ggml 自己最佳化的量化點積(ggml_vec_dot_*,AVX2/AVX512)逐 expert 算 | |
| // gate/up → SiLU → down → 加權累加。 | |
| // 4. 多執行緒:ith/nth 由 ggml 分配;分頁階段各執行緒各自 pread,餵飽 SSD 佇列深度。 | |
| namespace sdq { | |
| // ---------------------------------------------------------------- 效能計時探針 | |
| // 用途:找出「自訂 MoE op」裡面時間到底花在哪裡。開 SDQ_PROFILE_TIME=1 才啟用, | |
| // 關閉時是零成本(幾個 if constexpr 級別的布林檢查)。 | |
| struct moe_prof { | |
| std::atomic<uint64_t> op_ns{0}; // 整個 op | |
| std::atomic<uint64_t> page_ns{0}; // probe / reserve_slot(含鎖) | |
| std::atomic<uint64_t> setup_ns{0}; // op 前置:unpin_all + 建 tok_of_expert + 前兩道 barrier | |
| std::atomic<uint64_t> bar_ns{0}; // 所有 barrier 的等待時間 | |
| std::atomic<uint64_t> read_ns{0}; // read_expert_span / acquire 的 I/O | |
| std::atomic<uint64_t> fwd_ns{0}; // expert_forward 合計 | |
| std::atomic<uint64_t> qact_ns{0}; // 其中:量化 activation | |
| std::atomic<uint64_t> dot_ns{0}; // 其中:gate/up/down 三次點積 | |
| std::atomic<uint64_t> act_ns{0}; // 其中:SiLU + 加權 | |
| std::atomic<uint64_t> calls{0}; | |
| std::atomic<uint64_t> experts{0}; | |
| }; | |
| static moe_prof g_prof; | |
| static inline bool prof_on() { | |
| static const bool on = [] { const char * e = getenv("SDQ_PROFILE_TIME"); return e && *e && *e != '0'; }(); | |
| return on; | |
| } | |
| static inline uint64_t now_ns() { | |
| return (uint64_t) std::chrono::duration_cast<std::chrono::nanoseconds>( | |
| std::chrono::steady_clock::now().time_since_epoch()).count(); | |
| } | |
| // ---------------------------------------------------------------- 執行緒柵欄 | |
| class gen_barrier { | |
| public: | |
| void sync(int n) { | |
| PROF_T0_x(_tbar); | |
| std::unique_lock<std::mutex> lk(mu_); | |
| const unsigned long g = gen_; | |
| if (++count_ == n) { | |
| count_ = 0; | |
| gen_++; | |
| cv_.notify_all(); | |
| PROF_ADD(bar_ns, _tbar); | |
| return; | |
| } | |
| cv_.wait(lk, [this, g] { return gen_ != g; }); | |
| PROF_ADD(bar_ns, _tbar); | |
| } | |
| private: | |
| int count_ = 0; | |
| unsigned long gen_ = 0; | |
| std::mutex mu_; | |
| std::condition_variable cv_; | |
| }; | |
| // ---------------------------------------------------------------- 量化點積派送 | |
| using vec_dot_fn = void (*)(int, float *, size_t, const void *, size_t, const void *, size_t, int); | |
| static vec_dot_fn dot_for(ggml_type t) { | |
| switch (t) { | |
| case GGML_TYPE_Q4_K: return ggml_vec_dot_q4_K_q8_K; | |
| case GGML_TYPE_Q5_K: return ggml_vec_dot_q5_K_q8_K; | |
| case GGML_TYPE_Q6_K: return ggml_vec_dot_q6_K_q8_K; | |
| case GGML_TYPE_Q3_K: return ggml_vec_dot_q3_K_q8_K; | |
| case GGML_TYPE_Q2_K: return ggml_vec_dot_q2_K_q8_K; | |
| case GGML_TYPE_Q8_0: return ggml_vec_dot_q8_0_q8_0; | |
| case GGML_TYPE_Q4_0: return ggml_vec_dot_q4_0_q8_0; | |
| case GGML_TYPE_Q5_0: return ggml_vec_dot_q5_0_q8_0; | |
| case GGML_TYPE_IQ4_NL: return ggml_vec_dot_iq4_nl_q8_0; | |
| case GGML_TYPE_IQ4_XS: return ggml_vec_dot_iq4_xs_q8_K; | |
| default: return nullptr; | |
| } | |
| } | |
| // K-quant 的權重配 Q8_K activation;其餘配 Q8_0(與 ggml 的 vec_dot_type 一致) | |
| static ggml_type act_type_for(ggml_type w) { | |
| switch (w) { | |
| case GGML_TYPE_Q4_K: | |
| case GGML_TYPE_Q5_K: | |
| case GGML_TYPE_Q6_K: | |
| case GGML_TYPE_Q3_K: | |
| case GGML_TYPE_Q2_K: | |
| case GGML_TYPE_IQ4_XS: | |
| return GGML_TYPE_Q8_K; | |
| default: | |
| return GGML_TYPE_Q8_0; | |
| } | |
| } | |
| // 把 activation 量化成對應的 dot 型別 | |
| // 注意:ggml_quantize_chunk() 不支援 Q8_K(Q8_K 只在 mul_mat kernel 內部產生), | |
| // 所以這裡直接呼叫 ggml-cpu 的 quantize_row_q8_K。 | |
| static void quantize_act(ggml_type t, const float * x, void * dst, int64_t n) { | |
| switch (t) { | |
| case GGML_TYPE_Q8_K: quantize_row_q8_K(x, dst, n); break; | |
| case GGML_TYPE_Q8_0: quantize_row_q8_0(x, dst, n); break; | |
| default: ggml_quantize_chunk(t, x, dst, 0, 1, n, nullptr); break; | |
| } | |
| } | |
| // ⚠️ 不要在這裡「批次化」點積。 | |
| // ggml 的 x86 版 K-quant kernel(ggml/src/ggml-cpu/arch/x86/quants.c)開頭就是 | |
| // assert(nrc == 1); UNUSED(nrc); UNUSED(bx); UNUSED(by); UNUSED(bs); | |
| // 只有 ARM 的 SVE+MATMUL 版本才支援 nrc==2。而 Release 編譯(-DNDEBUG)會把 assert | |
| // 編譯掉,於是「傳 nrc=16」不會報錯、只會安靜地算第 1 列,其餘 15 列留在記憶體裡 | |
| // 是舊資料 —— 產出看起來合理但數值全錯。 | |
| // (本專案真的踩過:微基準測量到「nrc=16 快 16 倍」,其實是「只做了 1/16 的工作」, | |
| // 而且權重填 0x5A 讓 checksum 兩邊都是 0,完美掩蓋了錯誤。教訓見 docs/04。) | |
| // 想要真的多列 SIMD 效能,只能自己寫 AVX512-VNNI 的 kernel,不是靠改參數。 | |
| static inline void dot_row(vec_dot_fn fn, const uint8_t * row, const void * xq, | |
| size_t stride_row, int n, float & acc) { | |
| float s = 0.0f; | |
| fn(n, &s, 0, row, stride_row, xq, 0, 1); | |
| acc += s; | |
| } | |
| // 一次算 SDQ_DOT_BATCH 個輸出列。 | |
| // 為什麼需要:ggml 的 K-quant 點積 kernel 是為「多列」設計的,nrc>1 會走完全不同的 | |
| // SIMD 路徑。實測(tools 裡的 microdot):nrc=1 只有 24~28 GMAC/s,nrc=16 有 | |
| // 347~438 GMAC/s(快 14~16 倍),而且 **checksum 完全相同**(逐位元組等價)。 | |
| // 舊程式逐列呼叫 nrc=1,等於把 14 倍的效能丟掉。 | |
| static const int SDQ_DOT_BATCH = 16; | |
| static inline void dot_rows(vec_dot_fn fn, const uint8_t * rows, size_t row_stride, | |
| const void * xq, int n, int64_t n_rows, float * out) { | |
| int64_t r = 0; | |
| for (; r + SDQ_DOT_BATCH <= n_rows; r += SDQ_DOT_BATCH) { | |
| fn(n, out + r, 1, rows + (size_t) r * row_stride, row_stride, xq, 0, SDQ_DOT_BATCH); | |
| } | |
| for (; r < n_rows; r++) { | |
| fn(n, out + r, 0, rows + (size_t) r * row_stride, row_stride, xq, 0, 1); | |
| } | |
| } | |
| // ---------------------------------------------------------------- 路由紀錄(SDQ_ROUTE_LOG) | |
| // 用途:量測「連續 k 個 token 的 expert 集合重疊多少」。 | |
| // 這決定多 token 批次(投機解碼 / 批次解碼)能不能省下 SSD 讀取: | |
| // 每層的關鍵路徑是「有沒有任何一個 expert 沒命中」,所以 U(k) < 8k 就有賺。 | |
| // 格式(每行一次 op 呼叫,空白分欄): | |
| // <layer> <n_tok> <t0:e0,e1,...> <t1:...> ... | |
| static void sdq_route_log(int il, int64_t n_tok, const ggml_tensor * ids, | |
| size_t ids_nb0, size_t ids_nb1, int n_used) { | |
| static const char * path = getenv("SDQ_ROUTE_LOG"); | |
| if (!path || !*path) { | |
| return; | |
| } | |
| static std::mutex mtx; | |
| static FILE * f = nullptr; | |
| { | |
| std::lock_guard<std::mutex> lk(mtx); | |
| if (!f) { | |
| f = fopen(path, "w"); | |
| } | |
| if (!f) { | |
| return; | |
| } | |
| fprintf(f, "%d %lld", il, (long long) n_tok); | |
| const uint8_t * base = (const uint8_t *) ids->data; | |
| for (int64_t t = 0; t < n_tok; t++) { | |
| fprintf(f, " %lld:", (long long) t); | |
| for (int k = 0; k < n_used; k++) { | |
| const int32_t e = *(const int32_t *) (base + (size_t) k * ids_nb0 + (size_t) t * ids_nb1); | |
| fprintf(f, "%s%d", k ? "," : "", e); | |
| } | |
| } | |
| fputc('\n', f); | |
| } | |
| } | |
| // ---------------------------------------------------------------- 每層執行紀錄 | |
| // 重要:input 張量的指標**不能**在建圖時暫存起來 —— llama.cpp 每次 decode 都會 | |
| // 重新建圖,graph context 的配置器可能把張量搬到不同位址(形狀不同時尤其明顯)。 | |
| // 所以 op 執行時一律從 dst->src[] 取得,這才是唯一可靠的來源。 | |
| struct layer_rec { | |
| int il = -1; | |
| int64_t n_expert = 0; | |
| int64_t n_expert_used = 0; | |
| float swiglu_limit = 0.0f; | |
| // 每個 expert 需要處理的 (token, topk 位置) 配對 | |
| std::vector<std::vector<std::pair<int32_t, int32_t>>> tok_of_expert; | |
| // 每個 token 用到的 (expert, topk 位置) —— token 切分時用(避免輸出競爭) | |
| std::vector<std::vector<std::pair<int32_t, int32_t>>> exp_of_tok; | |
| std::vector<int32_t> need; // 本層要用的 expert 編號 | |
| std::vector<int32_t> need_sorted; // 依編號排序(合併讀取用) | |
| std::vector<int32_t> need_slot; // 對應的 arena 槽位 | |
| std::vector<std::vector<float>> partial; // mode A:每個執行緒一份累加緩衝 | |
| std::vector<int32_t> slot_of_expert; // [expert] -> arena 槽位(-1 = 沒分頁) | |
| std::vector<int32_t> need_slot_of_expert; // 合併讀取時暫存槽位 | |
| std::vector<int32_t> miss_sorted; // 需要從 SSD 讀的 expert(已排序) | |
| gen_barrier bar; | |
| std::mutex mtx; | |
| bool warned = false; | |
| }; | |
| static std::vector<layer_rec *> g_recs; | |
| static layer_rec * rec_for(int il) { | |
| static std::mutex m; | |
| std::lock_guard<std::mutex> lk(m); | |
| if ((int) g_recs.size() <= il) { | |
| g_recs.resize(il + 1, nullptr); | |
| } | |
| if (!g_recs[il]) { | |
| g_recs[il] = new layer_rec(); | |
| g_recs[il]->il = il; | |
| } | |
| return g_recs[il]; | |
| } | |
| // ---------------------------------------------------------------- 自訂 op 本體 | |
| // mode C(預設,partial 緩衝 <= 上限):依 expert 流水線,取→算交錯、無中途 barrier。 | |
| // mode B(partial 緩衝太大時的退路):先分頁完再一起算,會有 barrier 等待。 | |
| // 上限給一個很大的值=永遠走 mode C;給 0=強制 mode B(除錯用)。 | |
| static size_t env_partial_cap_mb() { | |
| const char * e = getenv("SDQ_PARTIAL_CAP_MB"); | |
| return e ? (size_t) atoi(e) : 4096; // MB,預設 4096 MB → 一律 mode C | |
| } | |
| // ---------------------------------------------------------------- 單一 expert 的數學 | |
| // 從 arena 的槽位記憶體算出一個 expert 對多個 token 的輸出(加權後直接累加到 out) | |
| static void expert_forward(const expert_desc & ed, const uint8_t * base, | |
| const float * cur, int64_t n_embd, | |
| const std::vector<std::pair<int32_t, int32_t>> & toks, | |
| const float * wts, size_t wts_nb0, size_t wts_nb1, int /*n_used*/, | |
| float swiglu_limit, | |
| float * out, std::vector<uint8_t> * scratch) { | |
| int64_t gate_rows = ed.ne1; | |
| int64_t up_rows = ed.ne1; | |
| size_t gate_row = ggml_row_size(ed.type[part::GATE], ed.ne0); | |
| size_t up_row = gate_row; | |
| if (ed.fused_gate_up) { | |
| gate_rows = ed.ne1 / 2; | |
| up_rows = gate_rows; | |
| } else { | |
| up_row = ggml_row_size(ed.type[part::UP], ed.ne0); | |
| } | |
| const vec_dot_fn f_gate = dot_for(ed.type[part::GATE]); | |
| const vec_dot_fn f_up = ed.fused_gate_up ? f_gate : dot_for(ed.type[part::UP]); | |
| const vec_dot_fn f_down = dot_for(ed.type[part::DOWN]); | |
| if (!f_gate || !f_up || !f_down) { | |
| fprintf(stderr, "[sdq] 不支援的專家量化型別:gate=%s up=%s down=%s\n", | |
| ggml_type_name(ed.type[part::GATE]), | |
| ggml_type_name(ed.type[part::UP]), | |
| ggml_type_name(ed.type[part::DOWN])); | |
| return; | |
| } | |
| const uint8_t * wg = base + ed.slot_off[part::GATE]; | |
| const uint8_t * wu = base + ed.slot_off[part::UP]; | |
| const uint8_t * wd = base + ed.slot_off[part::DOWN]; | |
| const size_t down_row = ggml_row_size(ed.type[part::DOWN], ed.down_ne0); | |
| const ggml_type act_gate = act_type_for(ed.type[part::GATE]); | |
| const ggml_type act_down = act_type_for(ed.type[part::DOWN]); | |
| const size_t xq_sz = ggml_row_size(act_gate, ed.ne0); | |
| const size_t hq_sz = ggml_row_size(act_down, ed.down_ne0); | |
| if (scratch->size() < xq_sz + hq_sz) { | |
| scratch->assign(xq_sz + hq_sz, 0); | |
| } | |
| uint8_t * xq = scratch->data(); | |
| uint8_t * hq = xq + xq_sz; | |
| std::vector<float> gate((size_t) gate_rows); | |
| std::vector<float> upv((size_t) up_rows); | |
| std::vector<float> h((size_t) ed.down_ne0); | |
| PROF_T0_x(_ta); // expert_forward 總計時 | |
| for (const auto & tk : toks) { | |
| const int32_t t = tk.first; | |
| const int32_t k = tk.second; | |
| PROF_T0_x(_tb); // 量化 activation | |
| quantize_act(act_gate, cur + (size_t) t * n_embd, xq, ed.ne0); | |
| PROF_ADD(qact_ns, _tb); | |
| PROF_T0_x(_tc); // gate / up / down 三次點積 | |
| for (int64_t r = 0; r < gate_rows; r++) { | |
| float acc = 0.0f; | |
| dot_row(f_gate, wg + r * gate_row, xq, gate_row, (int) ed.ne0, acc); | |
| gate[(size_t) r] = acc; | |
| } | |
| for (int64_t r = 0; r < up_rows; r++) { | |
| float acc = 0.0f; | |
| const uint8_t * src = ed.fused_gate_up ? (wg + (gate_rows + r) * gate_row) | |
| : (wu + r * up_row); | |
| dot_row(f_up, src, xq, ed.fused_gate_up ? gate_row : up_row, (int) ed.ne0, acc); | |
| upv[(size_t) r] = acc; | |
| } | |
| PROF_ADD(dot_ns, _tc); | |
| PROF_T0_x(_td); // SiLU + 加權求和 | |
| for (int64_t r = 0; r < ed.down_ne0; r++) { | |
| const float g = gate[(size_t) r]; | |
| const float u = upv[(size_t) r]; | |
| float act; | |
| if (swiglu_limit > 1e-6f) { | |
| const float uc = std::max(-swiglu_limit, std::min(swiglu_limit, u)); | |
| float gc = g / (1.0f + std::exp(-g)); | |
| gc = std::max(-swiglu_limit, std::min(swiglu_limit, gc)); | |
| act = gc * uc; | |
| } else { | |
| act = (g / (1.0f + std::exp(-g))) * u; | |
| } | |
| h[(size_t) r] = act; | |
| } | |
| quantize_act(act_down, h.data(), hq, ed.down_ne0); | |
| PROF_ADD(act_ns, _td); | |
| const float w = *(const float *) ((const uint8_t *) wts + k * wts_nb0 + (size_t) t * wts_nb1); | |
| float * o = out + (size_t) t * n_embd; | |
| for (int64_t r = 0; r < ed.down_ne1; r++) { | |
| float acc = 0.0f; | |
| dot_row(f_down, wd + r * down_row, hq, down_row, (int) ed.down_ne0, acc); | |
| o[r] += acc * w; | |
| } | |
| } | |
| PROF_ADD(fwd_ns, _ta); | |
| } | |
| static void moe_custom_op(ggml_tensor * dst, int ith, int nth, void * userdata) { | |
| layer_rec * R = (layer_rec *) userdata; | |
| pager & P = pager::instance(); | |
| // 從 op 的 src 取得輸入(保證是「這一次」真的被計算的張量) | |
| const ggml_tensor * src_cur = dst->src[0]; | |
| const ggml_tensor * src_ids = dst->src[1]; | |
| const ggml_tensor * src_wts = dst->src[2]; | |
| if (!src_cur || !src_ids || !src_wts) { | |
| if (ith == 0) { | |
| fprintf(stderr, "[sdq] op 缺少 src(%d)\n", R->il); | |
| } | |
| return; | |
| } | |
| const layer_desc * L = P.layer(R->il); | |
| PROF_T0_x(_tg); // op 總計時 | |
| g_prof.calls++; | |
| const int64_t n_embd = src_cur->ne[0]; | |
| const int64_t n_tok = src_cur->ne[2]; | |
| const int64_t n_used = R->n_expert_used; | |
| // ids 是 2D [n_expert_used, n_tokens] → 用 nb[0]/nb[1] | |
| // wts 是 3D [1, n_expert_used, n_tokens] → 用 nb[1]/nb[2](nb[0] 會是 4 而不是 n_used*4!) | |
| const size_t ids_nb0 = src_ids->nb[0], ids_nb1 = src_ids->nb[1]; | |
| const size_t wts_nb0 = src_wts->nb[src_wts->ne[0] == 1 ? 1 : 0]; | |
| const size_t wts_nb1 = src_wts->nb[src_wts->ne[0] == 1 ? 2 : 1]; | |
| const float * cur = (const float *) src_cur->data; | |
| const float * wts = (const float *) src_wts->data; | |
| if (!L || L->experts.empty() || dst->type != GGML_TYPE_F32) { | |
| if (ith == 0 && !R->warned) { | |
| R->warned = true; | |
| fprintf(stderr, "[sdq] 層 %d 沒有可用的分頁描述\n", R->il); | |
| } | |
| return; | |
| } | |
| if (R->il == 0 && getenv("SDQ_DBG_OP")) { | |
| fprintf(stderr, "[sdq-op] L%d n_tok=%lld nth=%d need=%d ids_nb1=%zu wts_nb1=%zu\n", | |
| R->il, (long long) n_tok, nth, (int) R->need.size(), ids_nb1, wts_nb1); | |
| } | |
| // ---- 1) 解除上一層的 pin(mode B 直接累加輸出,需要先清空) | |
| PROF_T0_x(_tsetup); | |
| if (ith == 0) { | |
| P.unpin_all(); | |
| } | |
| if (nth > 1) { R->bar.sync(nth); } | |
| // ---- 2) 依 expert 分組 token(只有執行緒 0 做,其餘在此等待) | |
| if (ith == 0) { | |
| std::lock_guard<std::mutex> lk(R->mtx); | |
| R->tok_of_expert.assign((size_t) R->n_expert, {}); | |
| R->exp_of_tok.assign((size_t) n_tok, {}); | |
| const uint8_t * ids_base = (const uint8_t *) src_ids->data; | |
| for (int64_t t = 0; t < n_tok; t++) { | |
| for (int64_t k = 0; k < n_used; k++) { | |
| const int32_t e = *(const int32_t *) (ids_base + k * ids_nb0 + (size_t) t * ids_nb1); | |
| if (e >= 0 && e < R->n_expert) { | |
| auto & v = R->tok_of_expert[(size_t) e]; | |
| bool seen = false; | |
| for (const auto & p : v) { | |
| if (p.first == (int32_t) t) { seen = true; break; } | |
| } | |
| if (!seen) { | |
| v.emplace_back((int32_t) t, (int32_t) k); | |
| } | |
| R->exp_of_tok[(size_t) t].emplace_back(e, (int32_t) k); | |
| } | |
| } | |
| } | |
| R->need.clear(); | |
| for (int64_t e = 0; e < R->n_expert; e++) { | |
| if (!R->tok_of_expert[(size_t) e].empty()) { | |
| R->need.push_back((int32_t) e); | |
| } | |
| } | |
| R->need_sorted = R->need; | |
| std::sort(R->need_sorted.begin(), R->need_sorted.end()); | |
| R->need_slot.assign(R->need.size(), -1); // 只能由 ith==0 配置,其餘執行緒唯讀 | |
| R->slot_of_expert.assign((size_t) R->n_expert, -1); | |
| R->need_slot_of_expert.assign((size_t) R->n_expert, -1); | |
| R->miss_sorted.clear(); | |
| R->miss_sorted.reserve((size_t) R->n_expert); | |
| if (R->partial.size() < 64) { | |
| R->partial.resize(64); | |
| } | |
| sdq_route_log(R->il, n_tok, src_ids, ids_nb0, ids_nb1, (int) n_used); | |
| } | |
| if (nth > 1) { R->bar.sync(nth); } | |
| PROF_ADD(setup_ns, _tsetup); | |
| const int n_need = (int) R->need.size(); | |
| // ---- 3+4) 分頁與計算融合成一條 pipeline | |
| // | |
| // 關鍵效能點:不要「先把整層的 expert 都讀完再開始算」。做法是每個執行緒 | |
| // 自己依序處理它負責的 expert:讀一個 → 立刻算它 → 再讀下一個。 | |
| // 這樣 SSD 讀取和 CPU 計算自然重疊,而且 8 條執行緒各自發出 pread, | |
| // SSD 佇列深度也夠(不需要再靠 queue depth)。 | |
| // | |
| // 平行切分:mode A 依 expert 切(每個執行緒有自己的累加緩衝,最後相加); | |
| // mode B 依 token 切(同一個 token 的所有 expert 由同一執行緒算, | |
| // 但必須先把該 token 的 expert 都備齊,所以先分頁再計算)。 | |
| const int64_t out_elems = n_embd * n_tok; | |
| const size_t partial_bytes = (size_t) out_elems * 4 * (size_t) nth; | |
| const bool mode_a = partial_bytes <= (size_t) env_partial_cap_mb() << 20; | |
| float * out = (float *) dst->data; | |
| if (!mode_a) { | |
| memset(dst->data, 0, (size_t) ggml_nbytes(dst)); | |
| } | |
| if (mode_a) { | |
| // ===================== mode C(預設):依 expert 流水線 ===================== | |
| // 三個設計重點,每個都來自實測: | |
| // | |
| // (1) **取→算交錯,中間沒有 barrier**。 | |
| // 舊的 mode B 是「先把整層 8 個 expert 全部讀完 → barrier → 才開始算」。 | |
| // 量測顯示 decode 有 66% 的時間卡在 barrier 上等最慢的執行緒 | |
| // (SDQ_PROFILE_TIME 的 barrier 等待 = op 時間的 70%)。改成交錯後 | |
| // 讀取與計算自然重疊,實測 2.79 → 3.49 tok/s。 | |
| // | |
| // (2) **分頁走 acquire(),不用 reserve_slot + read_expert_span**。 | |
| // reserve_slot() 會在資料還沒讀進來之前就把 (layer,expert) publish 到 | |
| // page_ 表;一旦該層的 pread 失敗或被中斷,這個槽就會留在 page_ 裡 | |
| // 卻是舊資料,之後的 probe() 會把它當成命中 → 靜默算錯。 | |
| // acquire() 是「先讀完再 publish」,成本相同但不會有這個失效模式。 | |
| // | |
| // (3) **i += nth 交錯分配**,不是切成連續區塊。 | |
| // 每個 expert 的 SSD 延遲差異很大,交錯分配讓快執行緒自然多幫忙。 | |
| // | |
| // 輸出不直接寫 dst(同一個 token 的不同 expert 會撞在一起), | |
| // 所以仍然用「每執行緒一份累加緩衝 + 最後歸約」。 | |
| std::vector<float> & mine = R->partial[(size_t) ith]; | |
| mine.assign((size_t) out_elems, 0.0f); | |
| std::vector<uint8_t> scratch; | |
| // 在取本層的 expert **之前**先替下一層排預取。 | |
| // | |
| // 實測結論(S7):**預設關閉,而且量到開啟反而慢 8%。** | |
| // 預測來自 prev_used_[il+1](上一個 token 在該層選的專家)。想用它躲在 | |
| // 「本層的讀取+計算」後面,但量到的結果是: | |
| // 開(5.68 tok/s,預取使用率 0.48 %)vs 關(6.20 tok/s,0.28 %) | |
| // 命中率幾乎一樣(70.5 % vs 70.7 %),但 SSD 位元組從 201.5 升到 216.0 MB/token | |
| // → 預測不夠準,多讀的全是白讀,而且還要和本層的 demand 讀搶頻寬。 | |
| // 另外開啟時同一提示詞的輸出不一致 → 預取佔用的槽位(busy)會讓 | |
| // acquire() 偶爾拿不到槽位而**靜默跳過一個 expert**,這是正確性問題。 | |
| // | |
| // 保留程式碼是為了讓後人不用重新發明;文件見 docs/06_slots.md §4。 | |
| // SDQ_PREFETCH_LAYER_AHEAD=1 可以打開來重現那個負結果。 | |
| static const bool layer_ahead = [] { | |
| const char * e = getenv("SDQ_PREFETCH_LAYER_AHEAD"); | |
| return e ? atoi(e) != 0 : false; | |
| }(); | |
| if (layer_ahead && ith == 0) { | |
| P.prefetch_layer_ahead(R->il + 1); | |
| } | |
| for (int i = ith; i < n_need; i += nth) { | |
| const int32_t e = R->need[(size_t) i]; | |
| const int32_t sl = P.acquire(R->il, e, true); | |
| if (sl < 0) { | |
| continue; | |
| } | |
| P.note_used(R->il, e); | |
| expert_forward(L->experts[(size_t) e], P.slot_ptr(sl), cur, n_embd, | |
| R->tok_of_expert[(size_t) e], wts, wts_nb0, wts_nb1, | |
| (int) n_used, R->swiglu_limit, mine.data(), &scratch); | |
| // expert_forward 回傳後這個槽位就沒人讀了(最後的歸約只讀 mine), | |
| // 所以立刻放掉 pin,不要讓它被釘到下一層開頭。 | |
| // 不放的話 prefill 一層會把整層用到的 expert 全部釘住(最多 256 個), | |
| // 大槽位不夠時就會中止(見 pager::release 的說明)。 | |
| P.release(sl); | |
| } | |
| if (nth > 1) { R->bar.sync(nth); } | |
| for (int64_t idx = ith; idx < out_elems; idx += nth) { | |
| float acc = 0.0f; | |
| for (int k = 0; k < nth; k++) { | |
| acc += R->partial[(size_t) k][(size_t) idx]; | |
| } | |
| out[idx] = acc; | |
| } | |
| } else { | |
| // mode B:token 多到 partial 緩衝太大時,依 token 切分 | |
| // 先把所有需要的 expert 分頁(各執行緒平行 pread) | |
| { | |
| PROF_T0_x(_th); // 分頁 / SSD 讀取 | |
| for (int i = ith; i < n_need; i += nth) { | |
| const int32_t e = R->need[(size_t) i]; | |
| const int s = P.acquire(R->il, e, true); | |
| g_prof.experts++; | |
| R->need_slot[(size_t) i] = s; | |
| if (s >= 0) { | |
| R->slot_of_expert[(size_t) e] = s; | |
| P.note_used(R->il, e); | |
| } | |
| } | |
| PROF_ADD(read_ns, _th); | |
| } | |
| if (nth > 1) { R->bar.sync(nth); } | |
| const float * cur_p = (const float *) src_cur->data; | |
| const float * wts_p = (const float *) src_wts->data; | |
| for (int64_t t = ith; t < n_tok; t += nth) { | |
| std::vector<uint8_t> scratch; | |
| for (const auto & ek : R->exp_of_tok[(size_t) t]) { | |
| const int slot = R->slot_of_expert[(size_t) ek.first]; | |
| if (slot < 0) { | |
| continue; | |
| } | |
| std::vector<std::pair<int32_t, int32_t>> one = {{(int32_t) t, ek.second}}; | |
| expert_forward(L->experts[(size_t) ek.first], P.slot_ptr(slot), cur_p, n_embd, | |
| one, wts_p, wts_nb0, wts_nb1, (int) n_used, | |
| R->swiglu_limit, out, &scratch); | |
| } | |
| } | |
| } | |
| // 除錯:把某一層的實際輸入/輸出 dump 出來,用 NumPy 對照。 | |
| // 必須先 barrier:上面的歸約是由各執行緒分片寫 out 的,沒有這一道會讀到半成品。 | |
| if (nth > 1) { R->bar.sync(nth); } | |
| if (const char * dump_path = getenv("SDQ_MOE_DUMP")) { | |
| const int dump_layer = getenv("SDQ_MOE_DUMP_LAYER") ? atoi(getenv("SDQ_MOE_DUMP_LAYER")) : -1; | |
| if ((dump_layer < 0 || R->il == dump_layer) && ith == 0) { | |
| std::string path = std::string(dump_path) + "." + std::to_string(R->il); | |
| FILE * f = fopen(path.c_str(), "wb"); | |
| if (f) { | |
| const int32_t meta[6] = {(int32_t) n_embd, (int32_t) n_tok, (int32_t) n_used, | |
| (int32_t) R->n_expert, 1, 2}; | |
| fwrite(meta, sizeof(int32_t), 6, f); | |
| fwrite(cur, sizeof(float), (size_t) (n_embd * n_tok), f); | |
| const uint8_t * ids_base = (const uint8_t *) src_ids->data; | |
| for (int64_t t = 0; t < n_tok; t++) { | |
| for (int64_t k = 0; k < n_used; k++) { | |
| const int32_t e = *(const int32_t *) (ids_base + k * ids_nb0 + (size_t) t * ids_nb1); | |
| fwrite(&e, sizeof(int32_t), 1, f); | |
| } | |
| } | |
| fwrite(wts, sizeof(float), (size_t) (n_used * n_tok), f); | |
| fwrite(out, sizeof(float), (size_t) (n_embd * n_tok), f); | |
| fclose(f); | |
| fprintf(stderr, "[sdq] dumped layer %d -> %s\n", R->il, path.c_str()); | |
| } | |
| } | |
| } | |
| if (nth > 1) { R->bar.sync(nth); } | |
| PROF_ADD(op_ns, _tg); | |
| } | |
| // 輸出計時結果(sdq_shutdown 呼叫) | |
| void sdq_print_time_profile() { | |
| if (!prof_on()) { | |
| return; | |
| } | |
| const double ms = 1e6; | |
| auto v = [ms](const std::atomic<uint64_t> & a) { return a.load() / ms; }; | |
| const double op = v(g_prof.op_ns); | |
| const double fwd = v(g_prof.fwd_ns); | |
| fprintf(stderr, "[sdq-prof] 自訂 MoE op(%llu 次呼叫、%llu 個 expert 分頁)\n", | |
| (unsigned long long) g_prof.calls.load(), (unsigned long long) g_prof.experts.load()); | |
| fprintf(stderr, "[sdq-prof] op 合計 %9.1f ms\n", op); | |
| fprintf(stderr, "[sdq-prof] expert_forward %9.1f ms (%5.1f%%) 其中:\n", fwd, op > 0 ? 100.0 * fwd / op : 0.0); | |
| fprintf(stderr, "[sdq-prof] 量化 activation%9.1f ms (%5.1f%% of fwd)\n", v(g_prof.qact_ns), fwd > 0 ? 100.0 * v(g_prof.qact_ns) / fwd : 0.0); | |
| fprintf(stderr, "[sdq-prof] 三次點積 %9.1f ms (%5.1f%% of fwd)\n", v(g_prof.dot_ns), fwd > 0 ? 100.0 * v(g_prof.dot_ns) / fwd : 0.0); | |
| fprintf(stderr, "[sdq-prof] SiLU+加權 %9.1f ms (%5.1f%% of fwd)\n", v(g_prof.act_ns), fwd > 0 ? 100.0 * v(g_prof.act_ns) / fwd : 0.0); | |
| fprintf(stderr, "[sdq-prof] SSD 讀取/分頁 %9.1f ms (%5.1f%%)\n", v(g_prof.read_ns), op > 0 ? 100.0 * v(g_prof.read_ns) / op : 0.0); | |
| fprintf(stderr, "[sdq-prof] 前置 setup %9.1f ms (%5.1f%%)\n", v(g_prof.setup_ns), op > 0 ? 100.0 * v(g_prof.setup_ns) / op : 0.0); | |
| fprintf(stderr, "[sdq-prof] barrier 等待 %9.1f ms (%5.1f%%)\n", v(g_prof.bar_ns), op > 0 ? 100.0 * v(g_prof.bar_ns) / op : 0.0); | |
| } | |
| // ---------------------------------------------------------------- graph 建構 | |
| ggml_tensor * sdq_build_moe(ggml_cgraph * gf, ggml_context * ctx, ggml_tensor * cur, | |
| ggml_tensor * ids, ggml_tensor * weights, int il, int64_t n_expert, | |
| int64_t n_expert_used, float swiglu_limit) { | |
| layer_rec * R = rec_for(il); | |
| R->n_expert = n_expert; | |
| R->n_expert_used = n_expert_used; | |
| R->swiglu_limit = swiglu_limit; | |
| // 呼叫端給的是 2D [n_embd, n_tokens],這裡跟上游一樣先 reshape 成 3D | |
| // [n_embd, 1, n_tokens](上游 mul_mat_id 路徑也是這樣拿 token 數的) | |
| ggml_tensor * cur3 = (cur->ne[1] == 1) ? cur : ggml_reshape_3d(ctx, cur, cur->ne[0], 1, cur->ne[1]); | |
| const int64_t n_tok = cur3->ne[2]; | |
| ggml_tensor * args[3] = { cur3, ids, weights }; | |
| ggml_tensor * out = ggml_custom_4d(ctx, GGML_TYPE_F32, cur3->ne[0], n_tok, 1, 1, | |
| args, 3, moe_custom_op, GGML_N_TASKS_MAX, R); | |
| ggml_set_name(out, "sdq_moe_out"); | |
| // 關鍵:把這個節點註冊進圖,ggml 才會為它配置獨立的 compute buffer。 | |
| // 少了這一行,layer>=1 的輸出會被其他張量覆蓋(layer 0 僥倖正確)。 | |
| ggml_build_forward_expand(gf, out); | |
| return out; | |
| } | |
| } // namespace sdq | |
| namespace sdq { | |
| // 自我測試:對指定 (layer, expert) 用 pager 的資料算一次 MoE 輸出, | |
| // 讓 Python 用純 NumPy 從同一段 GGUF 位元組重算並比對。 | |
| // 輸出:out[0..n_embd) = 輸出,out[n_embd..2*n_embd) = 輸入 activation | |
| int sdq_moe_selftest(int il, int ie, const float * x, float * out_io, | |
| int64_t n_embd, int64_t n_ff, int n_tok) { | |
| pager & P = pager::instance(); | |
| const layer_desc * L = P.layer(il); | |
| if (!L || ie < 0 || ie >= (int) L->experts.size()) { | |
| fprintf(stderr, "[sdq-selftest] 層/expert 不存在:%d/%d\n", il, ie); | |
| return 1; | |
| } | |
| const int slot = P.acquire(il, ie, true); | |
| if (slot < 0) { | |
| fprintf(stderr, "[sdq-selftest] 分頁失敗\n"); | |
| return 2; | |
| } | |
| const expert_desc & ed = L->experts[(size_t) ie]; | |
| if (ed.ne0 != n_embd || ed.ne1 != n_ff) { | |
| fprintf(stderr, "[sdq-selftest] 維度不符:ne0=%lld ne1=%lld(預期 %lld/%lld)\n", | |
| (long long) ed.ne0, (long long) ed.ne1, (long long) n_embd, (long long) n_ff); | |
| return 3; | |
| } | |
| std::vector<std::pair<int32_t, int32_t>> toks; | |
| std::vector<float> wts((size_t) n_tok, 1.0f); | |
| for (int t = 0; t < n_tok; t++) { | |
| toks.emplace_back(t, 0); | |
| } | |
| std::vector<float> out((size_t) n_embd * n_tok, 0.0f); | |
| std::vector<uint8_t> scratch; | |
| expert_forward(ed, P.slot_ptr(slot), x, n_embd, toks, wts.data(), 4, 4, 1, 0.0f, | |
| out.data(), &scratch); | |
| for (int64_t t = 0; t < n_tok; t++) { | |
| for (int64_t i = 0; i < n_embd; i++) { | |
| out_io[t * n_embd + i] = out[(size_t) t * n_embd + i]; | |
| out_io[n_tok * n_embd + t * n_embd + i] = x[(size_t) t * n_embd + i]; | |
| } | |
| } | |
| P.unpin_all(); | |
| fprintf(stderr, "[sdq-selftest] L%d E%d n_tok=%d 計算完成(gate %s / up %s / down %s)\n", | |
| il, ie, n_tok, ggml_type_name(ed.type[part::GATE]), ggml_type_name(ed.type[part::UP]), | |
| ggml_type_name(ed.type[part::DOWN])); | |
| return 0; | |
| } | |
| } // namespace sdq | |