sddqwen35a3b / cpp /sdq_moe.cpp
HelloSun's picture
S8: 新增 verify [2b/3] 預取路徑測試(預取開 + 512MB arena 強制回收 vs NumPy);新測試第一次跑就抓出大槽位 starvation(下限 16 < 工作集 24)與 pin 整層釘住兩個缺陷,修完 512MB arena 也能跑完 prefill;三層驗證全綠
f33b7a9 verified
Raw History Blame Contribute Delete
32.7 kB
// SDQ 被分頁的 MoE FFN(自訂 ggml op)
//
// 為什麼要自己算:上游 build_moe_ffn() 用 ggml_mul_mat_id 直接對
// blk.N.ffn_{gate,up,down}_exps.weight 這個「三維大張量」做矩陣乘,
// 權重必須整個在 RAM 裡。20.6 GiB 的專家權重在 8 GB 機器上不可能。
//
// 這裡改成:
// 1. 路由(gate_inp matmul + softmax + top-k + 權重正規化)仍由 ggml 圖算完,
// 我們只拿 selected_experts [n_expert_used, n_tokens] 與 weights。
// 2. 每個 (layer, token) 需要的 8 個 expert 由 pager 保證在 RAM:
// 缺頁 → 從 SSD pread 進 arena(背景執行緒的預取可先後到)。
// 3. 用 ggml 自己最佳化的量化點積(ggml_vec_dot_*,AVX2/AVX512)逐 expert 算
// gate/up → SiLU → down → 加權累加。
// 4. 多執行緒:ith/nth 由 ggml 分配;分頁階段各執行緒各自 pread,餵飽 SSD 佇列深度。
#include "sdq_pager.h"
#include "ggml-cpu/quants.h"
#include <algorithm>
#include <atomic>
#include <chrono>
#include <cmath>
#include <condition_variable>
#include <cstdio>
#include <cstring>
#include <mutex>
#include <utility>
#include <string>
#include <vector>
namespace sdq {
// ---------------------------------------------------------------- 效能計時探針
// 用途:找出「自訂 MoE op」裡面時間到底花在哪裡。開 SDQ_PROFILE_TIME=1 才啟用,
// 關閉時是零成本(幾個 if constexpr 級別的布林檢查)。
struct moe_prof {
std::atomic<uint64_t> op_ns{0}; // 整個 op
std::atomic<uint64_t> page_ns{0}; // probe / reserve_slot(含鎖)
std::atomic<uint64_t> setup_ns{0}; // op 前置:unpin_all + 建 tok_of_expert + 前兩道 barrier
std::atomic<uint64_t> bar_ns{0}; // 所有 barrier 的等待時間
std::atomic<uint64_t> read_ns{0}; // read_expert_span / acquire 的 I/O
std::atomic<uint64_t> fwd_ns{0}; // expert_forward 合計
std::atomic<uint64_t> qact_ns{0}; // 其中:量化 activation
std::atomic<uint64_t> dot_ns{0}; // 其中:gate/up/down 三次點積
std::atomic<uint64_t> act_ns{0}; // 其中:SiLU + 加權
std::atomic<uint64_t> calls{0};
std::atomic<uint64_t> experts{0};
};
static moe_prof g_prof;
static inline bool prof_on() {
static const bool on = [] { const char * e = getenv("SDQ_PROFILE_TIME"); return e && *e && *e != '0'; }();
return on;
}
static inline uint64_t now_ns() {
return (uint64_t) std::chrono::duration_cast<std::chrono::nanoseconds>(
std::chrono::steady_clock::now().time_since_epoch()).count();
}
#define PROF_T0_x(v) const uint64_t v = prof_on() ? now_ns() : 0
#define PROF_ADD(field, tv) do { if (prof_on()) g_prof.field += now_ns() - (tv); } while (0)
// ---------------------------------------------------------------- 執行緒柵欄
class gen_barrier {
public:
void sync(int n) {
PROF_T0_x(_tbar);
std::unique_lock<std::mutex> lk(mu_);
const unsigned long g = gen_;
if (++count_ == n) {
count_ = 0;
gen_++;
cv_.notify_all();
PROF_ADD(bar_ns, _tbar);
return;
}
cv_.wait(lk, [this, g] { return gen_ != g; });
PROF_ADD(bar_ns, _tbar);
}
private:
int count_ = 0;
unsigned long gen_ = 0;
std::mutex mu_;
std::condition_variable cv_;
};
// ---------------------------------------------------------------- 量化點積派送
using vec_dot_fn = void (*)(int, float *, size_t, const void *, size_t, const void *, size_t, int);
static vec_dot_fn dot_for(ggml_type t) {
switch (t) {
case GGML_TYPE_Q4_K: return ggml_vec_dot_q4_K_q8_K;
case GGML_TYPE_Q5_K: return ggml_vec_dot_q5_K_q8_K;
case GGML_TYPE_Q6_K: return ggml_vec_dot_q6_K_q8_K;
case GGML_TYPE_Q3_K: return ggml_vec_dot_q3_K_q8_K;
case GGML_TYPE_Q2_K: return ggml_vec_dot_q2_K_q8_K;
case GGML_TYPE_Q8_0: return ggml_vec_dot_q8_0_q8_0;
case GGML_TYPE_Q4_0: return ggml_vec_dot_q4_0_q8_0;
case GGML_TYPE_Q5_0: return ggml_vec_dot_q5_0_q8_0;
case GGML_TYPE_IQ4_NL: return ggml_vec_dot_iq4_nl_q8_0;
case GGML_TYPE_IQ4_XS: return ggml_vec_dot_iq4_xs_q8_K;
default: return nullptr;
}
}
// K-quant 的權重配 Q8_K activation;其餘配 Q8_0(與 ggml 的 vec_dot_type 一致)
static ggml_type act_type_for(ggml_type w) {
switch (w) {
case GGML_TYPE_Q4_K:
case GGML_TYPE_Q5_K:
case GGML_TYPE_Q6_K:
case GGML_TYPE_Q3_K:
case GGML_TYPE_Q2_K:
case GGML_TYPE_IQ4_XS:
return GGML_TYPE_Q8_K;
default:
return GGML_TYPE_Q8_0;
}
}
// 把 activation 量化成對應的 dot 型別
// 注意:ggml_quantize_chunk() 不支援 Q8_K(Q8_K 只在 mul_mat kernel 內部產生),
// 所以這裡直接呼叫 ggml-cpu 的 quantize_row_q8_K。
static void quantize_act(ggml_type t, const float * x, void * dst, int64_t n) {
switch (t) {
case GGML_TYPE_Q8_K: quantize_row_q8_K(x, dst, n); break;
case GGML_TYPE_Q8_0: quantize_row_q8_0(x, dst, n); break;
default: ggml_quantize_chunk(t, x, dst, 0, 1, n, nullptr); break;
}
}
// ⚠️ 不要在這裡「批次化」點積。
// ggml 的 x86 版 K-quant kernel(ggml/src/ggml-cpu/arch/x86/quants.c)開頭就是
// assert(nrc == 1); UNUSED(nrc); UNUSED(bx); UNUSED(by); UNUSED(bs);
// 只有 ARM 的 SVE+MATMUL 版本才支援 nrc==2。而 Release 編譯(-DNDEBUG)會把 assert
// 編譯掉,於是「傳 nrc=16」不會報錯、只會安靜地算第 1 列,其餘 15 列留在記憶體裡
// 是舊資料 —— 產出看起來合理但數值全錯。
// (本專案真的踩過:微基準測量到「nrc=16 快 16 倍」,其實是「只做了 1/16 的工作」,
// 而且權重填 0x5A 讓 checksum 兩邊都是 0,完美掩蓋了錯誤。教訓見 docs/04。)
// 想要真的多列 SIMD 效能,只能自己寫 AVX512-VNNI 的 kernel,不是靠改參數。
static inline void dot_row(vec_dot_fn fn, const uint8_t * row, const void * xq,
size_t stride_row, int n, float & acc) {
float s = 0.0f;
fn(n, &s, 0, row, stride_row, xq, 0, 1);
acc += s;
}
// 一次算 SDQ_DOT_BATCH 個輸出列。
// 為什麼需要:ggml 的 K-quant 點積 kernel 是為「多列」設計的,nrc>1 會走完全不同的
// SIMD 路徑。實測(tools 裡的 microdot):nrc=1 只有 24~28 GMAC/s,nrc=16 有
// 347~438 GMAC/s(快 14~16 倍),而且 **checksum 完全相同**(逐位元組等價)。
// 舊程式逐列呼叫 nrc=1,等於把 14 倍的效能丟掉。
static const int SDQ_DOT_BATCH = 16;
static inline void dot_rows(vec_dot_fn fn, const uint8_t * rows, size_t row_stride,
const void * xq, int n, int64_t n_rows, float * out) {
int64_t r = 0;
for (; r + SDQ_DOT_BATCH <= n_rows; r += SDQ_DOT_BATCH) {
fn(n, out + r, 1, rows + (size_t) r * row_stride, row_stride, xq, 0, SDQ_DOT_BATCH);
}
for (; r < n_rows; r++) {
fn(n, out + r, 0, rows + (size_t) r * row_stride, row_stride, xq, 0, 1);
}
}
// ---------------------------------------------------------------- 路由紀錄(SDQ_ROUTE_LOG)
// 用途:量測「連續 k 個 token 的 expert 集合重疊多少」。
// 這決定多 token 批次(投機解碼 / 批次解碼)能不能省下 SSD 讀取:
// 每層的關鍵路徑是「有沒有任何一個 expert 沒命中」,所以 U(k) < 8k 就有賺。
// 格式(每行一次 op 呼叫,空白分欄):
// <layer> <n_tok> <t0:e0,e1,...> <t1:...> ...
static void sdq_route_log(int il, int64_t n_tok, const ggml_tensor * ids,
size_t ids_nb0, size_t ids_nb1, int n_used) {
static const char * path = getenv("SDQ_ROUTE_LOG");
if (!path || !*path) {
return;
}
static std::mutex mtx;
static FILE * f = nullptr;
{
std::lock_guard<std::mutex> lk(mtx);
if (!f) {
f = fopen(path, "w");
}
if (!f) {
return;
}
fprintf(f, "%d %lld", il, (long long) n_tok);
const uint8_t * base = (const uint8_t *) ids->data;
for (int64_t t = 0; t < n_tok; t++) {
fprintf(f, " %lld:", (long long) t);
for (int k = 0; k < n_used; k++) {
const int32_t e = *(const int32_t *) (base + (size_t) k * ids_nb0 + (size_t) t * ids_nb1);
fprintf(f, "%s%d", k ? "," : "", e);
}
}
fputc('\n', f);
}
}
// ---------------------------------------------------------------- 每層執行紀錄
// 重要:input 張量的指標**不能**在建圖時暫存起來 —— llama.cpp 每次 decode 都會
// 重新建圖,graph context 的配置器可能把張量搬到不同位址(形狀不同時尤其明顯)。
// 所以 op 執行時一律從 dst->src[] 取得,這才是唯一可靠的來源。
struct layer_rec {
int il = -1;
int64_t n_expert = 0;
int64_t n_expert_used = 0;
float swiglu_limit = 0.0f;
// 每個 expert 需要處理的 (token, topk 位置) 配對
std::vector<std::vector<std::pair<int32_t, int32_t>>> tok_of_expert;
// 每個 token 用到的 (expert, topk 位置) —— token 切分時用(避免輸出競爭)
std::vector<std::vector<std::pair<int32_t, int32_t>>> exp_of_tok;
std::vector<int32_t> need; // 本層要用的 expert 編號
std::vector<int32_t> need_sorted; // 依編號排序(合併讀取用)
std::vector<int32_t> need_slot; // 對應的 arena 槽位
std::vector<std::vector<float>> partial; // mode A:每個執行緒一份累加緩衝
std::vector<int32_t> slot_of_expert; // [expert] -> arena 槽位(-1 = 沒分頁)
std::vector<int32_t> need_slot_of_expert; // 合併讀取時暫存槽位
std::vector<int32_t> miss_sorted; // 需要從 SSD 讀的 expert(已排序)
gen_barrier bar;
std::mutex mtx;
bool warned = false;
};
static std::vector<layer_rec *> g_recs;
static layer_rec * rec_for(int il) {
static std::mutex m;
std::lock_guard<std::mutex> lk(m);
if ((int) g_recs.size() <= il) {
g_recs.resize(il + 1, nullptr);
}
if (!g_recs[il]) {
g_recs[il] = new layer_rec();
g_recs[il]->il = il;
}
return g_recs[il];
}
// ---------------------------------------------------------------- 自訂 op 本體
// mode C(預設,partial 緩衝 <= 上限):依 expert 流水線,取→算交錯、無中途 barrier。
// mode B(partial 緩衝太大時的退路):先分頁完再一起算,會有 barrier 等待。
// 上限給一個很大的值=永遠走 mode C;給 0=強制 mode B(除錯用)。
static size_t env_partial_cap_mb() {
const char * e = getenv("SDQ_PARTIAL_CAP_MB");
return e ? (size_t) atoi(e) : 4096; // MB,預設 4096 MB → 一律 mode C
}
// ---------------------------------------------------------------- 單一 expert 的數學
// 從 arena 的槽位記憶體算出一個 expert 對多個 token 的輸出(加權後直接累加到 out)
static void expert_forward(const expert_desc & ed, const uint8_t * base,
const float * cur, int64_t n_embd,
const std::vector<std::pair<int32_t, int32_t>> & toks,
const float * wts, size_t wts_nb0, size_t wts_nb1, int /*n_used*/,
float swiglu_limit,
float * out, std::vector<uint8_t> * scratch) {
int64_t gate_rows = ed.ne1;
int64_t up_rows = ed.ne1;
size_t gate_row = ggml_row_size(ed.type[part::GATE], ed.ne0);
size_t up_row = gate_row;
if (ed.fused_gate_up) {
gate_rows = ed.ne1 / 2;
up_rows = gate_rows;
} else {
up_row = ggml_row_size(ed.type[part::UP], ed.ne0);
}
const vec_dot_fn f_gate = dot_for(ed.type[part::GATE]);
const vec_dot_fn f_up = ed.fused_gate_up ? f_gate : dot_for(ed.type[part::UP]);
const vec_dot_fn f_down = dot_for(ed.type[part::DOWN]);
if (!f_gate || !f_up || !f_down) {
fprintf(stderr, "[sdq] 不支援的專家量化型別:gate=%s up=%s down=%s\n",
ggml_type_name(ed.type[part::GATE]),
ggml_type_name(ed.type[part::UP]),
ggml_type_name(ed.type[part::DOWN]));
return;
}
const uint8_t * wg = base + ed.slot_off[part::GATE];
const uint8_t * wu = base + ed.slot_off[part::UP];
const uint8_t * wd = base + ed.slot_off[part::DOWN];
const size_t down_row = ggml_row_size(ed.type[part::DOWN], ed.down_ne0);
const ggml_type act_gate = act_type_for(ed.type[part::GATE]);
const ggml_type act_down = act_type_for(ed.type[part::DOWN]);
const size_t xq_sz = ggml_row_size(act_gate, ed.ne0);
const size_t hq_sz = ggml_row_size(act_down, ed.down_ne0);
if (scratch->size() < xq_sz + hq_sz) {
scratch->assign(xq_sz + hq_sz, 0);
}
uint8_t * xq = scratch->data();
uint8_t * hq = xq + xq_sz;
std::vector<float> gate((size_t) gate_rows);
std::vector<float> upv((size_t) up_rows);
std::vector<float> h((size_t) ed.down_ne0);
PROF_T0_x(_ta); // expert_forward 總計時
for (const auto & tk : toks) {
const int32_t t = tk.first;
const int32_t k = tk.second;
PROF_T0_x(_tb); // 量化 activation
quantize_act(act_gate, cur + (size_t) t * n_embd, xq, ed.ne0);
PROF_ADD(qact_ns, _tb);
PROF_T0_x(_tc); // gate / up / down 三次點積
for (int64_t r = 0; r < gate_rows; r++) {
float acc = 0.0f;
dot_row(f_gate, wg + r * gate_row, xq, gate_row, (int) ed.ne0, acc);
gate[(size_t) r] = acc;
}
for (int64_t r = 0; r < up_rows; r++) {
float acc = 0.0f;
const uint8_t * src = ed.fused_gate_up ? (wg + (gate_rows + r) * gate_row)
: (wu + r * up_row);
dot_row(f_up, src, xq, ed.fused_gate_up ? gate_row : up_row, (int) ed.ne0, acc);
upv[(size_t) r] = acc;
}
PROF_ADD(dot_ns, _tc);
PROF_T0_x(_td); // SiLU + 加權求和
for (int64_t r = 0; r < ed.down_ne0; r++) {
const float g = gate[(size_t) r];
const float u = upv[(size_t) r];
float act;
if (swiglu_limit > 1e-6f) {
const float uc = std::max(-swiglu_limit, std::min(swiglu_limit, u));
float gc = g / (1.0f + std::exp(-g));
gc = std::max(-swiglu_limit, std::min(swiglu_limit, gc));
act = gc * uc;
} else {
act = (g / (1.0f + std::exp(-g))) * u;
}
h[(size_t) r] = act;
}
quantize_act(act_down, h.data(), hq, ed.down_ne0);
PROF_ADD(act_ns, _td);
const float w = *(const float *) ((const uint8_t *) wts + k * wts_nb0 + (size_t) t * wts_nb1);
float * o = out + (size_t) t * n_embd;
for (int64_t r = 0; r < ed.down_ne1; r++) {
float acc = 0.0f;
dot_row(f_down, wd + r * down_row, hq, down_row, (int) ed.down_ne0, acc);
o[r] += acc * w;
}
}
PROF_ADD(fwd_ns, _ta);
}
static void moe_custom_op(ggml_tensor * dst, int ith, int nth, void * userdata) {
layer_rec * R = (layer_rec *) userdata;
pager & P = pager::instance();
// 從 op 的 src 取得輸入(保證是「這一次」真的被計算的張量)
const ggml_tensor * src_cur = dst->src[0];
const ggml_tensor * src_ids = dst->src[1];
const ggml_tensor * src_wts = dst->src[2];
if (!src_cur || !src_ids || !src_wts) {
if (ith == 0) {
fprintf(stderr, "[sdq] op 缺少 src(%d)\n", R->il);
}
return;
}
const layer_desc * L = P.layer(R->il);
PROF_T0_x(_tg); // op 總計時
g_prof.calls++;
const int64_t n_embd = src_cur->ne[0];
const int64_t n_tok = src_cur->ne[2];
const int64_t n_used = R->n_expert_used;
// ids 是 2D [n_expert_used, n_tokens] → 用 nb[0]/nb[1]
// wts 是 3D [1, n_expert_used, n_tokens] → 用 nb[1]/nb[2](nb[0] 會是 4 而不是 n_used*4!)
const size_t ids_nb0 = src_ids->nb[0], ids_nb1 = src_ids->nb[1];
const size_t wts_nb0 = src_wts->nb[src_wts->ne[0] == 1 ? 1 : 0];
const size_t wts_nb1 = src_wts->nb[src_wts->ne[0] == 1 ? 2 : 1];
const float * cur = (const float *) src_cur->data;
const float * wts = (const float *) src_wts->data;
if (!L || L->experts.empty() || dst->type != GGML_TYPE_F32) {
if (ith == 0 && !R->warned) {
R->warned = true;
fprintf(stderr, "[sdq] 層 %d 沒有可用的分頁描述\n", R->il);
}
return;
}
if (R->il == 0 && getenv("SDQ_DBG_OP")) {
fprintf(stderr, "[sdq-op] L%d n_tok=%lld nth=%d need=%d ids_nb1=%zu wts_nb1=%zu\n",
R->il, (long long) n_tok, nth, (int) R->need.size(), ids_nb1, wts_nb1);
}
// ---- 1) 解除上一層的 pin(mode B 直接累加輸出,需要先清空)
PROF_T0_x(_tsetup);
if (ith == 0) {
P.unpin_all();
}
if (nth > 1) { R->bar.sync(nth); }
// ---- 2) 依 expert 分組 token(只有執行緒 0 做,其餘在此等待)
if (ith == 0) {
std::lock_guard<std::mutex> lk(R->mtx);
R->tok_of_expert.assign((size_t) R->n_expert, {});
R->exp_of_tok.assign((size_t) n_tok, {});
const uint8_t * ids_base = (const uint8_t *) src_ids->data;
for (int64_t t = 0; t < n_tok; t++) {
for (int64_t k = 0; k < n_used; k++) {
const int32_t e = *(const int32_t *) (ids_base + k * ids_nb0 + (size_t) t * ids_nb1);
if (e >= 0 && e < R->n_expert) {
auto & v = R->tok_of_expert[(size_t) e];
bool seen = false;
for (const auto & p : v) {
if (p.first == (int32_t) t) { seen = true; break; }
}
if (!seen) {
v.emplace_back((int32_t) t, (int32_t) k);
}
R->exp_of_tok[(size_t) t].emplace_back(e, (int32_t) k);
}
}
}
R->need.clear();
for (int64_t e = 0; e < R->n_expert; e++) {
if (!R->tok_of_expert[(size_t) e].empty()) {
R->need.push_back((int32_t) e);
}
}
R->need_sorted = R->need;
std::sort(R->need_sorted.begin(), R->need_sorted.end());
R->need_slot.assign(R->need.size(), -1); // 只能由 ith==0 配置,其餘執行緒唯讀
R->slot_of_expert.assign((size_t) R->n_expert, -1);
R->need_slot_of_expert.assign((size_t) R->n_expert, -1);
R->miss_sorted.clear();
R->miss_sorted.reserve((size_t) R->n_expert);
if (R->partial.size() < 64) {
R->partial.resize(64);
}
sdq_route_log(R->il, n_tok, src_ids, ids_nb0, ids_nb1, (int) n_used);
}
if (nth > 1) { R->bar.sync(nth); }
PROF_ADD(setup_ns, _tsetup);
const int n_need = (int) R->need.size();
// ---- 3+4) 分頁與計算融合成一條 pipeline
//
// 關鍵效能點:不要「先把整層的 expert 都讀完再開始算」。做法是每個執行緒
// 自己依序處理它負責的 expert:讀一個 → 立刻算它 → 再讀下一個。
// 這樣 SSD 讀取和 CPU 計算自然重疊,而且 8 條執行緒各自發出 pread,
// SSD 佇列深度也夠(不需要再靠 queue depth)。
//
// 平行切分:mode A 依 expert 切(每個執行緒有自己的累加緩衝,最後相加);
// mode B 依 token 切(同一個 token 的所有 expert 由同一執行緒算,
// 但必須先把該 token 的 expert 都備齊,所以先分頁再計算)。
const int64_t out_elems = n_embd * n_tok;
const size_t partial_bytes = (size_t) out_elems * 4 * (size_t) nth;
const bool mode_a = partial_bytes <= (size_t) env_partial_cap_mb() << 20;
float * out = (float *) dst->data;
if (!mode_a) {
memset(dst->data, 0, (size_t) ggml_nbytes(dst));
}
if (mode_a) {
// ===================== mode C(預設):依 expert 流水線 =====================
// 三個設計重點,每個都來自實測:
//
// (1) **取→算交錯,中間沒有 barrier**。
// 舊的 mode B 是「先把整層 8 個 expert 全部讀完 → barrier → 才開始算」。
// 量測顯示 decode 有 66% 的時間卡在 barrier 上等最慢的執行緒
// (SDQ_PROFILE_TIME 的 barrier 等待 = op 時間的 70%)。改成交錯後
// 讀取與計算自然重疊,實測 2.79 → 3.49 tok/s。
//
// (2) **分頁走 acquire(),不用 reserve_slot + read_expert_span**。
// reserve_slot() 會在資料還沒讀進來之前就把 (layer,expert) publish 到
// page_ 表;一旦該層的 pread 失敗或被中斷,這個槽就會留在 page_ 裡
// 卻是舊資料,之後的 probe() 會把它當成命中 → 靜默算錯。
// acquire() 是「先讀完再 publish」,成本相同但不會有這個失效模式。
//
// (3) **i += nth 交錯分配**,不是切成連續區塊。
// 每個 expert 的 SSD 延遲差異很大,交錯分配讓快執行緒自然多幫忙。
//
// 輸出不直接寫 dst(同一個 token 的不同 expert 會撞在一起),
// 所以仍然用「每執行緒一份累加緩衝 + 最後歸約」。
std::vector<float> & mine = R->partial[(size_t) ith];
mine.assign((size_t) out_elems, 0.0f);
std::vector<uint8_t> scratch;
// 在取本層的 expert **之前**先替下一層排預取。
//
// 實測結論(S7):**預設關閉,而且量到開啟反而慢 8%。**
// 預測來自 prev_used_[il+1](上一個 token 在該層選的專家)。想用它躲在
// 「本層的讀取+計算」後面,但量到的結果是:
// 開(5.68 tok/s,預取使用率 0.48 %)vs 關(6.20 tok/s,0.28 %)
// 命中率幾乎一樣(70.5 % vs 70.7 %),但 SSD 位元組從 201.5 升到 216.0 MB/token
// → 預測不夠準,多讀的全是白讀,而且還要和本層的 demand 讀搶頻寬。
// 另外開啟時同一提示詞的輸出不一致 → 預取佔用的槽位(busy)會讓
// acquire() 偶爾拿不到槽位而**靜默跳過一個 expert**,這是正確性問題。
//
// 保留程式碼是為了讓後人不用重新發明;文件見 docs/06_slots.md §4。
// SDQ_PREFETCH_LAYER_AHEAD=1 可以打開來重現那個負結果。
static const bool layer_ahead = [] {
const char * e = getenv("SDQ_PREFETCH_LAYER_AHEAD");
return e ? atoi(e) != 0 : false;
}();
if (layer_ahead && ith == 0) {
P.prefetch_layer_ahead(R->il + 1);
}
for (int i = ith; i < n_need; i += nth) {
const int32_t e = R->need[(size_t) i];
const int32_t sl = P.acquire(R->il, e, true);
if (sl < 0) {
continue;
}
P.note_used(R->il, e);
expert_forward(L->experts[(size_t) e], P.slot_ptr(sl), cur, n_embd,
R->tok_of_expert[(size_t) e], wts, wts_nb0, wts_nb1,
(int) n_used, R->swiglu_limit, mine.data(), &scratch);
// expert_forward 回傳後這個槽位就沒人讀了(最後的歸約只讀 mine),
// 所以立刻放掉 pin,不要讓它被釘到下一層開頭。
// 不放的話 prefill 一層會把整層用到的 expert 全部釘住(最多 256 個),
// 大槽位不夠時就會中止(見 pager::release 的說明)。
P.release(sl);
}
if (nth > 1) { R->bar.sync(nth); }
for (int64_t idx = ith; idx < out_elems; idx += nth) {
float acc = 0.0f;
for (int k = 0; k < nth; k++) {
acc += R->partial[(size_t) k][(size_t) idx];
}
out[idx] = acc;
}
} else {
// mode B:token 多到 partial 緩衝太大時,依 token 切分
// 先把所有需要的 expert 分頁(各執行緒平行 pread)
{
PROF_T0_x(_th); // 分頁 / SSD 讀取
for (int i = ith; i < n_need; i += nth) {
const int32_t e = R->need[(size_t) i];
const int s = P.acquire(R->il, e, true);
g_prof.experts++;
R->need_slot[(size_t) i] = s;
if (s >= 0) {
R->slot_of_expert[(size_t) e] = s;
P.note_used(R->il, e);
}
}
PROF_ADD(read_ns, _th);
}
if (nth > 1) { R->bar.sync(nth); }
const float * cur_p = (const float *) src_cur->data;
const float * wts_p = (const float *) src_wts->data;
for (int64_t t = ith; t < n_tok; t += nth) {
std::vector<uint8_t> scratch;
for (const auto & ek : R->exp_of_tok[(size_t) t]) {
const int slot = R->slot_of_expert[(size_t) ek.first];
if (slot < 0) {
continue;
}
std::vector<std::pair<int32_t, int32_t>> one = {{(int32_t) t, ek.second}};
expert_forward(L->experts[(size_t) ek.first], P.slot_ptr(slot), cur_p, n_embd,
one, wts_p, wts_nb0, wts_nb1, (int) n_used,
R->swiglu_limit, out, &scratch);
}
}
}
// 除錯:把某一層的實際輸入/輸出 dump 出來,用 NumPy 對照。
// 必須先 barrier:上面的歸約是由各執行緒分片寫 out 的,沒有這一道會讀到半成品。
if (nth > 1) { R->bar.sync(nth); }
if (const char * dump_path = getenv("SDQ_MOE_DUMP")) {
const int dump_layer = getenv("SDQ_MOE_DUMP_LAYER") ? atoi(getenv("SDQ_MOE_DUMP_LAYER")) : -1;
if ((dump_layer < 0 || R->il == dump_layer) && ith == 0) {
std::string path = std::string(dump_path) + "." + std::to_string(R->il);
FILE * f = fopen(path.c_str(), "wb");
if (f) {
const int32_t meta[6] = {(int32_t) n_embd, (int32_t) n_tok, (int32_t) n_used,
(int32_t) R->n_expert, 1, 2};
fwrite(meta, sizeof(int32_t), 6, f);
fwrite(cur, sizeof(float), (size_t) (n_embd * n_tok), f);
const uint8_t * ids_base = (const uint8_t *) src_ids->data;
for (int64_t t = 0; t < n_tok; t++) {
for (int64_t k = 0; k < n_used; k++) {
const int32_t e = *(const int32_t *) (ids_base + k * ids_nb0 + (size_t) t * ids_nb1);
fwrite(&e, sizeof(int32_t), 1, f);
}
}
fwrite(wts, sizeof(float), (size_t) (n_used * n_tok), f);
fwrite(out, sizeof(float), (size_t) (n_embd * n_tok), f);
fclose(f);
fprintf(stderr, "[sdq] dumped layer %d -> %s\n", R->il, path.c_str());
}
}
}
if (nth > 1) { R->bar.sync(nth); }
PROF_ADD(op_ns, _tg);
}
// 輸出計時結果(sdq_shutdown 呼叫)
void sdq_print_time_profile() {
if (!prof_on()) {
return;
}
const double ms = 1e6;
auto v = [ms](const std::atomic<uint64_t> & a) { return a.load() / ms; };
const double op = v(g_prof.op_ns);
const double fwd = v(g_prof.fwd_ns);
fprintf(stderr, "[sdq-prof] 自訂 MoE op(%llu 次呼叫、%llu 個 expert 分頁)\n",
(unsigned long long) g_prof.calls.load(), (unsigned long long) g_prof.experts.load());
fprintf(stderr, "[sdq-prof] op 合計 %9.1f ms\n", op);
fprintf(stderr, "[sdq-prof] expert_forward %9.1f ms (%5.1f%%) 其中:\n", fwd, op > 0 ? 100.0 * fwd / op : 0.0);
fprintf(stderr, "[sdq-prof] 量化 activation%9.1f ms (%5.1f%% of fwd)\n", v(g_prof.qact_ns), fwd > 0 ? 100.0 * v(g_prof.qact_ns) / fwd : 0.0);
fprintf(stderr, "[sdq-prof] 三次點積 %9.1f ms (%5.1f%% of fwd)\n", v(g_prof.dot_ns), fwd > 0 ? 100.0 * v(g_prof.dot_ns) / fwd : 0.0);
fprintf(stderr, "[sdq-prof] SiLU+加權 %9.1f ms (%5.1f%% of fwd)\n", v(g_prof.act_ns), fwd > 0 ? 100.0 * v(g_prof.act_ns) / fwd : 0.0);
fprintf(stderr, "[sdq-prof] SSD 讀取/分頁 %9.1f ms (%5.1f%%)\n", v(g_prof.read_ns), op > 0 ? 100.0 * v(g_prof.read_ns) / op : 0.0);
fprintf(stderr, "[sdq-prof] 前置 setup %9.1f ms (%5.1f%%)\n", v(g_prof.setup_ns), op > 0 ? 100.0 * v(g_prof.setup_ns) / op : 0.0);
fprintf(stderr, "[sdq-prof] barrier 等待 %9.1f ms (%5.1f%%)\n", v(g_prof.bar_ns), op > 0 ? 100.0 * v(g_prof.bar_ns) / op : 0.0);
}
// ---------------------------------------------------------------- graph 建構
ggml_tensor * sdq_build_moe(ggml_cgraph * gf, ggml_context * ctx, ggml_tensor * cur,
ggml_tensor * ids, ggml_tensor * weights, int il, int64_t n_expert,
int64_t n_expert_used, float swiglu_limit) {
layer_rec * R = rec_for(il);
R->n_expert = n_expert;
R->n_expert_used = n_expert_used;
R->swiglu_limit = swiglu_limit;
// 呼叫端給的是 2D [n_embd, n_tokens],這裡跟上游一樣先 reshape 成 3D
// [n_embd, 1, n_tokens](上游 mul_mat_id 路徑也是這樣拿 token 數的)
ggml_tensor * cur3 = (cur->ne[1] == 1) ? cur : ggml_reshape_3d(ctx, cur, cur->ne[0], 1, cur->ne[1]);
const int64_t n_tok = cur3->ne[2];
ggml_tensor * args[3] = { cur3, ids, weights };
ggml_tensor * out = ggml_custom_4d(ctx, GGML_TYPE_F32, cur3->ne[0], n_tok, 1, 1,
args, 3, moe_custom_op, GGML_N_TASKS_MAX, R);
ggml_set_name(out, "sdq_moe_out");
// 關鍵:把這個節點註冊進圖,ggml 才會為它配置獨立的 compute buffer。
// 少了這一行,layer>=1 的輸出會被其他張量覆蓋(layer 0 僥倖正確)。
ggml_build_forward_expand(gf, out);
return out;
}
} // namespace sdq
namespace sdq {
// 自我測試:對指定 (layer, expert) 用 pager 的資料算一次 MoE 輸出,
// 讓 Python 用純 NumPy 從同一段 GGUF 位元組重算並比對。
// 輸出:out[0..n_embd) = 輸出,out[n_embd..2*n_embd) = 輸入 activation
int sdq_moe_selftest(int il, int ie, const float * x, float * out_io,
int64_t n_embd, int64_t n_ff, int n_tok) {
pager & P = pager::instance();
const layer_desc * L = P.layer(il);
if (!L || ie < 0 || ie >= (int) L->experts.size()) {
fprintf(stderr, "[sdq-selftest] 層/expert 不存在:%d/%d\n", il, ie);
return 1;
}
const int slot = P.acquire(il, ie, true);
if (slot < 0) {
fprintf(stderr, "[sdq-selftest] 分頁失敗\n");
return 2;
}
const expert_desc & ed = L->experts[(size_t) ie];
if (ed.ne0 != n_embd || ed.ne1 != n_ff) {
fprintf(stderr, "[sdq-selftest] 維度不符:ne0=%lld ne1=%lld(預期 %lld/%lld)\n",
(long long) ed.ne0, (long long) ed.ne1, (long long) n_embd, (long long) n_ff);
return 3;
}
std::vector<std::pair<int32_t, int32_t>> toks;
std::vector<float> wts((size_t) n_tok, 1.0f);
for (int t = 0; t < n_tok; t++) {
toks.emplace_back(t, 0);
}
std::vector<float> out((size_t) n_embd * n_tok, 0.0f);
std::vector<uint8_t> scratch;
expert_forward(ed, P.slot_ptr(slot), x, n_embd, toks, wts.data(), 4, 4, 1, 0.0f,
out.data(), &scratch);
for (int64_t t = 0; t < n_tok; t++) {
for (int64_t i = 0; i < n_embd; i++) {
out_io[t * n_embd + i] = out[(size_t) t * n_embd + i];
out_io[n_tok * n_embd + t * n_embd + i] = x[(size_t) t * n_embd + i];
}
}
P.unpin_all();
fprintf(stderr, "[sdq-selftest] L%d E%d n_tok=%d 計算完成(gate %s / up %s / down %s)\n",
il, ie, n_tok, ggml_type_name(ed.type[part::GATE]), ggml_type_name(ed.type[part::UP]),
ggml_type_name(ed.type[part::DOWN]));
return 0;
}
} // namespace sdq