File size: 2,236 Bytes
4809d46 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 | From 3e894f75872534b907b19175b6d2f3cdde1490a6 Mon Sep 17 00:00:00 2001
From: Gerald Corzo Perez <gerald.corzo@gmail.com>
Date: Fri, 21 Aug 2026 01:04:01 +0200
Subject: [PATCH] =?UTF-8?q?fix:=20gate=20K<4=20crash=20=E2=80=94=20views/a?=
=?UTF-8?q?ggregation=20loops=20must=20use=20gated=20n=5Fexpert=5Fused,=20?=
=?UTF-8?q?not=20hparams=20value=20(ggml=5Fview=5F2d=20out-of-bounds=20ass?=
=?UTF-8?q?ert)?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
---
src/llama-graph.cpp | 13 +++++++------
1 file changed, 7 insertions(+), 6 deletions(-)
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index 253998c59..4c1161110 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -2263,25 +2263,26 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
assert(n_expert_used > 0);
// order the views before the adds
- for (uint32_t i = 0; i < hparams.n_expert_used; ++i) {
+ // REMORA: use the (possibly gated) n_expert_used so the views match the
+ // gated `experts` shape (native hparams.n_expert_used would exceed it)
+ for (uint32_t i = 0; i < (uint32_t) n_expert_used; ++i) {
cur_experts[i] = ggml_view_2d(ctx0, experts, n_embd, n_tokens, experts->nb[2], i*experts->nb[1]);
ggml_build_forward_expand(gf, cur_experts[i]);
}
// aggregate experts
- // note: here we explicitly use hparams.n_expert_used instead of n_expert_used
- // to avoid potentially a large number of add nodes during warmup
- // ref: https://github.com/ggml-org/llama.cpp/pull/14753
+ // note: the gate (REMORA_GATE_K) may have reduced n_expert_used below the
+ // hparams value, so the aggregation must iterate the gated count
ggml_tensor * moe_out = cur_experts[0];
- for (uint32_t i = 1; i < hparams.n_expert_used; ++i) {
+ for (uint32_t i = 1; i < (uint32_t) n_expert_used; ++i) {
moe_out = ggml_add(ctx0, moe_out, cur_experts[i]);
ggml_build_forward_expand(gf, moe_out);
}
- if (hparams.n_expert_used == 1) {
+ if (n_expert_used == 1) {
// avoid returning a non-contiguous tensor
moe_out = ggml_cont(ctx0, moe_out);
}
--
2.50.1 (Apple Git-155)
|