Step5-Preview / step5-llamacpp.patch
avar6's picture
Add Step-5 llama.cpp support patch
5e94f14 verified
Raw
History Blame Contribute Delete
11.1 kB
Step-5 Preview support for llama.cpp (stepfun-ai/Step-5-Preview-BF16)
Adds HF->GGUF conversion and load/inference support for the Step-5 Preview MoE
model by reusing the existing Step3p5 (STEP35) graph path, plus the Step3-VL
perception encoder for vision. Verified end-to-end on the 600B-A27B checkpoint:
bf16 convert -> Q3_K_M quant -> load + generate (text), plus a 4.1 GB vision
mmproj (projector_type=step3vl).
Base commit: ce8caa6e60a03093351d6016a818720e0d46f0fb (ce8caa6e6)
Apply on a clean checkout at or near the base commit:
git apply -p1 step5-llamacpp.patch
(or: patch -p1 < step5-llamacpp.patch )
Files changed:
conversion/__init__.py : route Step4ForCausalLM + MMGPTStepRoboticsForCausalLM
into the step3 converter (text map + mmproj map).
conversion/base.py : map the Step-5 tokenizer hash to the deepseek-v3
pre-tokenizer (identical BPE config to DeepSeek-V3).
conversion/step3.py : Step5Model (text) + Step5VisionModel (mmproj) on the
STEP35 arch; per-layer rope_theta chosen by layer_type;
emits rope.dimension_count / rope.dimension_count_swa so
the head_dim/3 partial RoPE is honoured; drops the
sparse-GQA indexer tensors for a dense-attention fallback.
src/models/step35.cpp : only halve n_rot_full when rope.dimension_count is absent,
so models that declare it (Step-5) are taken verbatim.
Usage:
python convert_hf_to_gguf.py /path/to/Step5_safetensors --outtype bf16
python convert_hf_to_gguf.py /path/to/Step5_safetensors --mmproj --outtype bf16
Notes / limitations (dense fallback, no sparse attention yet):
* The sparse-GQA indexer (CSA block-compress + top-k over the full-attention
layers) is not modelled; those tensors are dropped and the affected layers run
dense attention (correct but slower). Remove the filter_tensors() hook in
Step5Model once attention_impl=sparse_gqa exists in the graph builder.
* MTP / NextN tensors convert through but are only used if a draft model is set.
* 23 of 95 layers stay full (non-sparse) attention; long-context quality will
differ from the reference until the indexer is implemented.
diff --git a/conversion/__init__.py b/conversion/__init__.py
index d48861e46..ca77c11f8 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -263,6 +263,8 @@ TEXT_MODEL_MAP: dict[str, str] = {
"StableLmForCausalLM": "stablelm",
"Starcoder2ForCausalLM": "starcoder",
"Step3p5ForCausalLM": "step3",
+ "Step4ForCausalLM": "step3",
+ "MMGPTStepRoboticsForCausalLM": "step3",
"StepVLForConditionalGeneration": "step3",
"Step3p7ForConditionalGeneration": "step3",
"T5EncoderModel": "t5",
@@ -345,6 +347,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
"RADIOModel": "nemotron",
"Sarashina2VisionForCausalLM": "sarashina2",
"SmolVLMForConditionalGeneration": "smolvlm",
+ "MMGPTStepRoboticsForCausalLM": "step3",
"StepVLForConditionalGeneration": "step3",
"Step3p7ForConditionalGeneration": "step3",
"UltravoxModel": "ultravox",
diff --git a/conversion/base.py b/conversion/base.py
index 6aca7f1d3..2912740a6 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -1762,6 +1762,11 @@ class TextModel(ModelBase):
if chkhsh == "877081d19cf6996e2c4ff0e1236341e9b7bde288f5311a56a937f0afbbb3aeb5":
# ref: https://huggingface.co/deepseek-ai/DeepSeek-V3
res = "deepseek-v3"
+ if chkhsh == "5841594bd6a8eeecd7207aeec6570831cc97ffaeba51e908bdaf560113177bae":
+ # ref: https://huggingface.co/stepfun-ai/Step-5-Preview-BF16
+ # its tokenizer pre-tokenizer config is identical to DeepSeek-V3's, and
+ # stepfun-ai/Step-3.7-Flash-GGUF ships tokenizer.ggml.pre = deepseek-v3
+ res = "deepseek-v3"
if chkhsh == "b3f499bb4255f8ca19fccd664443283318f2fd2414d5e0b040fbdd0cc195d6c5":
# ref: https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B
res = "deepseek-r1-qwen"
diff --git a/conversion/step3.py b/conversion/step3.py
index 93eb3134e..d4fcdb533 100644
--- a/conversion/step3.py
+++ b/conversion/step3.py
@@ -10,7 +10,7 @@ import torch
if TYPE_CHECKING:
from torch import Tensor
-from .base import MmprojModel, ModelBase, TextModel, _MISTRAL_COMMON_DATASET_MEAN, _MISTRAL_COMMON_DATASET_STD, gguf
+from .base import MmprojModel, ModelBase, TextModel, _MISTRAL_COMMON_DATASET_MEAN, _MISTRAL_COMMON_DATASET_STD, gguf, logger
from .qwen import Qwen3Model
@@ -132,12 +132,22 @@ class Step35Model(TextModel):
return super().index_tensors(remote_hf_model_id=remote_hf_model_id)
def set_gguf_parameters(self):
+ # Step3p5 checkpoints can carry a per-layer rope_theta list. llama.cpp models a
+ # single base for the full_attention layers plus one for the sliding_attention
+ # layers, so pick the value each layer type actually uses rather than the first
+ # two entries of the list (Step3p5/Step3.7 differ from Step-5 here: the latter
+ # uses 1e7 on full_attention and 1e4 on sliding_attention).
rope_theta = self.hparams.get("rope_theta")
if isinstance(rope_theta, list):
- self.hparams["rope_theta"] = float(rope_theta[0])
- self.hparams["local_rope_theta"] = float(rope_theta[1])
- self.rope_parameters["rope_theta"] = self.hparams["rope_theta"]
- self.rope_parameters["sliding_attention"] = {"rope_theta": self.hparams["local_rope_theta"]}
+ theta_by_type: dict[str, float] = {}
+ for lt, theta in zip(self.hparams.get("layer_types") or [], rope_theta):
+ theta_by_type.setdefault(lt, float(theta))
+ full_theta = theta_by_type.get("full_attention", float(rope_theta[0]))
+ swa_theta = theta_by_type.get("sliding_attention", full_theta)
+ self.hparams["rope_theta"] = full_theta
+ self.hparams["local_rope_theta"] = swa_theta
+ self.rope_parameters["rope_theta"] = full_theta
+ self.rope_parameters["sliding_attention"] = {"rope_theta": swa_theta}
super().set_gguf_parameters()
@@ -164,13 +174,15 @@ class Step35Model(TextModel):
arr = arr + [default] * (n - len(arr))
return arr[:n]
- layer_types = _pad(layer_types, self.block_count, "full_attention")
- partial_rotary_factors = _pad(
- partial_rotary_factors,
- self.block_count,
- 0.5, # full_attention default for Step3p5
+ # Rotary fraction used by the full_attention layers: 1/2 on Step3p5 and
+ # Step3.7-Flash, 1/3 on Step-5. sliding_attention layers stay fully rotary.
+ full_rotary_factor = next(
+ (float(f) for lt, f in zip(layer_types, partial_rotary_factors) if lt == "full_attention"),
+ 0.5,
)
- assert [1.0 if lt == "sliding_attention" else 0.5 for lt in layer_types] == partial_rotary_factors
+ layer_types = _pad(layer_types, self.block_count, "full_attention")
+ partial_rotary_factors = _pad(partial_rotary_factors, self.block_count, full_rotary_factor)
+ assert [1.0 if lt == "sliding_attention" else full_rotary_factor for lt in layer_types] == partial_rotary_factors
head_arr = [n_head_swa if lt == "sliding_attention" else n_head_base for lt in layer_types]
kv_arr = [n_kv_swa if lt == "sliding_attention" else n_kv_base for lt in layer_types]
swa_pat = [lt == "sliding_attention" for lt in layer_types]
@@ -183,6 +195,13 @@ class Step35Model(TextModel):
self.gguf_writer.add_value_length(self.hparams["head_dim"])
+ # Per-layer RoPE dims: full_attention layers rotate only part of head_dim, the
+ # sliding_attention layers rotate all of it. Without these keys llama.cpp falls
+ # back to its Step3p5 default of head_dim/2 for the full-attention layers.
+ head_dim = int(self.hparams["head_dim"])
+ self.gguf_writer.add_rope_dimension_count(int(head_dim * full_rotary_factor))
+ self.gguf_writer.add_rope_dimension_count_swa(head_dim)
+
# MoE params
self.gguf_writer.add_expert_count(self.hparams["moe_num_experts"])
self.gguf_writer.add_expert_used_count(self.hparams["moe_top_k"])
@@ -339,3 +358,32 @@ class Step35Model(TextModel):
rope_factors.extend([1.0] * (storage_dim // 2 - len(rope_factors)))
yield (self.format_tensor_name(gguf.MODEL_TENSOR.ROPE_FREQS), torch.tensor(rope_factors, dtype=torch.float32))
+
+
+@ModelBase.register("MMGPTStepRoboticsForCausalLM")
+@ModelBase.example("stepfun-ai/Step-5-Preview-BF16")
+class Step5VisionModel(Step3VLVisionModel):
+ """Step-5 reuses the Step3-VL perception encoder (728px / patch 14 / width 1536 /
+ 47 layers) with the same stride-2 downsampler pair and vit_large_projector."""
+
+
+@ModelBase.register("Step4ForCausalLM", "MMGPTStepRoboticsForCausalLM")
+@ModelBase.example("stepfun-ai/Step-5-Preview-BF16")
+class Step5Model(Step35Model):
+ model_arch = gguf.MODEL_ARCH.STEP35
+ """Step-5 text model: the Step3p5 trunk with 1/3 partial RoPE on the full-attention
+ layers, plus a sparse-GQA indexer (CSA block compression + top-k) on those layers.
+
+ The indexer is not modelled by llama.cpp yet, so its tensors are dropped here and
+ the affected layers fall back to dense attention. Drop this filter once
+ attention_impl=sparse_gqa is implemented in the graph builder."""
+
+ @classmethod
+ def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+ name, gen = item
+
+ if ".sparse_indexer" in name or name.endswith(".ssmax_s"):
+ logger.warning(f"dropping unsupported sparse-attention tensor (dense fallback): {name}")
+ return None
+
+ return super().filter_tensors(item)
diff --git a/src/models/step35.cpp b/src/models/step35.cpp
index ca68855d8..d29703682 100644
--- a/src/models/step35.cpp
+++ b/src/models/step35.cpp
@@ -5,8 +5,14 @@ void llama_model_step35::load_arch_hparams(llama_model_loader & ml) {
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
- // full_attention layer only use half of the RoPE dimensions
- hparams.n_rot_full = hparams.n_rot_full / 2;
+ // Step3p5 / Step3.7-Flash leave rope.dimension_count unset and use half of head_dim
+ // on the full-attention layers. Models that declare the value explicitly (Step-5
+ // uses head_dim/3) are taken verbatim. rope.dimension_count_swa is already applied
+ // to n_rot_swa by llama_model::load_hparams before this hook runs.
+ uint32_t n_rot_declared = 0;
+ if (!ml.get_key(LLM_KV_ROPE_DIMENSION_COUNT, n_rot_declared, false)) {
+ hparams.n_rot_full = hparams.n_rot_full / 2;
+ }
// MoE + SWA parameters
ml.get_key_or_arr(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp_arr, hparams.n_layer_all);