Step-5 Preview support for llama.cpp (stepfun-ai/Step-5-Preview-BF16) Adds HF->GGUF conversion and load/inference support for the Step-5 Preview MoE model by reusing the existing Step3p5 (STEP35) graph path, plus the Step3-VL perception encoder for vision. Verified end-to-end on the 600B-A27B checkpoint: bf16 convert -> Q3_K_M quant -> load + generate (text), plus a 4.1 GB vision mmproj (projector_type=step3vl). Base commit: ce8caa6e60a03093351d6016a818720e0d46f0fb (ce8caa6e6) Apply on a clean checkout at or near the base commit: git apply -p1 step5-llamacpp.patch (or: patch -p1 < step5-llamacpp.patch ) Files changed: conversion/__init__.py : route Step4ForCausalLM + MMGPTStepRoboticsForCausalLM into the step3 converter (text map + mmproj map). conversion/base.py : map the Step-5 tokenizer hash to the deepseek-v3 pre-tokenizer (identical BPE config to DeepSeek-V3). conversion/step3.py : Step5Model (text) + Step5VisionModel (mmproj) on the STEP35 arch; per-layer rope_theta chosen by layer_type; emits rope.dimension_count / rope.dimension_count_swa so the head_dim/3 partial RoPE is honoured; drops the sparse-GQA indexer tensors for a dense-attention fallback. src/models/step35.cpp : only halve n_rot_full when rope.dimension_count is absent, so models that declare it (Step-5) are taken verbatim. Usage: python convert_hf_to_gguf.py /path/to/Step5_safetensors --outtype bf16 python convert_hf_to_gguf.py /path/to/Step5_safetensors --mmproj --outtype bf16 Notes / limitations (dense fallback, no sparse attention yet): * The sparse-GQA indexer (CSA block-compress + top-k over the full-attention layers) is not modelled; those tensors are dropped and the affected layers run dense attention (correct but slower). Remove the filter_tensors() hook in Step5Model once attention_impl=sparse_gqa exists in the graph builder. * MTP / NextN tensors convert through but are only used if a draft model is set. * 23 of 95 layers stay full (non-sparse) attention; long-context quality will differ from the reference until the indexer is implemented. diff --git a/conversion/__init__.py b/conversion/__init__.py index d48861e46..ca77c11f8 100644 --- a/conversion/__init__.py +++ b/conversion/__init__.py @@ -263,6 +263,8 @@ TEXT_MODEL_MAP: dict[str, str] = { "StableLmForCausalLM": "stablelm", "Starcoder2ForCausalLM": "starcoder", "Step3p5ForCausalLM": "step3", + "Step4ForCausalLM": "step3", + "MMGPTStepRoboticsForCausalLM": "step3", "StepVLForConditionalGeneration": "step3", "Step3p7ForConditionalGeneration": "step3", "T5EncoderModel": "t5", @@ -345,6 +347,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = { "RADIOModel": "nemotron", "Sarashina2VisionForCausalLM": "sarashina2", "SmolVLMForConditionalGeneration": "smolvlm", + "MMGPTStepRoboticsForCausalLM": "step3", "StepVLForConditionalGeneration": "step3", "Step3p7ForConditionalGeneration": "step3", "UltravoxModel": "ultravox", diff --git a/conversion/base.py b/conversion/base.py index 6aca7f1d3..2912740a6 100644 --- a/conversion/base.py +++ b/conversion/base.py @@ -1762,6 +1762,11 @@ class TextModel(ModelBase): if chkhsh == "877081d19cf6996e2c4ff0e1236341e9b7bde288f5311a56a937f0afbbb3aeb5": # ref: https://huggingface.co/deepseek-ai/DeepSeek-V3 res = "deepseek-v3" + if chkhsh == "5841594bd6a8eeecd7207aeec6570831cc97ffaeba51e908bdaf560113177bae": + # ref: https://huggingface.co/stepfun-ai/Step-5-Preview-BF16 + # its tokenizer pre-tokenizer config is identical to DeepSeek-V3's, and + # stepfun-ai/Step-3.7-Flash-GGUF ships tokenizer.ggml.pre = deepseek-v3 + res = "deepseek-v3" if chkhsh == "b3f499bb4255f8ca19fccd664443283318f2fd2414d5e0b040fbdd0cc195d6c5": # ref: https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B res = "deepseek-r1-qwen" diff --git a/conversion/step3.py b/conversion/step3.py index 93eb3134e..d4fcdb533 100644 --- a/conversion/step3.py +++ b/conversion/step3.py @@ -10,7 +10,7 @@ import torch if TYPE_CHECKING: from torch import Tensor -from .base import MmprojModel, ModelBase, TextModel, _MISTRAL_COMMON_DATASET_MEAN, _MISTRAL_COMMON_DATASET_STD, gguf +from .base import MmprojModel, ModelBase, TextModel, _MISTRAL_COMMON_DATASET_MEAN, _MISTRAL_COMMON_DATASET_STD, gguf, logger from .qwen import Qwen3Model @@ -132,12 +132,22 @@ class Step35Model(TextModel): return super().index_tensors(remote_hf_model_id=remote_hf_model_id) def set_gguf_parameters(self): + # Step3p5 checkpoints can carry a per-layer rope_theta list. llama.cpp models a + # single base for the full_attention layers plus one for the sliding_attention + # layers, so pick the value each layer type actually uses rather than the first + # two entries of the list (Step3p5/Step3.7 differ from Step-5 here: the latter + # uses 1e7 on full_attention and 1e4 on sliding_attention). rope_theta = self.hparams.get("rope_theta") if isinstance(rope_theta, list): - self.hparams["rope_theta"] = float(rope_theta[0]) - self.hparams["local_rope_theta"] = float(rope_theta[1]) - self.rope_parameters["rope_theta"] = self.hparams["rope_theta"] - self.rope_parameters["sliding_attention"] = {"rope_theta": self.hparams["local_rope_theta"]} + theta_by_type: dict[str, float] = {} + for lt, theta in zip(self.hparams.get("layer_types") or [], rope_theta): + theta_by_type.setdefault(lt, float(theta)) + full_theta = theta_by_type.get("full_attention", float(rope_theta[0])) + swa_theta = theta_by_type.get("sliding_attention", full_theta) + self.hparams["rope_theta"] = full_theta + self.hparams["local_rope_theta"] = swa_theta + self.rope_parameters["rope_theta"] = full_theta + self.rope_parameters["sliding_attention"] = {"rope_theta": swa_theta} super().set_gguf_parameters() @@ -164,13 +174,15 @@ class Step35Model(TextModel): arr = arr + [default] * (n - len(arr)) return arr[:n] - layer_types = _pad(layer_types, self.block_count, "full_attention") - partial_rotary_factors = _pad( - partial_rotary_factors, - self.block_count, - 0.5, # full_attention default for Step3p5 + # Rotary fraction used by the full_attention layers: 1/2 on Step3p5 and + # Step3.7-Flash, 1/3 on Step-5. sliding_attention layers stay fully rotary. + full_rotary_factor = next( + (float(f) for lt, f in zip(layer_types, partial_rotary_factors) if lt == "full_attention"), + 0.5, ) - assert [1.0 if lt == "sliding_attention" else 0.5 for lt in layer_types] == partial_rotary_factors + layer_types = _pad(layer_types, self.block_count, "full_attention") + partial_rotary_factors = _pad(partial_rotary_factors, self.block_count, full_rotary_factor) + assert [1.0 if lt == "sliding_attention" else full_rotary_factor for lt in layer_types] == partial_rotary_factors head_arr = [n_head_swa if lt == "sliding_attention" else n_head_base for lt in layer_types] kv_arr = [n_kv_swa if lt == "sliding_attention" else n_kv_base for lt in layer_types] swa_pat = [lt == "sliding_attention" for lt in layer_types] @@ -183,6 +195,13 @@ class Step35Model(TextModel): self.gguf_writer.add_value_length(self.hparams["head_dim"]) + # Per-layer RoPE dims: full_attention layers rotate only part of head_dim, the + # sliding_attention layers rotate all of it. Without these keys llama.cpp falls + # back to its Step3p5 default of head_dim/2 for the full-attention layers. + head_dim = int(self.hparams["head_dim"]) + self.gguf_writer.add_rope_dimension_count(int(head_dim * full_rotary_factor)) + self.gguf_writer.add_rope_dimension_count_swa(head_dim) + # MoE params self.gguf_writer.add_expert_count(self.hparams["moe_num_experts"]) self.gguf_writer.add_expert_used_count(self.hparams["moe_top_k"]) @@ -339,3 +358,32 @@ class Step35Model(TextModel): rope_factors.extend([1.0] * (storage_dim // 2 - len(rope_factors))) yield (self.format_tensor_name(gguf.MODEL_TENSOR.ROPE_FREQS), torch.tensor(rope_factors, dtype=torch.float32)) + + +@ModelBase.register("MMGPTStepRoboticsForCausalLM") +@ModelBase.example("stepfun-ai/Step-5-Preview-BF16") +class Step5VisionModel(Step3VLVisionModel): + """Step-5 reuses the Step3-VL perception encoder (728px / patch 14 / width 1536 / + 47 layers) with the same stride-2 downsampler pair and vit_large_projector.""" + + +@ModelBase.register("Step4ForCausalLM", "MMGPTStepRoboticsForCausalLM") +@ModelBase.example("stepfun-ai/Step-5-Preview-BF16") +class Step5Model(Step35Model): + model_arch = gguf.MODEL_ARCH.STEP35 + """Step-5 text model: the Step3p5 trunk with 1/3 partial RoPE on the full-attention + layers, plus a sparse-GQA indexer (CSA block compression + top-k) on those layers. + + The indexer is not modelled by llama.cpp yet, so its tensors are dropped here and + the affected layers fall back to dense attention. Drop this filter once + attention_impl=sparse_gqa is implemented in the graph builder.""" + + @classmethod + def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None: + name, gen = item + + if ".sparse_indexer" in name or name.endswith(".ssmax_s"): + logger.warning(f"dropping unsupported sparse-attention tensor (dense fallback): {name}") + return None + + return super().filter_tensors(item) diff --git a/src/models/step35.cpp b/src/models/step35.cpp index ca68855d8..d29703682 100644 --- a/src/models/step35.cpp +++ b/src/models/step35.cpp @@ -5,8 +5,14 @@ void llama_model_step35::load_arch_hparams(llama_model_loader & ml) { hparams.swa_type = LLAMA_SWA_TYPE_STANDARD; - // full_attention layer only use half of the RoPE dimensions - hparams.n_rot_full = hparams.n_rot_full / 2; + // Step3p5 / Step3.7-Flash leave rope.dimension_count unset and use half of head_dim + // on the full-attention layers. Models that declare the value explicitly (Step-5 + // uses head_dim/3) are taken verbatim. rope.dimension_count_swa is already applied + // to n_rot_swa by llama_model::load_hparams before this hook runs. + uint32_t n_rot_declared = 0; + if (!ml.get_key(LLM_KV_ROPE_DIMENSION_COUNT, n_rot_declared, false)) { + hparams.n_rot_full = hparams.n_rot_full / 2; + } // MoE + SWA parameters ml.get_key_or_arr(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp_arr, hparams.n_layer_all);