File size: 11,119 Bytes
5e94f14
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
Step-5 Preview support for llama.cpp (stepfun-ai/Step-5-Preview-BF16)

Adds HF->GGUF conversion and load/inference support for the Step-5 Preview MoE
model by reusing the existing Step3p5 (STEP35) graph path, plus the Step3-VL
perception encoder for vision. Verified end-to-end on the 600B-A27B checkpoint:
bf16 convert -> Q3_K_M quant -> load + generate (text), plus a 4.1 GB vision
mmproj (projector_type=step3vl).

Base commit: ce8caa6e60a03093351d6016a818720e0d46f0fb (ce8caa6e6)

Apply on a clean checkout at or near the base commit:
    git apply -p1 step5-llamacpp.patch
    (or:  patch -p1 < step5-llamacpp.patch )

Files changed:
  conversion/__init__.py : route Step4ForCausalLM + MMGPTStepRoboticsForCausalLM
                           into the step3 converter (text map + mmproj map).
  conversion/base.py     : map the Step-5 tokenizer hash to the deepseek-v3
                           pre-tokenizer (identical BPE config to DeepSeek-V3).
  conversion/step3.py    : Step5Model (text) + Step5VisionModel (mmproj) on the
                           STEP35 arch; per-layer rope_theta chosen by layer_type;
                           emits rope.dimension_count / rope.dimension_count_swa so
                           the head_dim/3 partial RoPE is honoured; drops the
                           sparse-GQA indexer tensors for a dense-attention fallback.
  src/models/step35.cpp  : only halve n_rot_full when rope.dimension_count is absent,
                           so models that declare it (Step-5) are taken verbatim.

Usage:
    python convert_hf_to_gguf.py /path/to/Step5_safetensors --outtype bf16
    python convert_hf_to_gguf.py /path/to/Step5_safetensors --mmproj --outtype bf16

Notes / limitations (dense fallback, no sparse attention yet):
  * The sparse-GQA indexer (CSA block-compress + top-k over the full-attention
    layers) is not modelled; those tensors are dropped and the affected layers run
    dense attention (correct but slower). Remove the filter_tensors() hook in
    Step5Model once attention_impl=sparse_gqa exists in the graph builder.
  * MTP / NextN tensors convert through but are only used if a draft model is set.
  * 23 of 95 layers stay full (non-sparse) attention; long-context quality will
    differ from the reference until the indexer is implemented.

diff --git a/conversion/__init__.py b/conversion/__init__.py
index d48861e46..ca77c11f8 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -263,6 +263,8 @@ TEXT_MODEL_MAP: dict[str, str] = {
     "StableLmForCausalLM": "stablelm",
     "Starcoder2ForCausalLM": "starcoder",
     "Step3p5ForCausalLM": "step3",
+    "Step4ForCausalLM": "step3",
+    "MMGPTStepRoboticsForCausalLM": "step3",
     "StepVLForConditionalGeneration": "step3",
     "Step3p7ForConditionalGeneration": "step3",
     "T5EncoderModel": "t5",
@@ -345,6 +347,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
     "RADIOModel": "nemotron",
     "Sarashina2VisionForCausalLM": "sarashina2",
     "SmolVLMForConditionalGeneration": "smolvlm",
+    "MMGPTStepRoboticsForCausalLM": "step3",
     "StepVLForConditionalGeneration": "step3",
     "Step3p7ForConditionalGeneration": "step3",
     "UltravoxModel": "ultravox",
diff --git a/conversion/base.py b/conversion/base.py
index 6aca7f1d3..2912740a6 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -1762,6 +1762,11 @@ class TextModel(ModelBase):
         if chkhsh == "877081d19cf6996e2c4ff0e1236341e9b7bde288f5311a56a937f0afbbb3aeb5":
             # ref: https://huggingface.co/deepseek-ai/DeepSeek-V3
             res = "deepseek-v3"
+        if chkhsh == "5841594bd6a8eeecd7207aeec6570831cc97ffaeba51e908bdaf560113177bae":
+            # ref: https://huggingface.co/stepfun-ai/Step-5-Preview-BF16
+            # its tokenizer pre-tokenizer config is identical to DeepSeek-V3's, and
+            # stepfun-ai/Step-3.7-Flash-GGUF ships tokenizer.ggml.pre = deepseek-v3
+            res = "deepseek-v3"
         if chkhsh == "b3f499bb4255f8ca19fccd664443283318f2fd2414d5e0b040fbdd0cc195d6c5":
             # ref: https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B
             res = "deepseek-r1-qwen"
diff --git a/conversion/step3.py b/conversion/step3.py
index 93eb3134e..d4fcdb533 100644
--- a/conversion/step3.py
+++ b/conversion/step3.py
@@ -10,7 +10,7 @@ import torch
 if TYPE_CHECKING:
     from torch import Tensor
 
-from .base import MmprojModel, ModelBase, TextModel, _MISTRAL_COMMON_DATASET_MEAN, _MISTRAL_COMMON_DATASET_STD, gguf
+from .base import MmprojModel, ModelBase, TextModel, _MISTRAL_COMMON_DATASET_MEAN, _MISTRAL_COMMON_DATASET_STD, gguf, logger
 
 from .qwen import Qwen3Model
 
@@ -132,12 +132,22 @@ class Step35Model(TextModel):
         return super().index_tensors(remote_hf_model_id=remote_hf_model_id)
 
     def set_gguf_parameters(self):
+        # Step3p5 checkpoints can carry a per-layer rope_theta list. llama.cpp models a
+        # single base for the full_attention layers plus one for the sliding_attention
+        # layers, so pick the value each layer type actually uses rather than the first
+        # two entries of the list (Step3p5/Step3.7 differ from Step-5 here: the latter
+        # uses 1e7 on full_attention and 1e4 on sliding_attention).
         rope_theta = self.hparams.get("rope_theta")
         if isinstance(rope_theta, list):
-            self.hparams["rope_theta"] = float(rope_theta[0])
-            self.hparams["local_rope_theta"] = float(rope_theta[1])
-            self.rope_parameters["rope_theta"] = self.hparams["rope_theta"]
-            self.rope_parameters["sliding_attention"] = {"rope_theta": self.hparams["local_rope_theta"]}
+            theta_by_type: dict[str, float] = {}
+            for lt, theta in zip(self.hparams.get("layer_types") or [], rope_theta):
+                theta_by_type.setdefault(lt, float(theta))
+            full_theta = theta_by_type.get("full_attention", float(rope_theta[0]))
+            swa_theta = theta_by_type.get("sliding_attention", full_theta)
+            self.hparams["rope_theta"] = full_theta
+            self.hparams["local_rope_theta"] = swa_theta
+            self.rope_parameters["rope_theta"] = full_theta
+            self.rope_parameters["sliding_attention"] = {"rope_theta": swa_theta}
 
         super().set_gguf_parameters()
 
@@ -164,13 +174,15 @@ class Step35Model(TextModel):
                 arr = arr + [default] * (n - len(arr))
             return arr[:n]
 
-        layer_types = _pad(layer_types, self.block_count, "full_attention")
-        partial_rotary_factors = _pad(
-            partial_rotary_factors,
-            self.block_count,
-            0.5,  # full_attention default for Step3p5
+        # Rotary fraction used by the full_attention layers: 1/2 on Step3p5 and
+        # Step3.7-Flash, 1/3 on Step-5. sliding_attention layers stay fully rotary.
+        full_rotary_factor = next(
+            (float(f) for lt, f in zip(layer_types, partial_rotary_factors) if lt == "full_attention"),
+            0.5,
         )
-        assert [1.0 if lt == "sliding_attention" else 0.5 for lt in layer_types] == partial_rotary_factors
+        layer_types = _pad(layer_types, self.block_count, "full_attention")
+        partial_rotary_factors = _pad(partial_rotary_factors, self.block_count, full_rotary_factor)
+        assert [1.0 if lt == "sliding_attention" else full_rotary_factor for lt in layer_types] == partial_rotary_factors
         head_arr = [n_head_swa if lt == "sliding_attention" else n_head_base for lt in layer_types]
         kv_arr = [n_kv_swa if lt == "sliding_attention" else n_kv_base for lt in layer_types]
         swa_pat = [lt == "sliding_attention" for lt in layer_types]
@@ -183,6 +195,13 @@ class Step35Model(TextModel):
 
         self.gguf_writer.add_value_length(self.hparams["head_dim"])
 
+        # Per-layer RoPE dims: full_attention layers rotate only part of head_dim, the
+        # sliding_attention layers rotate all of it. Without these keys llama.cpp falls
+        # back to its Step3p5 default of head_dim/2 for the full-attention layers.
+        head_dim = int(self.hparams["head_dim"])
+        self.gguf_writer.add_rope_dimension_count(int(head_dim * full_rotary_factor))
+        self.gguf_writer.add_rope_dimension_count_swa(head_dim)
+
         # MoE params
         self.gguf_writer.add_expert_count(self.hparams["moe_num_experts"])
         self.gguf_writer.add_expert_used_count(self.hparams["moe_top_k"])
@@ -339,3 +358,32 @@ class Step35Model(TextModel):
             rope_factors.extend([1.0] * (storage_dim // 2 - len(rope_factors)))
 
         yield (self.format_tensor_name(gguf.MODEL_TENSOR.ROPE_FREQS), torch.tensor(rope_factors, dtype=torch.float32))
+
+
+@ModelBase.register("MMGPTStepRoboticsForCausalLM")
+@ModelBase.example("stepfun-ai/Step-5-Preview-BF16")
+class Step5VisionModel(Step3VLVisionModel):
+    """Step-5 reuses the Step3-VL perception encoder (728px / patch 14 / width 1536 /
+    47 layers) with the same stride-2 downsampler pair and vit_large_projector."""
+
+
+@ModelBase.register("Step4ForCausalLM", "MMGPTStepRoboticsForCausalLM")
+@ModelBase.example("stepfun-ai/Step-5-Preview-BF16")
+class Step5Model(Step35Model):
+    model_arch = gguf.MODEL_ARCH.STEP35
+    """Step-5 text model: the Step3p5 trunk with 1/3 partial RoPE on the full-attention
+    layers, plus a sparse-GQA indexer (CSA block compression + top-k) on those layers.
+
+    The indexer is not modelled by llama.cpp yet, so its tensors are dropped here and
+    the affected layers fall back to dense attention. Drop this filter once
+    attention_impl=sparse_gqa is implemented in the graph builder."""
+
+    @classmethod
+    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+        name, gen = item
+
+        if ".sparse_indexer" in name or name.endswith(".ssmax_s"):
+            logger.warning(f"dropping unsupported sparse-attention tensor (dense fallback): {name}")
+            return None
+
+        return super().filter_tensors(item)
diff --git a/src/models/step35.cpp b/src/models/step35.cpp
index ca68855d8..d29703682 100644
--- a/src/models/step35.cpp
+++ b/src/models/step35.cpp
@@ -5,8 +5,14 @@ void llama_model_step35::load_arch_hparams(llama_model_loader & ml) {
 
     hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
 
-    // full_attention layer only use half of the RoPE dimensions
-    hparams.n_rot_full = hparams.n_rot_full / 2;
+    // Step3p5 / Step3.7-Flash leave rope.dimension_count unset and use half of head_dim
+    // on the full-attention layers. Models that declare the value explicitly (Step-5
+    // uses head_dim/3) are taken verbatim. rope.dimension_count_swa is already applied
+    // to n_rot_swa by llama_model::load_hparams before this hook runs.
+    uint32_t n_rot_declared = 0;
+    if (!ml.get_key(LLM_KV_ROPE_DIMENSION_COUNT, n_rot_declared, false)) {
+        hparams.n_rot_full = hparams.n_rot_full / 2;
+    }
 
     // MoE + SWA parameters
     ml.get_key_or_arr(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp_arr, hparams.n_layer_all);