File size: 5,813 Bytes
c99f13f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
import re

GGUF_TYPE_NAMES = {
    0: "F32",
    1: "F16",
    2: "Q4_0", 3: "Q4_1", 6: "Q5_0", 7: "Q5_1",
    8: "Q8_0",
    10: "Q4_K", 11: "Q5_K", 12: "Q6_K",
    13: "Q5_K_M", 14: "Q4_K_M",
    15: "IQ4_XS", 16: "IQ4_NL",
    20: "IQ3_XXS",
    24: "IQ2_XXS",
    30: "IQ1_S",
}

GGUF_TYPE_NAMES_INV = {v: k for k, v in GGUF_TYPE_NAMES.items()}

# Ordered from worst to best quality
TIER_ORDER = [
    "IQ1_S", "IQ2_XXS", "IQ2_XS", "IQ2_S",
    "IQ3_XXS", "Q3_K", "IQ3_S",
    "IQ4_XS", "IQ4_NL", "Q4_K", "Q5_K", "Q6_K", "Q8_0", "F16",
]

# Exact bits per weight from ggml block structs (ggml_type_sizef * 8)
# Does NOT include GGUF overhead — GGUF_OVERHEAD_FACTOR is applied separately
TIER_BPW = {
    "IQ1_S": 1.5625,
    "IQ2_XXS": 2.0625,
    "IQ2_XS": 2.3125,
    "IQ2_S": 2.5,
    "IQ3_XXS": 3.0625,
    "Q3_K": 3.4375,
    "IQ3_S": 3.44,
    "IQ4_XS": 4.25,
    "IQ4_NL": 4.5,
    "Q4_K": 4.5,
    "Q5_K": 5.5,
    "Q6_K": 6.5625,
    "Q8_0": 8.5,
    "F16": 16.0,
}
GGUF_OVERHEAD_FACTOR = 1.0

# Quality retention per tier (1.0 = F16, no loss). Used for non-linear utility.
QUALITY_WEIGHTS = {
    "F16": 1.000,
    "Q8_0": 0.995,
    "Q6_K": 0.990,
    "Q5_K": 0.978,
    "IQ4_XS": 0.940,
    "IQ4_NL": 0.945,
    "Q4_K": 0.960,
    "Q3_K": 0.900,
}

# Quant quality rank: higher = better (same order as TIER_ORDER)
QUANT_RANK = {tier: i for i, tier in enumerate(TIER_ORDER)}

# Per-class hard floor — NEVER go below this without --allow-q3-or-lower
# Matches hand-tuned v2 patterns: gate=Q6_K, attn_proj=Q6_K, ffn=IQ4_XS
CLASS_HARD_FLOORS = {
    "gate": "Q8_0",
    "attn_proj": "Q8_0",
    "ffn_gate_up": "IQ4_XS",
    "ffn_down": "Q6_K",
    "norms": "F16",
    "ssm_params": "F16",
    "mtp": "IQ4_XS",
    "embd": "Q5_K",
}

# Per-class start tier — where greedy begins (Q5_K for all, like OptA base type)
# Greedy upgrades from here toward CLASS_MAX_TIER
CLASS_START_TIER = {
    "gate": "Q8_0",
    "attn_proj": "Q8_0",
    "ffn_gate_up": "Q5_K",
    "ffn_down": "Q6_K",
    "norms": "F16",
    "ssm_params": "F16",
    "mtp": "Q5_K",
    "embd": "Q5_K",
}

# Per-class max tier — never exceed this (for greedy upgrades)
# Deep layers can be upgraded to Q8_0 for important tensors
CLASS_MAX_TIER = {
    "gate": "Q8_0",
    "attn_proj": "Q8_0",
    "ffn_gate_up": "Q8_0",
    "ffn_down": "Q8_0",
    "norms": "F16",
    "ssm_params": "F16",
    "mtp": "Q8_0",
    "embd": "Q5_K",
}

# Classes that can go to Q3_K when --allow-q3-or-lower is set
CAN_Q3 = {"ffn_gate", "ffn_up", "ffn_down", "attn_output", "ssm_out"}
ALLOW_LOWER_FLOOR = "IQ2_XXS"  # lowest starting tier for --allow-q3-or-lower

# Default floor for unclassified tensors / unknown class
DEFAULT_FLOOR = "Q4_K"

# Tier for MTP head deployment
MTP_DEPLOY_TIER = "Q8_0"

# Preferred tier for extreme importance spikes (>50% of total importance)
SPIKE_IMPORTANCE_RATIO = 0.50
SPIKE_PREFERRED_TIER = "F16"

TENSOR_CLASS = {
    # Qwen 3.5 hybrid
    "attn_gate": "gate",
    "ssm_alpha": "gate",
    "ssm_beta": "gate",
    "attn_q": "attn_proj",
    "attn_k": "attn_proj",
    "attn_v": "attn_proj",
    "attn_qkv": "attn_proj",
    "attn_output": "attn_proj",
    "ffn_gate": "ffn_gate_up",
    "ffn_up": "ffn_gate_up",
    "ffn_down": "ffn_down",
    "ssm_out": "ffn_down",
    "ssm_conv1d": "norms",
    "router": "norms",
    "ssm_dt": "ssm_params",
    "ssm_a": "ssm_params",
    "nextn": "mtp",
    "ffn_gate_exps": "ffn_gate_up",
    "ffn_up_exps": "ffn_gate_up",
    "ffn_down_exps": "ffn_down",
    "ffn_gate_inp": "norms",

    # Standard llama.cpp tensor names
    "q_proj": "attn_proj",
    "k_proj": "attn_proj",
    "v_proj": "attn_proj",
    "o_proj": "attn_proj",
    "gate_proj": "ffn_gate_up",
    "up_proj": "ffn_gate_up",
    "down_proj": "ffn_down",
}

ARCH_FEATURES = {
    "qwen35": {
        "has_qkv": True,
        "has_ssm": True,
        "has_mtp": True,
        "has_moe": False,
        "is_qat": False,
        "prefix": "blk",
        "n_layers": 32,
    },
    "mellum2": {
        "has_qkv": False,
        "has_ssm": False,
        "has_mtp": False,
        "has_moe": True,
        "is_qat": False,
        "prefix": "blk",
        "n_layers": 28,
    },
    "gemma4": {
        "has_qkv": False,
        "has_ssm": False,
        "has_mtp": False,
        "has_moe": False,
        "is_qat": True,
        "prefix": "blk",
        "n_layers": 48,
    },
}


def strip_weight(name: str) -> str:
    return name.lstrip(".").removesuffix(".weight").removesuffix(".bias")


def get_tensor_type(name: str) -> str:
    parts = strip_weight(name).split(".")
    if len(parts) >= 2 and parts[0] in ("blk", "BLK"):
        return parts[2] if len(parts) >= 3 else "unknown"
    if "token_embd" in name:
        return "token_embd"
    if name.startswith("output") and "norm" not in name:
        return "output"
    return name


def get_tensor_class(ttype: str) -> str:
    if ttype in TENSOR_CLASS:
        return TENSOR_CLASS[ttype]
    if "norm" in ttype or "scale" in ttype:
        return "norms"
    if ttype.startswith("ssm_"):
        return "ssm_params"
    if ttype in ("token_embd", "output", "embed_tokens", "lm_head",
                 "vision_embedder", "audio_embedder"):
        return "embd"
    return "unknown"


def is_mtp_tensor(name: str, n_layers: int = 32) -> bool:
    if "nextn" in name:
        return True
    if n_layers > 40:
        return False  # Deep models (Gemma4, 48 layers) use separate drafter, not in-model MTP
    layer = get_layer_number(name)
    return layer is not None and layer >= n_layers


def get_layer_number(name: str) -> int | None:
    parts = strip_weight(name).split(".")
    if len(parts) >= 2 and parts[0] in ("blk", "BLK"):
        try:
            return int(parts[1])
        except ValueError:
            return None
    return None