Bruce001/ASHQ1-bucket / constants.py
Bruce001's picture
download
raw
5.81 kB
import re
GGUF_TYPE_NAMES = {
0: "F32",
1: "F16",
2: "Q4_0", 3: "Q4_1", 6: "Q5_0", 7: "Q5_1",
8: "Q8_0",
10: "Q4_K", 11: "Q5_K", 12: "Q6_K",
13: "Q5_K_M", 14: "Q4_K_M",
15: "IQ4_XS", 16: "IQ4_NL",
20: "IQ3_XXS",
24: "IQ2_XXS",
30: "IQ1_S",
}
GGUF_TYPE_NAMES_INV = {v: k for k, v in GGUF_TYPE_NAMES.items()}
# Ordered from worst to best quality
TIER_ORDER = [
"IQ1_S", "IQ2_XXS", "IQ2_XS", "IQ2_S",
"IQ3_XXS", "Q3_K", "IQ3_S",
"IQ4_XS", "IQ4_NL", "Q4_K", "Q5_K", "Q6_K", "Q8_0", "F16",
]
# Exact bits per weight from ggml block structs (ggml_type_sizef * 8)
# Does NOT include GGUF overhead — GGUF_OVERHEAD_FACTOR is applied separately
TIER_BPW = {
"IQ1_S": 1.5625,
"IQ2_XXS": 2.0625,
"IQ2_XS": 2.3125,
"IQ2_S": 2.5,
"IQ3_XXS": 3.0625,
"Q3_K": 3.4375,
"IQ3_S": 3.44,
"IQ4_XS": 4.25,
"IQ4_NL": 4.5,
"Q4_K": 4.5,
"Q5_K": 5.5,
"Q6_K": 6.5625,
"Q8_0": 8.5,
"F16": 16.0,
}
GGUF_OVERHEAD_FACTOR = 1.0
# Quality retention per tier (1.0 = F16, no loss). Used for non-linear utility.
QUALITY_WEIGHTS = {
"F16": 1.000,
"Q8_0": 0.995,
"Q6_K": 0.990,
"Q5_K": 0.978,
"IQ4_XS": 0.940,
"IQ4_NL": 0.945,
"Q4_K": 0.960,
"Q3_K": 0.900,
}
# Quant quality rank: higher = better (same order as TIER_ORDER)
QUANT_RANK = {tier: i for i, tier in enumerate(TIER_ORDER)}
# Per-class hard floor — NEVER go below this without --allow-q3-or-lower
# Matches hand-tuned v2 patterns: gate=Q6_K, attn_proj=Q6_K, ffn=IQ4_XS
CLASS_HARD_FLOORS = {
"gate": "Q8_0",
"attn_proj": "Q8_0",
"ffn_gate_up": "IQ4_XS",
"ffn_down": "Q6_K",
"norms": "F16",
"ssm_params": "F16",
"mtp": "IQ4_XS",
"embd": "Q5_K",
}
# Per-class start tier — where greedy begins (Q5_K for all, like OptA base type)
# Greedy upgrades from here toward CLASS_MAX_TIER
CLASS_START_TIER = {
"gate": "Q8_0",
"attn_proj": "Q8_0",
"ffn_gate_up": "Q5_K",
"ffn_down": "Q6_K",
"norms": "F16",
"ssm_params": "F16",
"mtp": "Q5_K",
"embd": "Q5_K",
}
# Per-class max tier — never exceed this (for greedy upgrades)
# Deep layers can be upgraded to Q8_0 for important tensors
CLASS_MAX_TIER = {
"gate": "Q8_0",
"attn_proj": "Q8_0",
"ffn_gate_up": "Q8_0",
"ffn_down": "Q8_0",
"norms": "F16",
"ssm_params": "F16",
"mtp": "Q8_0",
"embd": "Q5_K",
}
# Classes that can go to Q3_K when --allow-q3-or-lower is set
CAN_Q3 = {"ffn_gate", "ffn_up", "ffn_down", "attn_output", "ssm_out"}
ALLOW_LOWER_FLOOR = "IQ2_XXS" # lowest starting tier for --allow-q3-or-lower
# Default floor for unclassified tensors / unknown class
DEFAULT_FLOOR = "Q4_K"
# Tier for MTP head deployment
MTP_DEPLOY_TIER = "Q8_0"
# Preferred tier for extreme importance spikes (>50% of total importance)
SPIKE_IMPORTANCE_RATIO = 0.50
SPIKE_PREFERRED_TIER = "F16"
TENSOR_CLASS = {
# Qwen 3.5 hybrid
"attn_gate": "gate",
"ssm_alpha": "gate",
"ssm_beta": "gate",
"attn_q": "attn_proj",
"attn_k": "attn_proj",
"attn_v": "attn_proj",
"attn_qkv": "attn_proj",
"attn_output": "attn_proj",
"ffn_gate": "ffn_gate_up",
"ffn_up": "ffn_gate_up",
"ffn_down": "ffn_down",
"ssm_out": "ffn_down",
"ssm_conv1d": "norms",
"router": "norms",
"ssm_dt": "ssm_params",
"ssm_a": "ssm_params",
"nextn": "mtp",
"ffn_gate_exps": "ffn_gate_up",
"ffn_up_exps": "ffn_gate_up",
"ffn_down_exps": "ffn_down",
"ffn_gate_inp": "norms",
# Standard llama.cpp tensor names
"q_proj": "attn_proj",
"k_proj": "attn_proj",
"v_proj": "attn_proj",
"o_proj": "attn_proj",
"gate_proj": "ffn_gate_up",
"up_proj": "ffn_gate_up",
"down_proj": "ffn_down",
}
ARCH_FEATURES = {
"qwen35": {
"has_qkv": True,
"has_ssm": True,
"has_mtp": True,
"has_moe": False,
"is_qat": False,
"prefix": "blk",
"n_layers": 32,
},
"mellum2": {
"has_qkv": False,
"has_ssm": False,
"has_mtp": False,
"has_moe": True,
"is_qat": False,
"prefix": "blk",
"n_layers": 28,
},
"gemma4": {
"has_qkv": False,
"has_ssm": False,
"has_mtp": False,
"has_moe": False,
"is_qat": True,
"prefix": "blk",
"n_layers": 48,
},
}
def strip_weight(name: str) -> str:
return name.lstrip(".").removesuffix(".weight").removesuffix(".bias")
def get_tensor_type(name: str) -> str:
parts = strip_weight(name).split(".")
if len(parts) >= 2 and parts[0] in ("blk", "BLK"):
return parts[2] if len(parts) >= 3 else "unknown"
if "token_embd" in name:
return "token_embd"
if name.startswith("output") and "norm" not in name:
return "output"
return name
def get_tensor_class(ttype: str) -> str:
if ttype in TENSOR_CLASS:
return TENSOR_CLASS[ttype]
if "norm" in ttype or "scale" in ttype:
return "norms"
if ttype.startswith("ssm_"):
return "ssm_params"
if ttype in ("token_embd", "output", "embed_tokens", "lm_head",
"vision_embedder", "audio_embedder"):
return "embd"
return "unknown"
def is_mtp_tensor(name: str, n_layers: int = 32) -> bool:
if "nextn" in name:
return True
if n_layers > 40:
return False # Deep models (Gemma4, 48 layers) use separate drafter, not in-model MTP
layer = get_layer_number(name)
return layer is not None and layer >= n_layers
def get_layer_number(name: str) -> int | None:
parts = strip_weight(name).split(".")
if len(parts) >= 2 and parts[0] in ("blk", "BLK"):
try:
return int(parts[1])
except ValueError:
return None
return None

Xet Storage Details

Size:
5.81 kB
·
Xet hash:
b2a3ca01a6b6996b087cacbbc86ee3b5f15b1218cd89cf8f8b744f818f8a6f20

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.