// The six exclusion conventions, ported from tools/quantize/pack_check.py. // // A second implementation of rules that took a day and four wrong // verdicts to settle. It is kept honest mechanically rather than by // care: conventions.cases.json holds the inputs, the Python side // generates conventions.golden.json from them, and parity.mjs fails // this file the moment the two disagree. // // Conventions read out of vLLM 0.23.1rc1.dev552+g4559c43a9 on // 2026-08-11. They are not a stable public API upstream. export const KNOWN_METHODS = new Set([ 'compressed-tensors', 'compressed_tensors', 'awq', 'gptq', 'gptq_marlin', 'bitsandbytes', 'bnb', 'modelopt', 'modelopt_fp4', 'modelopt_mxfp8', 'modelopt_mixed', 'auto-round', 'auto_round', 'intel/auto-round', ]); // Tensor name tails that exist only for a quantized weight. const QUANTIZED_SUFFIXES = new Set([ 'weight_packed', 'weight_scale', 'weight_shape', 'weight_zero_point', 'weight_global_scale', 'qweight', 'qzeros', 'scales', 'g_idx', ]); const WEIGHT_TAILS = new Set([...QUANTIZED_SUFFIXES, 'weight', 'bias']); // Where each convention keeps its exclusion list. export const EXCLUSION_FIELD = { 'compressed-tensors': 'ignore', 'compressed_tensors': 'ignore', 'awq': 'modules_to_not_convert', 'gptq': 'dynamic', 'gptq_marlin': 'dynamic', 'bitsandbytes': 'llm_int8_skip_modules', 'bnb': 'llm_int8_skip_modules', 'modelopt': 'exclude_modules', 'modelopt_fp4': 'exclude_modules', 'modelopt_mxfp8': 'exclude_modules', 'modelopt_mixed': 'exclude_modules', 'auto-round': 'extra_config', 'auto_round': 'extra_config', }; /** * Name the matching convention this config actually uses. * * `quant_method` selects vLLM's config class, so it decides the * convention. Falling back to `format` is wrong: a compressed-tensors * pack writes `format: "pack-quantized"`, a name matching no branch, * which would report every correct pack as unexempted. */ export function resolveMethod(quant) { const method = String(quant?.quant_method ?? '').trim(); if (method) return method; if ('config_groups' in quant || 'ignore' in quant) return 'compressed-tensors'; if ('modules_to_not_convert' in quant) return 'awq'; if ('dynamic' in quant || 'modules_in_block_to_quantize' in quant) return 'gptq'; if ('llm_int8_skip_modules' in quant) return 'bitsandbytes'; if ('exclude_modules' in quant || 'ignored_layers' in quant) return 'modelopt'; if ('extra_config' in quant || 'block_name_to_quantize' in quant) return 'auto-round'; return Object.keys(quant).length === 0 ? 'none' : 'unrecognised'; } /** * Start-anchored match, the way every runtime here does it. * * compressed-tensors and gptq both use Python's `re.match`, which * anchors at the start but not the end. `^(?:…)` reproduces that. * `re:layers.0` does NOT match `mtp.layers.0.…`; `re:.*layers.0` does. */ function regexMatch(pattern, module) { try { return new RegExp('^(?:' + pattern + ')').test(module); } catch { return false; // an invalid pattern matches nothing, as in Python } } /** * Would this pack's config exempt `module` from quantization? * * Each branch reproduces what the vLLM loader for that method does. * Reading one convention into another is how a checker invents * verdicts: a bare `mtp` covers the whole head under awq's substring * rule and covers nothing under compressed-tensors' exact-match rule. */ export function covers(quant, method, module) { if (method === 'compressed-tensors' || method === 'compressed_tensors') { for (const raw of quant.ignore ?? []) { const entry = String(raw); if (entry.startsWith('re:')) { if (regexMatch(entry.slice(3), module)) return true; } else if (entry === module) { return true; } } return false; } if (method === 'awq') { return (quant.modules_to_not_convert ?? []) .some((e) => module.includes(String(e))); } if (method === 'gptq' || method === 'gptq_marlin') { // Ordered, and the FIRST match wins: a positive rule ahead of a // `-:` rule stops the search and the exclusion never applies. for (const pattern of Object.keys(quant.dynamic ?? {})) { if (pattern.startsWith('-:')) { if (regexMatch(pattern.slice(2), module)) return true; } else if (regexMatch(pattern.replace(/^\+:/, ''), module)) { return false; } } return false; } if (method === 'bitsandbytes' || method === 'bnb') { const parts = module.split('.'); const prefixes = new Set( parts.map((_, i) => parts.slice(0, i + 1).join('.'))); const skip = (quant.llm_int8_skip_modules ?? []).map(String); return skip.some((e) => parts.includes(e) || prefixes.has(e)); } if (method.startsWith('modelopt')) { const entries = quant.exclude_modules?.length ? quant.exclude_modules : quant.ignore?.length ? quant.ignore : quant.ignored_layers?.length ? quant.ignored_layers : []; for (const raw of entries) { const entry = String(raw); if (entry === module || module.includes(entry)) return true; if (fnmatch(module, entry)) return true; } return false; } if (method === 'auto-round' || method === 'auto_round' || method === 'intel/auto-round') { for (const [name, override] of Object.entries(quant.extra_config ?? {})) { if (name === module && (override?.bits ?? 4) >= 16) return true; } let allow = quant.block_name_to_quantize || quant.to_quant_block_names; if (allow) { if (typeof allow === 'string') allow = allow.split(','); return !allow.some((p) => module.startsWith(String(p))); } return false; } // An unrecognised method is reported as uncovered rather than // guessed at, so a new format shows up as a gap, not a verdict. return false; } /** Shell-glob match, as Python's fnmatch does it — `*` crosses dots. */ function fnmatch(name, pattern) { let out = ''; for (let i = 0; i < pattern.length; i += 1) { const c = pattern[i]; if (c === '*') out += '.*'; else if (c === '?') out += '.'; else if (c === '[') { const close = pattern.indexOf(']', i + 1); if (close < 0) { out += '\\['; } else { let body = pattern.slice(i + 1, close); i = close; if (body.startsWith('!')) body = '^' + body.slice(1); out += '[' + body + ']'; } } else out += c.replace(/[.+^${}()|\\]/g, '\\$&'); } try { return new RegExp('^(?:' + out + ')$').test(name); } catch { return false; } } /** Reduce tensor names to the modules that own them. */ export function modulePaths(names) { const out = new Set(); for (const name of names) { const i = name.lastIndexOf('.'); const head = i < 0 ? '' : name.slice(0, i); const tail = i < 0 ? name : name.slice(i + 1); out.add(head && WEIGHT_TAILS.has(tail) ? head : name); } return out; } /** * Every module path in the pack, parents included. * * An exclusion list is written against the module tree of the loaded * model, where a container like `…layers.0.linear_attn` is a real * module owning no weight of its own. Checking against leaf paths * alone called 48 correct entries unmatched on a pack measured at * 86.5% draft acceptance. */ export function moduleTree(names) { const out = new Set(); for (const name of names) { const parts = name.split('.'); for (let i = 1; i < parts.length; i += 1) { out.add(parts.slice(0, i).join('.')); } } return out; } /** * Do these two paths name the same module at different depths? * * An exclusion list is written against the names of the loaded model, * and `save_pretrained` can drop a wrapper level: a pack listing * `model.vision_tower.…` stores its tensors as `vision_tower.…`. One * level, and it made 171 correct entries look unmatched. Comparing on * a dot-boundary suffix keeps that precise — it never matches a * different module that merely shares a tail fragment. */ export function sameModule(entry, module) { if (entry === module) return true; const [longer, shorter] = entry.length > module.length ? [entry, module] : [module, entry]; return longer.endsWith('.' + shorter); } /** * Exclusion entries that protect nothing in this pack. * * Asks a different question from `covers`. That one decides whether * the runtime treats a module as excluded, and answers strictly. This * asks whether the entry refers to anything real at all, so it accepts * the wrapper-level difference above. */ export function unmatchedExclusions(quant, method, modules, tied = false) { const field = EXCLUSION_FIELD[method]; if (!field) return []; const raw = quant[field]; if (!raw || (Array.isArray(raw) && raw.length === 0) || (!Array.isArray(raw) && Object.keys(raw).length === 0)) return []; const entries = Array.isArray(raw) ? raw : Object.keys(raw); const list = [...modules]; const stray = []; for (const rawEntry of entries) { const text = String(rawEntry); // With tied embeddings there is no separate lm_head tensor, so an // lm_head entry matches nothing and protects nothing — correctly. // Reporting it would train a reader to ignore this check. if (tied && text.replace(/[.*]+$/, '').endsWith('lm_head')) continue; let hit; if (text.startsWith('re:') || text.startsWith('-:') || text.startsWith('+:')) { // A regex is only ever applied to the runtime's own module // names, so it gets the strict test and no suffix leniency. const solo = Array.isArray(raw) ? { [field]: [rawEntry] } : { [field]: { [rawEntry]: raw[rawEntry] } }; hit = list.some((m) => covers(solo, method, m)); } else { hit = list.some((m) => sameModule(text, m)); } if (!hit) stray.push(text); } return stray; }