quantization-explorer / index.html
Agenten's picture
Upload 2 files
762c004 verified
Raw History Blame Contribute Delete
21.8 kB
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<meta name="description" content="Explore AI model quantization: FP16, BF16, FP8, INT8, INT4, bitsandbytes, GPTQ, AWQ and GGUF. Compare memory, quality and deployment trade-offs.">
<meta name="theme-color" content="#07111f">
<title>Quantization Explorer — FP8, INT8, INT4 & Open-Weight Deployment</title>
<style>
:root{
--bg:#07111f;--panel:#0d1b2d;--panel2:#10233a;--text:#edf7ff;--muted:#9fb4c8;
--line:#24445f;--cyan:#55d9ff;--blue:#6b8cff;--green:#79f2c0;--gold:#ffd580;
--red:#ff9e9e;--shadow:0 18px 60px rgba(0,0,0,.25)
}
*{box-sizing:border-box}
html{scroll-behavior:smooth}
body{
margin:0;color:var(--text);
font:16px/1.65 Inter,ui-sans-serif,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;
background:
radial-gradient(circle at 12% 0%,rgba(85,217,255,.12),transparent 29%),
radial-gradient(circle at 90% 12%,rgba(107,140,255,.13),transparent 26%),
var(--bg)
}
a{color:var(--cyan);text-decoration:none}
a:hover{text-decoration:underline}
.wrap{max-width:1180px;margin:auto;padding:0 22px}
.hero{padding:72px 0 35px}
.badge{display:inline-flex;padding:7px 12px;border:1px solid var(--line);border-radius:999px;background:rgba(13,27,45,.75);color:#c8efff;font-size:14px}
h1{font-size:clamp(42px,7vw,78px);line-height:1;letter-spacing:-.055em;margin:20px 0;max-width:980px}
.gradient{background:linear-gradient(90deg,var(--cyan),#b6c3ff);-webkit-background-clip:text;background-clip:text;color:transparent}
.lead{font-size:clamp(18px,2.2vw,24px);max-width:900px;color:#cbdbe8;margin:0 0 28px}
.cta{display:flex;gap:12px;flex-wrap:wrap}
.btn{display:inline-block;padding:11px 16px;border:1px solid var(--line);border-radius:12px;font-weight:750}
.btn.primary{border:0;color:white;background:linear-gradient(135deg,#157aa8,#5269df)}
.quick{display:grid;grid-template-columns:repeat(4,1fr);gap:14px;margin:32px 0 56px}
.card,.section{border:1px solid var(--line);background:linear-gradient(180deg,rgba(16,35,58,.93),rgba(10,25,42,.93));border-radius:20px;box-shadow:var(--shadow)}
.card{padding:18px}
.card strong{display:block;font-size:23px;color:#fff}
.card span,.muted{color:var(--muted)}
.section{padding:28px;margin:22px 0}
.eyebrow{text-transform:uppercase;letter-spacing:.14em;font-size:12px;color:var(--cyan);font-weight:850}
h2{font-size:clamp(28px,4vw,45px);letter-spacing:-.03em;margin:6px 0 10px}
h3{font-size:22px;margin:0 0 8px}
.grid2{display:grid;grid-template-columns:1fr 1fr;gap:18px}
.grid3{display:grid;grid-template-columns:repeat(3,1fr);gap:14px}
.grid4{display:grid;grid-template-columns:repeat(4,1fr);gap:14px}
.callout{padding:15px 17px;border-left:4px solid var(--cyan);border-radius:9px;background:rgba(85,217,255,.06);margin:18px 0}
.warning{border-left-color:var(--gold);background:rgba(255,213,128,.06)}
.diagram{padding:22px;border:1px solid #2d5876;border-radius:17px;background:#091829;overflow:auto;margin:20px 0}
.flow{display:flex;gap:9px;align-items:center;min-width:900px}
.node{min-width:135px;padding:14px 12px;text-align:center;border:1px solid #34617d;background:#102842;border-radius:13px;font-weight:800}
.arrow{font-size:24px;color:var(--cyan)}
.tablewrap{overflow:auto}
table{width:100%;border-collapse:collapse;min-width:760px}
th,td{padding:13px;border-bottom:1px solid #23435e;text-align:left;vertical-align:top}
th{font-size:12px;text-transform:uppercase;letter-spacing:.07em;color:#c9efff}
.pill{display:inline-block;padding:5px 9px;margin:3px;border:1px solid #315b78;border-radius:999px;background:#102842;color:#c8ecff;font-size:13px}
.code{white-space:pre-wrap;padding:17px;border:1px solid #223f58;border-radius:14px;background:#06101c;color:#bfeeff;font:14px/1.6 ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;overflow:auto}
.barwrap{display:grid;grid-template-columns:repeat(5,1fr);gap:12px;align-items:end;height:220px;margin:28px 0 42px}
.bar{position:relative;border-radius:12px 12px 4px 4px;background:linear-gradient(180deg,var(--cyan),#5368de);min-height:28px}
.bar b{position:absolute;top:10px;left:0;right:0;text-align:center;color:#06111f}
.bar small{position:absolute;bottom:-30px;left:0;right:0;text-align:center;color:var(--muted)}
.controls{display:grid;grid-template-columns:1fr 1fr;gap:16px}
label{display:block;color:#c8efff;font-weight:700;margin-bottom:7px}
input,select,button{
width:100%;background:#0a1c2f;color:var(--text);border:1px solid #315b78;border-radius:11px;padding:11px 12px;font:inherit
}
.result{margin-top:18px;padding:18px;border:1px solid #315b78;background:#091a2b;border-radius:15px}
.big{font-size:34px;font-weight:850;color:white}
.tabs{display:flex;gap:9px;flex-wrap:wrap;margin:16px 0}
.tab{width:auto;cursor:pointer;border:1px solid #315b78;background:#0c2035;color:#d9f2ff;border-radius:999px;padding:8px 12px;font-weight:700}
.tab.active{background:linear-gradient(135deg,#177aa6,#4e66d8);border-color:transparent}
.answer{padding:18px;border:1px solid #315b78;border-radius:15px;background:#0a1c2f;min-height:112px}
.good{color:var(--green);font-weight:800}
.caution{color:var(--gold);font-weight:800}
.bad{color:var(--red);font-weight:800}
footer{padding:50px 0 68px;color:var(--muted)}
@media(max-width:820px){
.quick,.grid2,.grid3,.grid4,.controls{grid-template-columns:1fr}
.hero{padding-top:48px}.section{padding:20px}
.barwrap{grid-template-columns:repeat(5,minmax(52px,1fr))}
}
</style>
</head>
<body>
<div class="wrap">
<header class="hero">
<div class="badge">Open Weight · Quantization Explorer</div>
<h1>Make models smaller. Understand the <span class="gradient">trade-offs.</span></h1>
<p class="lead">A practical guide to model quantization — from FP16 and BF16 to FP8, INT8, INT4, bitsandbytes, GPTQ, AWQ and GGUF-based local inference.</p>
<div class="cta">
<a class="btn primary" href="#basics">Start exploring</a>
<a class="btn" href="https://huggingface.co/open-weight" target="_blank" rel="noopener">Open Weight organization ↗</a>
</div>
</header>
<div class="quick">
<div class="card"><strong>FP8</strong><span>8-bit floating-point workflows</span></div>
<div class="card"><strong>INT8</strong><span>Lower-memory integer inference</span></div>
<div class="card"><strong>INT4</strong><span>High compression for deployment</span></div>
<div class="card"><strong>Trade-offs</strong><span>Memory · quality · speed · support</span></div>
</div>
<section class="section" id="basics">
<div class="eyebrow">01 · Foundation</div>
<h2>What is model quantization?</h2>
<p><strong>Quantization reduces the precision used to represent model weights or activations.</strong> The goal is usually to reduce memory requirements and make models easier or cheaper to run while preserving as much model quality as possible.</p>
<div class="callout">Hugging Face describes quantization as lowering model memory requirements by storing weights at lower precision while trying to preserve accuracy.</div>
<div class="diagram">
<div class="flow">
<div class="node">FP32 / BF16 / FP16</div><div class="arrow">→</div>
<div class="node">Quantization method</div><div class="arrow">→</div>
<div class="node">FP8 / INT8 / INT4</div><div class="arrow">→</div>
<div class="node">Lower memory</div><div class="arrow">+</div>
<div class="node">Potential speed gains</div><div class="arrow">+</div>
<div class="node">Trade-offs</div>
</div>
</div>
</section>
<section class="section">
<div class="eyebrow">02 · Precision</div>
<h2>Bits per parameter: the basic intuition</h2>
<p>The chart below shows <strong>theoretical raw weight storage</strong> relative to FP32. It ignores runtime overhead, metadata, KV cache, activations and mixed-precision components.</p>
<div class="barwrap">
<div class="bar" style="height:100%"><b>32</b><small>FP32</small></div>
<div class="bar" style="height:50%"><b>16</b><small>FP16/BF16</small></div>
<div class="bar" style="height:25%"><b>8</b><small>FP8</small></div>
<div class="bar" style="height:25%"><b>8</b><small>INT8</small></div>
<div class="bar" style="height:12.5%"><b>4</b><small>INT4</small></div>
</div>
<div class="callout warning"><strong>Lower bit width does not guarantee faster inference.</strong> Actual performance depends on kernels, hardware, memory bandwidth, runtime support and the quantization method.</div>
</section>
<section class="section">
<div class="eyebrow">03 · Memory estimator</div>
<h2>Estimate raw weight storage</h2>
<p>Use this simple calculator to estimate the theoretical storage of model weights at a chosen bit width.</p>
<div class="controls">
<div>
<label for="params">Model parameters (billions)</label>
<input id="params" type="number" min="0.1" step="0.1" value="8">
</div>
<div>
<label for="bits">Bits per parameter</label>
<select id="bits">
<option value="32">FP32 — 32 bit</option>
<option value="16" selected>FP16 / BF16 — 16 bit</option>
<option value="8">FP8 / INT8 — 8 bit</option>
<option value="4">INT4 — 4 bit</option>
<option value="2">2 bit — method dependent</option>
</select>
</div>
</div>
<div class="result">
<div class="big" id="memoryOut">16.00 GB</div>
<div class="muted">Approximate decimal GB for raw parameters only. Real deployment memory can be higher.</div>
</div>
</section>
<section class="section">
<div class="eyebrow">04 · Main approaches</div>
<h2>Quantization is not one technique</h2>
<div class="grid3">
<div class="card">
<h3>On-the-fly</h3>
<p class="muted">Quantize during model loading rather than distributing a separately pre-quantized checkpoint.</p>
<span class="pill">bitsandbytes</span>
</div>
<div class="card">
<h3>Post-training</h3>
<p class="muted">Quantize an already trained model, often using calibration or optimization to reduce error.</p>
<span class="pill">GPTQ</span><span class="pill">AWQ</span>
</div>
<div class="card">
<h3>Runtime ecosystem</h3>
<p class="muted">Convert and quantize for a deployment stack such as GGUF / llama.cpp.</p>
<span class="pill">GGUF</span><span class="pill">llama.cpp</span>
</div>
</div>
</section>
<section class="section">
<div class="eyebrow">05 · bitsandbytes</div>
<h2>4-bit and 8-bit loading in Transformers</h2>
<p>Hugging Face documents <strong>bitsandbytes</strong> as providing memory-efficient 8-bit and 4-bit linear layers and quantization integrations for Transformers.</p>
<div class="grid2">
<div class="card">
<h3>LLM.int8()</h3>
<p class="muted">An 8-bit method designed to preserve higher precision for sensitive computations instead of naively forcing everything into INT8.</p>
</div>
<div class="card">
<h3>QLoRA</h3>
<p class="muted">Uses 4-bit quantization with trainable low-rank adapter parameters, making parameter-efficient adaptation possible with a smaller memory footprint.</p>
</div>
</div>
<p><a href="https://huggingface.co/docs/transformers/en/quantization/bitsandbytes" target="_blank" rel="noopener">Hugging Face bitsandbytes documentation ↗</a></p>
</section>
<section class="section">
<div class="eyebrow">06 · GPTQ</div>
<h2>Error-aware post-training quantization</h2>
<p>Current Transformers documentation uses <strong>GPT-QModel</strong> as the maintained GPTQ backend. GPTQ is a post-training method that quantizes weight matrices while optimizing to reduce quantization error.</p>
<div class="callout">Hugging Face notes that current GPTQ workflows can quantize weights to low-bit representations such as INT4 and dequantize them during inference in optimized kernels.</div>
<p><a href="https://huggingface.co/docs/transformers/quantization/gptq" target="_blank" rel="noopener">Hugging Face GPTQ documentation ↗</a></p>
</section>
<section class="section">
<div class="eyebrow">07 · AWQ</div>
<h2>Activation-aware weight quantization</h2>
<p><strong>AWQ</strong> focuses on preserving weights that are especially important to model behavior while compressing the model to low-bit representations.</p>
<p>Transformers documents AWQ as an activation-aware approach designed for 4-bit compression with limited performance degradation.</p>
<p><a href="https://huggingface.co/docs/transformers/quantization/awq" target="_blank" rel="noopener">Hugging Face AWQ documentation ↗</a></p>
</section>
<section class="section">
<div class="eyebrow">08 · GGUF / llama.cpp</div>
<h2>Quantization for local and portable inference</h2>
<p>The llama.cpp ecosystem provides many quantized GGUF tensor types and tooling for converting higher-precision GGUF models into smaller quantized variants.</p>
<div class="code">High-precision model
↓
Convert to GGUF
↓
llama-quantize
↓
Q8 / Q6 / Q5 / Q4 / lower-bit variants
↓
Evaluate quality + performance
↓
Run with llama.cpp</div>
<p class="muted">llama.cpp documents integer quantization from very low bit widths through 8-bit variants. The practical choice depends on model family, quality target and hardware.</p>
<p><a href="https://github.com/ggml-org/llama.cpp/tree/master/tools/quantize" target="_blank" rel="noopener">llama.cpp quantization tools ↗</a></p>
</section>
<section class="section">
<div class="eyebrow">09 · Comparison</div>
<h2>Common quantization directions</h2>
<div class="tablewrap">
<table>
<thead><tr><th>Approach</th><th>Typical bit width</th><th>Strength</th><th>Watch for</th></tr></thead>
<tbody>
<tr><td>bitsandbytes</td><td>4 / 8</td><td>Convenient Transformers integration and on-the-fly loading</td><td>Hardware/backend support and training limitations</td></tr>
<tr><td>GPTQ</td><td>Commonly 4; other bit widths supported by current backends</td><td>Post-training compression with error-aware optimization</td><td>Kernel, model and checkpoint compatibility</td></tr>
<tr><td>AWQ</td><td>4</td><td>Activation-aware preservation of important weights</td><td>Toolchain and runtime compatibility</td></tr>
<tr><td>GGUF / llama.cpp</td><td>Multiple low-bit types</td><td>Strong local-inference ecosystem and many quantization variants</td><td>Model architecture support and quality/runtime trade-offs</td></tr>
<tr><td>FP8</td><td>8</td><td>Lower precision while remaining floating point</td><td>Hardware and kernel support</td></tr>
</tbody>
</table>
</div>
</section>
<section class="section">
<div class="eyebrow">10 · Decision helper</div>
<h2>Which direction should you investigate?</h2>
<p>Select a deployment goal. This is a starting point, not a universal recommendation.</p>
<div class="tabs">
<button class="tab active" data-answer="hf">Transformers simplicity</button>
<button class="tab" data-answer="local">Local / consumer hardware</button>
<button class="tab" data-answer="gpu">GPU server inference</button>
<button class="tab" data-answer="tune">Fine-tuning</button>
</div>
<div class="answer" id="answer">
<strong>Start by evaluating bitsandbytes.</strong>
<p class="muted">Its 4-bit and 8-bit Transformers integration makes it a practical entry point when your model and hardware are supported.</p>
</div>
</section>
<section class="section">
<div class="eyebrow">11 · Trade-offs</div>
<h2>What should you measure?</h2>
<div class="grid4">
<div class="card"><h3>Memory</h3><p class="muted">How much RAM or VRAM is actually required?</p></div>
<div class="card"><h3>Quality</h3><p class="muted">How much task performance changes after quantization?</p></div>
<div class="card"><h3>Latency</h3><p class="muted">Does the runtime and hardware actually become faster?</p></div>
<div class="card"><h3>Compatibility</h3><p class="muted">Can your serving stack load and accelerate the chosen format?</p></div>
</div>
<div class="callout warning"><strong>Always benchmark on the real workload.</strong> A smaller checkpoint can still perform worse operationally if the runtime lacks optimized kernels for that quantization.</div>
</section>
<section class="section">
<div class="eyebrow">12 · Common mistakes</div>
<h2>Quantization misconceptions</h2>
<div class="grid3">
<div class="card"><h3>Bits ≠ method</h3><p class="muted">Two 4-bit methods can behave very differently.</p></div>
<div class="card"><h3>Smaller ≠ faster</h3><p class="muted">Speed depends on kernels, hardware and runtime support.</p></div>
<div class="card"><h3>Format ≠ quantization</h3><p class="muted">GGUF or Safetensors are serialization formats; quantization describes numerical representation and method.</p></div>
<div class="card"><h3>Memory ≠ file size only</h3><p class="muted">KV cache, activations and runtime overhead also matter.</p></div>
<div class="card"><h3>Quality loss is task-specific</h3><p class="muted">Benchmark the model on the tasks that matter to you.</p></div>
<div class="card"><h3>Support changes</h3><p class="muted">Quantization libraries and hardware backends evolve quickly.</p></div>
</div>
</section>
<section class="section">
<div class="eyebrow">13 · Quick checklist</div>
<h2>Before choosing a quantization</h2>
<div class="tablewrap">
<table>
<thead><tr><th>Question</th><th>Why it matters</th></tr></thead>
<tbody>
<tr><td>What hardware will run the model?</td><td>Backend support and optimized kernels differ by platform.</td></tr>
<tr><td>Which runtime will serve it?</td><td>Not every runtime supports every quantization method.</td></tr>
<tr><td>What memory limit do you have?</td><td>Defines how aggressive compression may need to be.</td></tr>
<tr><td>What quality loss is acceptable?</td><td>Lower bit widths can affect downstream performance.</td></tr>
<tr><td>Do you need fine-tuning?</td><td>Some workflows support PEFT or adapter training better than others.</td></tr>
<tr><td>Do you need portability?</td><td>A highly optimized method may tie you to a specific runtime or hardware stack.</td></tr>
</tbody>
</table>
</div>
</section>
<section class="section">
<div class="eyebrow">Next</div>
<h2>Continue the Open Weight series</h2>
<div class="grid3">
<div class="card"><h3>Open Weight Explorer</h3><p class="muted">Understand tensors, model weights and the deployment stack.</p></div>
<div class="card"><h3>Weight Format Explorer</h3><p class="muted">Safetensors, GGUF, metadata, sharding and conversion.</p></div>
<div class="card"><h3>Model Portability Explorer</h3><p class="muted">Formats, runtimes, hardware and compatibility.</p></div>
</div>
</section>
<section class="section">
<div class="eyebrow">Primary sources</div>
<h2>Technical references</h2>
<p><a href="https://huggingface.co/docs/transformers/quantization/overview" target="_blank" rel="noopener">Hugging Face Transformers — Quantization overview ↗</a></p>
<p><a href="https://huggingface.co/docs/transformers/en/quantization/bitsandbytes" target="_blank" rel="noopener">Hugging Face — bitsandbytes ↗</a></p>
<p><a href="https://huggingface.co/docs/transformers/quantization/gptq" target="_blank" rel="noopener">Hugging Face — GPTQ ↗</a></p>
<p><a href="https://huggingface.co/docs/transformers/quantization/awq" target="_blank" rel="noopener">Hugging Face — AWQ ↗</a></p>
<p><a href="https://github.com/ggml-org/llama.cpp/tree/master/tools/quantize" target="_blank" rel="noopener">llama.cpp — quantization tools ↗</a></p>
</section>
<footer>
<strong style="color:white">Open Weight</strong><br>
Open weights. Portable models. Deployable AI.<br><br>
Collaboration: open-weight AI, model infrastructure, inference, deployment, research and ecosystem partnerships.<br>
Contact: <a href="mailto:agenten@magenta.de">agenten@magenta.de</a>
</footer>
</div>
<script>
function updateMemory(){
const p = Math.max(0, parseFloat(document.getElementById('params').value)||0);
const b = Math.max(0, parseFloat(document.getElementById('bits').value)||0);
const gb = p * b / 8;
document.getElementById('memoryOut').textContent = gb.toFixed(2) + ' GB';
}
document.getElementById('params').addEventListener('input',updateMemory);
document.getElementById('bits').addEventListener('change',updateMemory);
const answers={
hf:`<strong>Start by evaluating bitsandbytes.</strong><p class="muted">Its 4-bit and 8-bit Transformers integration makes it a practical entry point when your model and hardware are supported.</p>`,
local:`<strong>Investigate GGUF / llama.cpp quantization.</strong><p class="muted">The ecosystem offers many low-bit formats for local inference across supported CPUs, GPUs and Apple Silicon workflows.</p>`,
gpu:`<strong>Choose the serving runtime first.</strong><p class="muted">Then compare the methods it accelerates well — such as supported FP8, GPTQ, AWQ, bitsandbytes or other native quantization paths.</p>`,
tune:`<strong>Look closely at 4-bit PEFT / QLoRA workflows.</strong><p class="muted">bitsandbytes is a common entry point because it combines low-bit loading with trainable adapter parameters.</p>`
};
document.querySelectorAll('.tab').forEach(btn=>{
btn.addEventListener('click',()=>{
document.querySelectorAll('.tab').forEach(x=>x.classList.remove('active'));
btn.classList.add('active');
document.getElementById('answer').innerHTML=answers[btn.dataset.answer];
});
});
updateMemory();
</script>
</body>
</html>