File size: 21,753 Bytes
762c004
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<meta name="description" content="Explore AI model quantization: FP16, BF16, FP8, INT8, INT4, bitsandbytes, GPTQ, AWQ and GGUF. Compare memory, quality and deployment trade-offs.">
<meta name="theme-color" content="#07111f">
<title>Quantization Explorer — FP8, INT8, INT4 & Open-Weight Deployment</title>
<style>
:root{
  --bg:#07111f;--panel:#0d1b2d;--panel2:#10233a;--text:#edf7ff;--muted:#9fb4c8;
  --line:#24445f;--cyan:#55d9ff;--blue:#6b8cff;--green:#79f2c0;--gold:#ffd580;
  --red:#ff9e9e;--shadow:0 18px 60px rgba(0,0,0,.25)
}
*{box-sizing:border-box}
html{scroll-behavior:smooth}
body{
  margin:0;color:var(--text);
  font:16px/1.65 Inter,ui-sans-serif,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;
  background:
    radial-gradient(circle at 12% 0%,rgba(85,217,255,.12),transparent 29%),
    radial-gradient(circle at 90% 12%,rgba(107,140,255,.13),transparent 26%),
    var(--bg)
}
a{color:var(--cyan);text-decoration:none}
a:hover{text-decoration:underline}
.wrap{max-width:1180px;margin:auto;padding:0 22px}
.hero{padding:72px 0 35px}
.badge{display:inline-flex;padding:7px 12px;border:1px solid var(--line);border-radius:999px;background:rgba(13,27,45,.75);color:#c8efff;font-size:14px}
h1{font-size:clamp(42px,7vw,78px);line-height:1;letter-spacing:-.055em;margin:20px 0;max-width:980px}
.gradient{background:linear-gradient(90deg,var(--cyan),#b6c3ff);-webkit-background-clip:text;background-clip:text;color:transparent}
.lead{font-size:clamp(18px,2.2vw,24px);max-width:900px;color:#cbdbe8;margin:0 0 28px}
.cta{display:flex;gap:12px;flex-wrap:wrap}
.btn{display:inline-block;padding:11px 16px;border:1px solid var(--line);border-radius:12px;font-weight:750}
.btn.primary{border:0;color:white;background:linear-gradient(135deg,#157aa8,#5269df)}
.quick{display:grid;grid-template-columns:repeat(4,1fr);gap:14px;margin:32px 0 56px}
.card,.section{border:1px solid var(--line);background:linear-gradient(180deg,rgba(16,35,58,.93),rgba(10,25,42,.93));border-radius:20px;box-shadow:var(--shadow)}
.card{padding:18px}
.card strong{display:block;font-size:23px;color:#fff}
.card span,.muted{color:var(--muted)}
.section{padding:28px;margin:22px 0}
.eyebrow{text-transform:uppercase;letter-spacing:.14em;font-size:12px;color:var(--cyan);font-weight:850}
h2{font-size:clamp(28px,4vw,45px);letter-spacing:-.03em;margin:6px 0 10px}
h3{font-size:22px;margin:0 0 8px}
.grid2{display:grid;grid-template-columns:1fr 1fr;gap:18px}
.grid3{display:grid;grid-template-columns:repeat(3,1fr);gap:14px}
.grid4{display:grid;grid-template-columns:repeat(4,1fr);gap:14px}
.callout{padding:15px 17px;border-left:4px solid var(--cyan);border-radius:9px;background:rgba(85,217,255,.06);margin:18px 0}
.warning{border-left-color:var(--gold);background:rgba(255,213,128,.06)}
.diagram{padding:22px;border:1px solid #2d5876;border-radius:17px;background:#091829;overflow:auto;margin:20px 0}
.flow{display:flex;gap:9px;align-items:center;min-width:900px}
.node{min-width:135px;padding:14px 12px;text-align:center;border:1px solid #34617d;background:#102842;border-radius:13px;font-weight:800}
.arrow{font-size:24px;color:var(--cyan)}
.tablewrap{overflow:auto}
table{width:100%;border-collapse:collapse;min-width:760px}
th,td{padding:13px;border-bottom:1px solid #23435e;text-align:left;vertical-align:top}
th{font-size:12px;text-transform:uppercase;letter-spacing:.07em;color:#c9efff}
.pill{display:inline-block;padding:5px 9px;margin:3px;border:1px solid #315b78;border-radius:999px;background:#102842;color:#c8ecff;font-size:13px}
.code{white-space:pre-wrap;padding:17px;border:1px solid #223f58;border-radius:14px;background:#06101c;color:#bfeeff;font:14px/1.6 ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;overflow:auto}
.barwrap{display:grid;grid-template-columns:repeat(5,1fr);gap:12px;align-items:end;height:220px;margin:28px 0 42px}
.bar{position:relative;border-radius:12px 12px 4px 4px;background:linear-gradient(180deg,var(--cyan),#5368de);min-height:28px}
.bar b{position:absolute;top:10px;left:0;right:0;text-align:center;color:#06111f}
.bar small{position:absolute;bottom:-30px;left:0;right:0;text-align:center;color:var(--muted)}
.controls{display:grid;grid-template-columns:1fr 1fr;gap:16px}
label{display:block;color:#c8efff;font-weight:700;margin-bottom:7px}
input,select,button{
  width:100%;background:#0a1c2f;color:var(--text);border:1px solid #315b78;border-radius:11px;padding:11px 12px;font:inherit
}
.result{margin-top:18px;padding:18px;border:1px solid #315b78;background:#091a2b;border-radius:15px}
.big{font-size:34px;font-weight:850;color:white}
.tabs{display:flex;gap:9px;flex-wrap:wrap;margin:16px 0}
.tab{width:auto;cursor:pointer;border:1px solid #315b78;background:#0c2035;color:#d9f2ff;border-radius:999px;padding:8px 12px;font-weight:700}
.tab.active{background:linear-gradient(135deg,#177aa6,#4e66d8);border-color:transparent}
.answer{padding:18px;border:1px solid #315b78;border-radius:15px;background:#0a1c2f;min-height:112px}
.good{color:var(--green);font-weight:800}
.caution{color:var(--gold);font-weight:800}
.bad{color:var(--red);font-weight:800}
footer{padding:50px 0 68px;color:var(--muted)}
@media(max-width:820px){
  .quick,.grid2,.grid3,.grid4,.controls{grid-template-columns:1fr}
  .hero{padding-top:48px}.section{padding:20px}
  .barwrap{grid-template-columns:repeat(5,minmax(52px,1fr))}
}
</style>
</head>
<body>
<div class="wrap">
<header class="hero">
  <div class="badge">Open Weight · Quantization Explorer</div>
  <h1>Make models smaller. Understand the <span class="gradient">trade-offs.</span></h1>
  <p class="lead">A practical guide to model quantization — from FP16 and BF16 to FP8, INT8, INT4, bitsandbytes, GPTQ, AWQ and GGUF-based local inference.</p>
  <div class="cta">
    <a class="btn primary" href="#basics">Start exploring</a>
    <a class="btn" href="https://huggingface.co/open-weight" target="_blank" rel="noopener">Open Weight organization ↗</a>
  </div>
</header>

<div class="quick">
  <div class="card"><strong>FP8</strong><span>8-bit floating-point workflows</span></div>
  <div class="card"><strong>INT8</strong><span>Lower-memory integer inference</span></div>
  <div class="card"><strong>INT4</strong><span>High compression for deployment</span></div>
  <div class="card"><strong>Trade-offs</strong><span>Memory · quality · speed · support</span></div>
</div>

<section class="section" id="basics">
  <div class="eyebrow">01 · Foundation</div>
  <h2>What is model quantization?</h2>
  <p><strong>Quantization reduces the precision used to represent model weights or activations.</strong> The goal is usually to reduce memory requirements and make models easier or cheaper to run while preserving as much model quality as possible.</p>
  <div class="callout">Hugging Face describes quantization as lowering model memory requirements by storing weights at lower precision while trying to preserve accuracy.</div>
  <div class="diagram">
    <div class="flow">
      <div class="node">FP32 / BF16 / FP16</div><div class="arrow">→</div>
      <div class="node">Quantization method</div><div class="arrow">→</div>
      <div class="node">FP8 / INT8 / INT4</div><div class="arrow">→</div>
      <div class="node">Lower memory</div><div class="arrow">+</div>
      <div class="node">Potential speed gains</div><div class="arrow">+</div>
      <div class="node">Trade-offs</div>
    </div>
  </div>
</section>

<section class="section">
  <div class="eyebrow">02 · Precision</div>
  <h2>Bits per parameter: the basic intuition</h2>
  <p>The chart below shows <strong>theoretical raw weight storage</strong> relative to FP32. It ignores runtime overhead, metadata, KV cache, activations and mixed-precision components.</p>
  <div class="barwrap">
    <div class="bar" style="height:100%"><b>32</b><small>FP32</small></div>
    <div class="bar" style="height:50%"><b>16</b><small>FP16/BF16</small></div>
    <div class="bar" style="height:25%"><b>8</b><small>FP8</small></div>
    <div class="bar" style="height:25%"><b>8</b><small>INT8</small></div>
    <div class="bar" style="height:12.5%"><b>4</b><small>INT4</small></div>
  </div>
  <div class="callout warning"><strong>Lower bit width does not guarantee faster inference.</strong> Actual performance depends on kernels, hardware, memory bandwidth, runtime support and the quantization method.</div>
</section>

<section class="section">
  <div class="eyebrow">03 · Memory estimator</div>
  <h2>Estimate raw weight storage</h2>
  <p>Use this simple calculator to estimate the theoretical storage of model weights at a chosen bit width.</p>
  <div class="controls">
    <div>
      <label for="params">Model parameters (billions)</label>
      <input id="params" type="number" min="0.1" step="0.1" value="8">
    </div>
    <div>
      <label for="bits">Bits per parameter</label>
      <select id="bits">
        <option value="32">FP32 — 32 bit</option>
        <option value="16" selected>FP16 / BF16 — 16 bit</option>
        <option value="8">FP8 / INT8 — 8 bit</option>
        <option value="4">INT4 — 4 bit</option>
        <option value="2">2 bit — method dependent</option>
      </select>
    </div>
  </div>
  <div class="result">
    <div class="big" id="memoryOut">16.00 GB</div>
    <div class="muted">Approximate decimal GB for raw parameters only. Real deployment memory can be higher.</div>
  </div>
</section>

<section class="section">
  <div class="eyebrow">04 · Main approaches</div>
  <h2>Quantization is not one technique</h2>
  <div class="grid3">
    <div class="card">
      <h3>On-the-fly</h3>
      <p class="muted">Quantize during model loading rather than distributing a separately pre-quantized checkpoint.</p>
      <span class="pill">bitsandbytes</span>
    </div>
    <div class="card">
      <h3>Post-training</h3>
      <p class="muted">Quantize an already trained model, often using calibration or optimization to reduce error.</p>
      <span class="pill">GPTQ</span><span class="pill">AWQ</span>
    </div>
    <div class="card">
      <h3>Runtime ecosystem</h3>
      <p class="muted">Convert and quantize for a deployment stack such as GGUF / llama.cpp.</p>
      <span class="pill">GGUF</span><span class="pill">llama.cpp</span>
    </div>
  </div>
</section>

<section class="section">
  <div class="eyebrow">05 · bitsandbytes</div>
  <h2>4-bit and 8-bit loading in Transformers</h2>
  <p>Hugging Face documents <strong>bitsandbytes</strong> as providing memory-efficient 8-bit and 4-bit linear layers and quantization integrations for Transformers.</p>
  <div class="grid2">
    <div class="card">
      <h3>LLM.int8()</h3>
      <p class="muted">An 8-bit method designed to preserve higher precision for sensitive computations instead of naively forcing everything into INT8.</p>
    </div>
    <div class="card">
      <h3>QLoRA</h3>
      <p class="muted">Uses 4-bit quantization with trainable low-rank adapter parameters, making parameter-efficient adaptation possible with a smaller memory footprint.</p>
    </div>
  </div>
  <p><a href="https://huggingface.co/docs/transformers/en/quantization/bitsandbytes" target="_blank" rel="noopener">Hugging Face bitsandbytes documentation ↗</a></p>
</section>

<section class="section">
  <div class="eyebrow">06 · GPTQ</div>
  <h2>Error-aware post-training quantization</h2>
  <p>Current Transformers documentation uses <strong>GPT-QModel</strong> as the maintained GPTQ backend. GPTQ is a post-training method that quantizes weight matrices while optimizing to reduce quantization error.</p>
  <div class="callout">Hugging Face notes that current GPTQ workflows can quantize weights to low-bit representations such as INT4 and dequantize them during inference in optimized kernels.</div>
  <p><a href="https://huggingface.co/docs/transformers/quantization/gptq" target="_blank" rel="noopener">Hugging Face GPTQ documentation ↗</a></p>
</section>

<section class="section">
  <div class="eyebrow">07 · AWQ</div>
  <h2>Activation-aware weight quantization</h2>
  <p><strong>AWQ</strong> focuses on preserving weights that are especially important to model behavior while compressing the model to low-bit representations.</p>
  <p>Transformers documents AWQ as an activation-aware approach designed for 4-bit compression with limited performance degradation.</p>
  <p><a href="https://huggingface.co/docs/transformers/quantization/awq" target="_blank" rel="noopener">Hugging Face AWQ documentation ↗</a></p>
</section>

<section class="section">
  <div class="eyebrow">08 · GGUF / llama.cpp</div>
  <h2>Quantization for local and portable inference</h2>
  <p>The llama.cpp ecosystem provides many quantized GGUF tensor types and tooling for converting higher-precision GGUF models into smaller quantized variants.</p>
  <div class="code">High-precision model
       ↓
Convert to GGUF
       ↓
llama-quantize
       ↓
Q8 / Q6 / Q5 / Q4 / lower-bit variants
       ↓
Evaluate quality + performance
       ↓
Run with llama.cpp</div>
  <p class="muted">llama.cpp documents integer quantization from very low bit widths through 8-bit variants. The practical choice depends on model family, quality target and hardware.</p>
  <p><a href="https://github.com/ggml-org/llama.cpp/tree/master/tools/quantize" target="_blank" rel="noopener">llama.cpp quantization tools ↗</a></p>
</section>

<section class="section">
  <div class="eyebrow">09 · Comparison</div>
  <h2>Common quantization directions</h2>
  <div class="tablewrap">
    <table>
      <thead><tr><th>Approach</th><th>Typical bit width</th><th>Strength</th><th>Watch for</th></tr></thead>
      <tbody>
        <tr><td>bitsandbytes</td><td>4 / 8</td><td>Convenient Transformers integration and on-the-fly loading</td><td>Hardware/backend support and training limitations</td></tr>
        <tr><td>GPTQ</td><td>Commonly 4; other bit widths supported by current backends</td><td>Post-training compression with error-aware optimization</td><td>Kernel, model and checkpoint compatibility</td></tr>
        <tr><td>AWQ</td><td>4</td><td>Activation-aware preservation of important weights</td><td>Toolchain and runtime compatibility</td></tr>
        <tr><td>GGUF / llama.cpp</td><td>Multiple low-bit types</td><td>Strong local-inference ecosystem and many quantization variants</td><td>Model architecture support and quality/runtime trade-offs</td></tr>
        <tr><td>FP8</td><td>8</td><td>Lower precision while remaining floating point</td><td>Hardware and kernel support</td></tr>
      </tbody>
    </table>
  </div>
</section>

<section class="section">
  <div class="eyebrow">10 · Decision helper</div>
  <h2>Which direction should you investigate?</h2>
  <p>Select a deployment goal. This is a starting point, not a universal recommendation.</p>
  <div class="tabs">
    <button class="tab active" data-answer="hf">Transformers simplicity</button>
    <button class="tab" data-answer="local">Local / consumer hardware</button>
    <button class="tab" data-answer="gpu">GPU server inference</button>
    <button class="tab" data-answer="tune">Fine-tuning</button>
  </div>
  <div class="answer" id="answer">
    <strong>Start by evaluating bitsandbytes.</strong>
    <p class="muted">Its 4-bit and 8-bit Transformers integration makes it a practical entry point when your model and hardware are supported.</p>
  </div>
</section>

<section class="section">
  <div class="eyebrow">11 · Trade-offs</div>
  <h2>What should you measure?</h2>
  <div class="grid4">
    <div class="card"><h3>Memory</h3><p class="muted">How much RAM or VRAM is actually required?</p></div>
    <div class="card"><h3>Quality</h3><p class="muted">How much task performance changes after quantization?</p></div>
    <div class="card"><h3>Latency</h3><p class="muted">Does the runtime and hardware actually become faster?</p></div>
    <div class="card"><h3>Compatibility</h3><p class="muted">Can your serving stack load and accelerate the chosen format?</p></div>
  </div>
  <div class="callout warning"><strong>Always benchmark on the real workload.</strong> A smaller checkpoint can still perform worse operationally if the runtime lacks optimized kernels for that quantization.</div>
</section>

<section class="section">
  <div class="eyebrow">12 · Common mistakes</div>
  <h2>Quantization misconceptions</h2>
  <div class="grid3">
    <div class="card"><h3>Bits ≠ method</h3><p class="muted">Two 4-bit methods can behave very differently.</p></div>
    <div class="card"><h3>Smaller ≠ faster</h3><p class="muted">Speed depends on kernels, hardware and runtime support.</p></div>
    <div class="card"><h3>Format ≠ quantization</h3><p class="muted">GGUF or Safetensors are serialization formats; quantization describes numerical representation and method.</p></div>
    <div class="card"><h3>Memory ≠ file size only</h3><p class="muted">KV cache, activations and runtime overhead also matter.</p></div>
    <div class="card"><h3>Quality loss is task-specific</h3><p class="muted">Benchmark the model on the tasks that matter to you.</p></div>
    <div class="card"><h3>Support changes</h3><p class="muted">Quantization libraries and hardware backends evolve quickly.</p></div>
  </div>
</section>

<section class="section">
  <div class="eyebrow">13 · Quick checklist</div>
  <h2>Before choosing a quantization</h2>
  <div class="tablewrap">
    <table>
      <thead><tr><th>Question</th><th>Why it matters</th></tr></thead>
      <tbody>
        <tr><td>What hardware will run the model?</td><td>Backend support and optimized kernels differ by platform.</td></tr>
        <tr><td>Which runtime will serve it?</td><td>Not every runtime supports every quantization method.</td></tr>
        <tr><td>What memory limit do you have?</td><td>Defines how aggressive compression may need to be.</td></tr>
        <tr><td>What quality loss is acceptable?</td><td>Lower bit widths can affect downstream performance.</td></tr>
        <tr><td>Do you need fine-tuning?</td><td>Some workflows support PEFT or adapter training better than others.</td></tr>
        <tr><td>Do you need portability?</td><td>A highly optimized method may tie you to a specific runtime or hardware stack.</td></tr>
      </tbody>
    </table>
  </div>
</section>

<section class="section">
  <div class="eyebrow">Next</div>
  <h2>Continue the Open Weight series</h2>
  <div class="grid3">
    <div class="card"><h3>Open Weight Explorer</h3><p class="muted">Understand tensors, model weights and the deployment stack.</p></div>
    <div class="card"><h3>Weight Format Explorer</h3><p class="muted">Safetensors, GGUF, metadata, sharding and conversion.</p></div>
    <div class="card"><h3>Model Portability Explorer</h3><p class="muted">Formats, runtimes, hardware and compatibility.</p></div>
  </div>
</section>

<section class="section">
  <div class="eyebrow">Primary sources</div>
  <h2>Technical references</h2>
  <p><a href="https://huggingface.co/docs/transformers/quantization/overview" target="_blank" rel="noopener">Hugging Face Transformers — Quantization overview ↗</a></p>
  <p><a href="https://huggingface.co/docs/transformers/en/quantization/bitsandbytes" target="_blank" rel="noopener">Hugging Face — bitsandbytes ↗</a></p>
  <p><a href="https://huggingface.co/docs/transformers/quantization/gptq" target="_blank" rel="noopener">Hugging Face — GPTQ ↗</a></p>
  <p><a href="https://huggingface.co/docs/transformers/quantization/awq" target="_blank" rel="noopener">Hugging Face — AWQ ↗</a></p>
  <p><a href="https://github.com/ggml-org/llama.cpp/tree/master/tools/quantize" target="_blank" rel="noopener">llama.cpp — quantization tools ↗</a></p>
</section>

<footer>
  <strong style="color:white">Open Weight</strong><br>
  Open weights. Portable models. Deployable AI.<br><br>
  Collaboration: open-weight AI, model infrastructure, inference, deployment, research and ecosystem partnerships.<br>
  Contact: <a href="mailto:agenten@magenta.de">agenten@magenta.de</a>
</footer>
</div>

<script>
function updateMemory(){
  const p = Math.max(0, parseFloat(document.getElementById('params').value)||0);
  const b = Math.max(0, parseFloat(document.getElementById('bits').value)||0);
  const gb = p * b / 8;
  document.getElementById('memoryOut').textContent = gb.toFixed(2) + ' GB';
}
document.getElementById('params').addEventListener('input',updateMemory);
document.getElementById('bits').addEventListener('change',updateMemory);

const answers={
  hf:`<strong>Start by evaluating bitsandbytes.</strong><p class="muted">Its 4-bit and 8-bit Transformers integration makes it a practical entry point when your model and hardware are supported.</p>`,
  local:`<strong>Investigate GGUF / llama.cpp quantization.</strong><p class="muted">The ecosystem offers many low-bit formats for local inference across supported CPUs, GPUs and Apple Silicon workflows.</p>`,
  gpu:`<strong>Choose the serving runtime first.</strong><p class="muted">Then compare the methods it accelerates well — such as supported FP8, GPTQ, AWQ, bitsandbytes or other native quantization paths.</p>`,
  tune:`<strong>Look closely at 4-bit PEFT / QLoRA workflows.</strong><p class="muted">bitsandbytes is a common entry point because it combines low-bit loading with trainable adapter parameters.</p>`
};
document.querySelectorAll('.tab').forEach(btn=>{
  btn.addEventListener('click',()=>{
    document.querySelectorAll('.tab').forEach(x=>x.classList.remove('active'));
    btn.classList.add('active');
    document.getElementById('answer').innerHTML=answers[btn.dataset.answer];
  });
});
updateMemory();
</script>
</body>
</html>