Buckets:

download
raw
61.7 kB
<meta charset="utf-8" /><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;Optimization&quot;,&quot;local&quot;:&quot;optimization&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Optimization Support Matrix&quot;,&quot;local&quot;:&quot;optimization-support-matrix&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Weight-only Quantization&quot;,&quot;local&quot;:&quot;weight-only-quantization&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;8-bit&quot;,&quot;local&quot;:&quot;8-bit&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;4-bit&quot;,&quot;local&quot;:&quot;4-bit&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Full quantization&quot;,&quot;local&quot;:&quot;full-quantization&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Speech-to-text Models Quantization&quot;,&quot;local&quot;:&quot;speech-to-text-models-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Hybrid quantization&quot;,&quot;local&quot;:&quot;hybrid-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Mixed Quantization&quot;,&quot;local&quot;:&quot;mixed-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Pipeline Quantization&quot;,&quot;local&quot;:&quot;pipeline-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/>
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/entry/start.CatJ25dz.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/B-VqpFX9.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/Cbbc2wwc.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/entry/app.KpdgG_wN.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/DeDI0Q53.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/6YgYxE4i.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/B9oc0ti6.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/nodes/0.CgngmqY0.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/CczWW-MX.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/nodes/7.pe1zTaFq.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/yrm_jVWY.js" rel="modulepreload">
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/chunks/CHJ7u4WY.js" rel="modulepreload">
<!--r7r3m9--><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;Optimization&quot;,&quot;local&quot;:&quot;optimization&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Optimization Support Matrix&quot;,&quot;local&quot;:&quot;optimization-support-matrix&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Weight-only Quantization&quot;,&quot;local&quot;:&quot;weight-only-quantization&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;8-bit&quot;,&quot;local&quot;:&quot;8-bit&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;4-bit&quot;,&quot;local&quot;:&quot;4-bit&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Full quantization&quot;,&quot;local&quot;:&quot;full-quantization&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Speech-to-text Models Quantization&quot;,&quot;local&quot;:&quot;speech-to-text-models-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Hybrid quantization&quot;,&quot;local&quot;:&quot;hybrid-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Mixed Quantization&quot;,&quot;local&quot;:&quot;mixed-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Pipeline Quantization&quot;,&quot;local&quot;:&quot;pipeline-quantization&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/><!---->
<link href="/docs/optimum.intel/pr_1908/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg><!----></button></div> <!--[-1--><!--]--></div><!----> <!--[0--><h1 class="relative group"><a id="optimization" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#optimization"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Optimization</span></h1><!--]--><!----> <p>🤗 Optimum Intel provides an <code>openvino</code> package that enables you to apply a variety of model quantization methods on many models hosted on the 🤗 hub using the <a href="https://docs.openvino.ai/2024/openvino-workflow/model-optimization.html" rel="nofollow">NNCF</a> framework.</p> <p>Quantization is a technique to reduce the computational and memory costs of running inference by representing the weights and / or the activations with lower precision data types like 8-bit or 4-bit.</p> <!--[1--><h2 class="relative group"><a id="optimization-support-matrix" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#optimization-support-matrix"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Optimization Support Matrix</span></h2><!--]--><!----> <p>Click on a ✅ to copy the command/code for the corresponding optimization case.</p> <div id="copyMsg" style="display: none; position: fixed; top: 20px; left: 50%; transform: translateX(-50%); background: rgba(60, 179, 113, 0.95); color: white; padding: 10px 20px; border-radius: 6px; box-shadow: 0 4px 12px rgba(0, 0, 0, 0.15); z-index: 1000; font-family: 'Segoe UI', sans-serif; font-size: 14px; font-weight: 500; letter-spacing: 0.3px; white-space: nowrap;">Command copied to clipboard</div> <table><thead><tr><th rowspan="3">Task<br/>(OV Model Class)</th><th colspan="4">Weight-only Quantization</th><th colspan="2" rowspan="2">Hybrid Quantization</th><th colspan="2" rowspan="2">Full Quantization</th><th colspan="2" rowspan="2">Mixed Quantization</th></tr><tr><th colspan="2">Data-free</th><th colspan="2">Data-aware</th></tr><tr><th>CLI</th><th>Python</th><th>CLI</th><th>Python</th><th>CLI</th><th>Python</th><th>CLI</th><th>Python</th><th>CLI</th><th>Python</th></tr></thead><tbody><tr><td style="text-align: center; vertical-align: middle;">text-generation<br/>(OVModelForCausalLM)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;">-</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td></tr><tr><td style="text-align: center; vertical-align: middle;">image-text-to-text<br/>(OVModelForVisualCausalLM)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td></tr><tr><td style="text-align: center; vertical-align: middle;">text-to-image, text-to-video<br/>(OVDiffusionPipeline)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td></tr><tr><td style="text-align: center; vertical-align: middle;">automatic-speech-recognition<br/>(OVModelForSpeechSeq2Seq)</td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td></tr><tr><td style="text-align: center; vertical-align: middle;">feature-extraction<br/>(OVModelForFeatureExtraction)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;">-</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td></tr><tr><td style="text-align: center; vertical-align: middle;">feature-extraction<br/>(OVSentenceTransformer)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;">-</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td></tr><tr><td style="text-align: center; vertical-align: middle;">fill-mask<br/>(OVModelForMaskedLM)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;">-</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td></tr><tr><td style="text-align: center; vertical-align: middle;">text2text-generation<br/>(OVModelForSeq2SeqLM)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;">-</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td></tr><tr><td style="text-align: center; vertical-align: middle;">zero-shot-image-classification<br/>(OVModelForZeroShotImageClassification)</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;">-</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td></tr><tr><td style="text-align: center; vertical-align: middle;">feature-extraction<br/>(OVSamModel)</td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;">-</td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td></tr><tr><td style="text-align: center; vertical-align: middle;">text-to-audio<br/>(OVModelForTextToSpeechSeq2Seq)</td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"><button></button></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td><td style="text-align: center; vertical-align: middle;"></td></tr></tbody></table> <!--[1--><h2 class="relative group"><a id="weight-only-quantization" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#weight-only-quantization"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Weight-only Quantization</span></h2><!--]--><!----> <p>Quantization can be applied on the model’s Linear, Convolutional and Embedding layers, enabling the loading of large models on memory-limited devices. For example, when applying 8-bit quantization, the resulting model will be x4 smaller than its fp32 counterpart. For 4-bit quantization, the reduction in memory could theoretically reach x8, but is closer to x6 in practice.</p> <!--[2--><h3 class="relative group"><a id="8-bit" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#8-bit"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>8-bit</span></h3><!--]--><!----> <p>For the 8-bit weight quantization you can provide <code>quantization_config</code> equal to <code>OVWeightQuantizationConfig(bits=8)</code> to load your model’s weights in 8-bit:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> optimum.intel <span class="hljs-keyword">import</span> OVModelForCausalLM, OVWeightQuantizationConfig
model_id = <span class="hljs-string">&quot;helenai/gpt2-ov&quot;</span>
quantization_config = OVWeightQuantizationConfig(bits=<span class="hljs-number">8</span>)
model = OVModelForCausalLM.from_pretrained(model_id, quantization_config=quantization_config)
<span class="hljs-comment"># Saves the int8 model that will be x4 smaller than its fp32 counterpart</span>
model.save_pretrained(saving_directory)<!----></pre></div><!----> <p>Weights of language models inside vision-language pipelines can be quantized in a similar way:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->model = OVModelForVisualCausalLM.from_pretrained(
<span class="hljs-string">&quot;llava-hf/llava-v1.6-mistral-7b-hf&quot;</span>,
quantization_config=quantization_config
)<!----></pre></div><!----> <blockquote class="warning"><p>If quantization_config is not provided, model will be exported in 8 bits by default when it has more than 1 billion parameters. You can disable it with <code>load_in_8bit=False</code>.</p><!----></blockquote><!----> <!--[2--><h3 class="relative group"><a id="4-bit" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#4-bit"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>4-bit</span></h3><!--]--><!----> <p>4-bit weight quantization can be achieved in a similar way:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> optimum.intel <span class="hljs-keyword">import</span> OVModelForCausalLM
model = OVModelForCausalLM.from_pretrained(model_id, quantization_config={<span class="hljs-string">&quot;bits&quot;</span>: <span class="hljs-number">4</span>})<!----></pre></div><!----> <p>For some models, we provide preconfigured 4-bit weight-only quantization <a href="https://github.com/huggingface/optimum-intel/blob/main/optimum/intel/openvino/configuration.py" rel="nofollow">configurations</a> that offer a good trade-off between quality and speed. This default 4-bit configuration is applied automatically when you specify <code>quantization_config={"bits": 4}</code>.</p> <p>Or for vision-language pipelines:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->model = OVModelForVisualCausalLM.from_pretrained(
<span class="hljs-string">&quot;llava-hf/llava-v1.6-mistral-7b-hf&quot;</span>,
quantization_config={<span class="hljs-string">&quot;bits&quot;</span>: <span class="hljs-number">4</span>}
)<!----></pre></div><!----> <p>You can tune quantization parameters to achieve a better performance accuracy trade-off as follows:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> optimum.intel <span class="hljs-keyword">import</span> OVWeightQuantizationConfig
quantization_config = OVWeightQuantizationConfig(
bits=<span class="hljs-number">4</span>,
sym=<span class="hljs-literal">False</span>,
ratio=<span class="hljs-number">0.8</span>,
quant_method=<span class="hljs-string">&quot;awq&quot;</span>,
dataset=<span class="hljs-string">&quot;wikitext2&quot;</span>
)<!----></pre></div><!----> <p>Note: <code>OVWeightQuantizationConfig</code> also accepts keyword arguments that are not listed in its constructor. In this case such arguments will be passed directly to <code>nncf.compress_weights()</code> call. This is useful for passing additional parameters to the quantization algorithm.</p> <p>By default the quantization scheme will be <a href="https://github.com/openvinotoolkit/nncf/blob/develop/docs/usage/training_time_compression/other_algorithms/LegacyQuantization.md#asymmetric-quantization" rel="nofollow">asymmetric</a>, to make it <a href="https://github.com/openvinotoolkit/nncf/blob/develop/docs/usage/training_time_compression/other_algorithms/LegacyQuantization.md#symmetric-quantization" rel="nofollow">symmetric</a> you can add <code>sym=True</code>.</p> <p>For 4-bit quantization you can also specify the following arguments in the quantization configuration :</p> <ul><li>The <code>group_size</code> parameter will define the group size to use for quantization, <code>-1</code> it will results in per-column quantization.</li> <li>The <code>ratio</code> parameter controls the ratio between 4-bit and 8-bit quantization. If set to 0.9, it means that 90% of the layers will be quantized to <code>int4</code> while 10% will be quantized to <code>int8</code>.</li></ul> <p>Smaller <code>group_size</code> and <code>ratio</code> values usually improve accuracy at the sacrifice of the model size and inference latency.</p> <p>Quality of 4-bit weight compressed model can further be improved by employing one of the following data-dependent methods:</p> <ul><li><strong>AWQ</strong> which stands for Activation Aware Quantization is an algorithm that tunes model weights for more accurate 4-bit compression. It slightly improves generation quality of compressed LLMs, but requires significant additional time and memory for tuning weights on a calibration dataset. Please note that it is possible that there will be no matching patterns in the model to apply AWQ, in such case it will be skipped. There is also a data-free version of AWQ available that relies on per-column magnitudes of weights instead of activations.</li> <li><strong>Scale Estimation</strong> is a method that tunes quantization scales to minimize the <code>L2</code> error between the original and compressed layers. Providing a dataset is required to run scale estimation. Using this method also incurs additional time and memory overhead.</li> <li><strong>GPTQ</strong> optimizes compressed weights in a layer-wise fashion to minimize the difference between activations of a compressed and original layer.</li> <li><strong>LoRA Correction</strong> mitigates quantization noise introduced during weight compression by leveraging low-rank adaptation.</li></ul> <p>Data-aware algorithms can be applied together or separately. For that, provide corresponding arguments to the 4-bit <code>OVWeightQuantizationConfig</code> together with a dataset. For example:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->quantization_config = OVWeightQuantizationConfig(
bits=<span class="hljs-number">4</span>,
sym=<span class="hljs-literal">False</span>,
ratio=<span class="hljs-number">0.8</span>,
quant_method=<span class="hljs-string">&quot;awq&quot;</span>,
scale_estimation=<span class="hljs-literal">True</span>,
gptq=<span class="hljs-literal">True</span>,
dataset=<span class="hljs-string">&quot;wikitext2&quot;</span>
)<!----></pre></div><!----> <p>Note: GPTQ and LoRA Correction algorithms can’t be applied simultaneously.</p> <!--[1--><h2 class="relative group"><a id="full-quantization" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#full-quantization"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Full quantization</span></h2><!--]--><!----> <p>When applying post-training full quantization, both the weights and the activations are quantized.
To apply quantization on the activations, an additional calibration step is needed which consists in feeding a <code>calibration_dataset</code> to the network in order to estimate the quantization activations parameters.</p> <p>Here is how to apply full quantization on a fine-tuned DistilBERT given your own <code>calibration_dataset</code>:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer
<span class="hljs-keyword">from</span> optimum.intel <span class="hljs-keyword">import</span> OVQuantizer, OVModelForSequenceClassification, OVConfig, OVQuantizationConfig
model_id = <span class="hljs-string">&quot;distilbert-base-uncased-finetuned-sst-2-english&quot;</span>
model = OVModelForSequenceClassification.from_pretrained(model_id, export=<span class="hljs-literal">True</span>)
tokenizer = AutoTokenizer.from_pretrained(model_id)
<span class="hljs-comment"># The directory where the quantized model will be saved</span>
save_dir = <span class="hljs-string">&quot;ptq_model&quot;</span>
quantizer = OVQuantizer.from_pretrained(model)
<span class="hljs-comment"># Apply full quantization and export the resulting quantized model to OpenVINO IR format</span>
ov_config = OVConfig(quantization_config=OVQuantizationConfig())
quantizer.quantize(ov_config=ov_config, calibration_dataset=calibration_dataset, save_directory=save_dir)
<span class="hljs-comment"># Save the tokenizer</span>
tokenizer.save_pretrained(save_dir)<!----></pre></div><!----> <p>The calibration dataset can also be created easily using your <code>OVQuantizer</code>:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> functools <span class="hljs-keyword">import</span> partial
<span class="hljs-keyword">def</span> <span class="hljs-title function_">preprocess_function</span>(<span class="hljs-params">examples, tokenizer</span>):
<span class="hljs-keyword">return</span> tokenizer(examples[<span class="hljs-string">&quot;sentence&quot;</span>], padding=<span class="hljs-string">&quot;max_length&quot;</span>, max_length=<span class="hljs-number">128</span>, truncation=<span class="hljs-literal">True</span>)
<span class="hljs-comment"># Create the calibration dataset used to perform full quantization</span>
calibration_dataset = quantizer.get_calibration_dataset(
<span class="hljs-string">&quot;glue&quot;</span>,
dataset_config_name=<span class="hljs-string">&quot;sst2&quot;</span>,
preprocess_function=partial(preprocess_function, tokenizer=tokenizer),
num_samples=<span class="hljs-number">300</span>,
dataset_split=<span class="hljs-string">&quot;train&quot;</span>,
)<!----></pre></div><!----> <p>The <code>quantize()</code> method applies post-training quantization and export the resulting quantized model to the OpenVINO Intermediate Representation (IR). The resulting graph is represented with two files: an XML file describing the network topology and a binary file describing the weights. The resulting model can be run on any target Intel device.</p> <!--[2--><h3 class="relative group"><a id="speech-to-text-models-quantization" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#speech-to-text-models-quantization"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Speech-to-text Models Quantization</span></h3><!--]--><!----> <p>The speech-to-text Whisper model can be quantized without the need for preparing a custom calibration dataset. Please see example below.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->model_id = <span class="hljs-string">&quot;openai/whisper-tiny&quot;</span>
ov_model = OVModelForSpeechSeq2Seq.from_pretrained(
model_id,
quantization_config=OVQuantizationConfig(
num_samples=<span class="hljs-number">10</span>,
dataset=<span class="hljs-string">&quot;librispeech&quot;</span>,
processor=model_id,
smooth_quant_alpha=<span class="hljs-number">0.95</span>,
)
)<!----></pre></div><!----> <p>With this, encoder and decoder models of the Whisper pipeline will be fully quantized, including activations.</p> <!--[1--><h2 class="relative group"><a id="hybrid-quantization" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#hybrid-quantization"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Hybrid quantization</span></h2><!--]--><!----> <p>Traditional optimization methods like post-training 8-bit quantization do not work well for Stable Diffusion (SD) models and can lead to poor generation results. On the other hand, weight compression does not improve performance significantly when applied to Stable Diffusion models, as the size of activations is comparable to weights.
The U-Net component takes up most of the overall execution time of the pipeline. Thus, optimizing just this one component can bring substantial benefits in terms of inference speed while keeping acceptable accuracy without fine-tuning. Quantizing the rest of the diffusion pipeline does not significantly improve inference performance but could potentially lead to substantial accuracy degradation.
Therefore, the proposal is to apply quantization in <em>hybrid mode</em> for the U-Net model and weight-only quantization for the rest of the pipeline components :</p> <ul><li>U-Net : quantization applied on both the weights and activations</li> <li>The text encoder, VAE encoder / decoder : quantization applied on the weights</li></ul> <p>The hybrid mode involves the quantization of weights in MatMul and Embedding layers, and activations of other layers, facilitating accuracy preservation post-optimization while reducing the model size.</p> <p>The <code>quantization_config</code> is utilized to define optimization parameters for optimizing the SD pipeline. To enable hybrid quantization, specify the quantization dataset in the <code>quantization_config</code>. If the dataset is not defined, weight-only quantization will be applied on all components.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> optimum.intel <span class="hljs-keyword">import</span> OVStableDiffusionPipeline, OVWeightQuantizationConfig
model = OVStableDiffusionPipeline.from_pretrained(
model_id,
export=<span class="hljs-literal">True</span>,
quantization_config=OVWeightQuantizationConfig(bits=<span class="hljs-number">8</span>, dataset=<span class="hljs-string">&quot;conceptual_captions&quot;</span>),
)<!----></pre></div><!----> <p>For more details, please refer to the corresponding NNCF <a href="https://github.com/openvinotoolkit/nncf/blob/develop/docs/usage/post_training_compression/weights_compression/Usage.md" rel="nofollow">documentation</a>.</p> <!--[1--><h2 class="relative group"><a id="mixed-quantization" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#mixed-quantization"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Mixed Quantization</span></h2><!--]--><!----> <p>Mixed quantization is a technique that combines weight-only quantization with full quantization. During mixed quantization we separately quantize:</p> <ol><li>weights of weighted layers to one precision, and</li> <li>activations (and possibly, weights, if some were skipped at the first step) of other supported layers to another precision.</li></ol> <p>By default, weights of all weighted layers are quantized in the first step. In the second step activations of weighted and non-weighted layers are quantized. If some layers are instructed to be ignored in the first step with <code>weight_quantization_config.ignored_scope</code> parameter, both weights and activations of these layers are quantized to the precision given in the <code>full_quantization_config</code>.</p> <p>When running this kind of optimization through Python API, <code>OVMixedQuantizationConfig</code> should be used. In such case the precision for the first step should be provided with <code>weight_quantization_config</code> argument and the precision for the second step with <code>full_quantization_config</code> argument. For example:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!---->model = OVModelForCausalLM.from_pretrained(
<span class="hljs-string">&#x27;TinyLlama/TinyLlama-1.1B-Chat-v1.0&#x27;</span>,
quantization_config=OVMixedQuantizationConfig(
weight_quantization_config=OVWeightQuantizationConfig(bits=<span class="hljs-number">4</span>, dtype=<span class="hljs-string">&#x27;cb4&#x27;</span>),
full_quantization_config=OVQuantizationConfig(dtype=<span class="hljs-string">&#x27;f8e4m3&#x27;</span>, dataset=<span class="hljs-string">&#x27;wikitext2&#x27;</span>)
)
)<!----></pre></div><!----> <p>To apply mixed quantization through CLI, the <code>--quant-mode</code> argument should be used. For example:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-bash "><!---->optimum-cli <span class="hljs-built_in">export</span> openvino -m TinyLlama/TinyLlama-1.1B-Chat-v1.0 --quant-mode cb4_f8e4m3 --dataset wikitext2 ./save_dir<!----></pre></div><!----> <p>Don’t forget to provide a dataset since it is required for the calibration procedure during full quantization.</p> <!--[1--><h2 class="relative group"><a id="pipeline-quantization" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#pipeline-quantization"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Pipeline Quantization</span></h2><!--]--><!----> <p>There are multimodal pipelines that consist of multiple components, such as Stable Diffusion or Visual Language models. In these cases, there may be a need to apply different quantization methods to different components of the pipeline. For example, you may want to apply int4 data-aware weight-only quantization to a language model in visual-language pipeline, while applying int8 weight-only quantization to other components. In this case you can use the <code>OVPipelineQuantizationConfig</code> class to specify the quantization configuration for each component of the pipeline.</p> <p>For example, the code below quantizes weights and activations of a language model inside InternVL2-1B, compresses weights of a text embedding model and skips any quantization for vision embedding model.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> optimum.intel <span class="hljs-keyword">import</span> OVModelForVisualCausalLM
<span class="hljs-keyword">from</span> optimum.intel <span class="hljs-keyword">import</span> OVPipelineQuantizationConfig, OVQuantizationConfig, OVWeightQuantizationConfig
model_id = <span class="hljs-string">&quot;OpenGVLab/InternVL2-1B&quot;</span>
model = OVModelForVisualCausalLM.from_pretrained(
model_id,
export=<span class="hljs-literal">True</span>,
trust_remote_code=<span class="hljs-literal">True</span>,
quantization_config=OVPipelineQuantizationConfig(
quantization_configs={
<span class="hljs-string">&quot;lm_model&quot;</span>: OVQuantizationConfig(bits=<span class="hljs-number">8</span>),
<span class="hljs-string">&quot;text_embeddings_model&quot;</span>: OVWeightQuantizationConfig(bits=<span class="hljs-number">8</span>),
},
dataset=<span class="hljs-string">&quot;textvqa&quot;</span>,
)
)<!----></pre></div><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]-->
<script>
{
__sveltekit_1ofgn6n = {
base: "/docs/optimum.intel/pr_1908/en",
assets: "/docs/optimum.intel/pr_1908/en"
};
const element = document.currentScript.parentElement;
Promise.all([
import("/docs/optimum.intel/pr_1908/en/_app/immutable/entry/start.CatJ25dz.js"),
import("/docs/optimum.intel/pr_1908/en/_app/immutable/entry/app.KpdgG_wN.js")
]).then(([kit, app]) => {
kit.start(app, element, {
node_ids: [0, 7],
data: [null,null],
form: null,
error: null
});
});
}
</script>

Xet Storage Details

Size:
61.7 kB
·
Xet hash:
842093feb0060783813bf53a88e088f7b3e1073c3381336c829edc0f30b0bb48

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.