Buckets:
| <meta charset="utf-8" /><meta name="hf:doc:metadata" content="{"title":"vLLM","local":"vllm","sections":[{"title":"Configuration","local":"configuration","sections":[],"depth":2},{"title":"Supported models","local":"supported-models","sections":[],"depth":2},{"title":"Parallelism and Scaling","local":"parallelism-and-scaling","sections":[{"title":"Default Behavior on Inference Endpoints","local":"default-behavior-on-inference-endpoints","sections":[],"depth":3},{"title":"Tensor Parallelism (TP)","local":"tensor-parallelism-tp","sections":[],"depth":3},{"title":"Data Parallelism (DP)","local":"data-parallelism-dp","sections":[],"depth":3},{"title":"Combining TP and DP","local":"combining-tp-and-dp","sections":[{"title":"Optimizing for Throughput","local":"optimizing-for-throughput","sections":[],"depth":4}],"depth":3},{"title":"Choosing the Right Configuration","local":"choosing-the-right-configuration","sections":[],"depth":3},{"title":"Common Mistakes","local":"common-mistakes","sections":[],"depth":3}],"depth":2},{"title":"References","local":"references","sections":[],"depth":2}],"depth":1}"/> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/entry/start.D6CMpMkr.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/CqyW0xB4.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/DkE5gPSR.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/entry/app.CK3kxEa6.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/DBl7_-Yh.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/BNeWj5ME.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/DG0Iw8HZ.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/nodes/0.CyhimaJ6.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/B3lpbLb1.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/nodes/10.CbOAQtag.js" rel="modulepreload"> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/chunks/DQX3K91K.js" rel="modulepreload"> | |
| <!--1beq9yi--><meta name="hf:doc:metadata" content="{"title":"vLLM","local":"vllm","sections":[{"title":"Configuration","local":"configuration","sections":[],"depth":2},{"title":"Supported models","local":"supported-models","sections":[],"depth":2},{"title":"Parallelism and Scaling","local":"parallelism-and-scaling","sections":[{"title":"Default Behavior on Inference Endpoints","local":"default-behavior-on-inference-endpoints","sections":[],"depth":3},{"title":"Tensor Parallelism (TP)","local":"tensor-parallelism-tp","sections":[],"depth":3},{"title":"Data Parallelism (DP)","local":"data-parallelism-dp","sections":[],"depth":3},{"title":"Combining TP and DP","local":"combining-tp-and-dp","sections":[{"title":"Optimizing for Throughput","local":"optimizing-for-throughput","sections":[],"depth":4}],"depth":3},{"title":"Choosing the Right Configuration","local":"choosing-the-right-configuration","sections":[],"depth":3},{"title":"Common Mistakes","local":"common-mistakes","sections":[],"depth":3}],"depth":2},{"title":"References","local":"references","sections":[],"depth":2}],"depth":1}"/><!----> | |
| <link href="/docs/inference-endpoints/pr_170/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg><!----></button></div> <!--[-1--><!--]--></div><!----> <!--[0--><h1 class="relative group"><a id="vllm" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#vllm"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>vLLM</span></h1><!--]--><!----> <p>vLLM is a high-performance, memory-efficient inference engine for open-source LLMs. It delivers efficient scheduling, KV-cache handling, | |
| batching, and decoding—all wrapped in a production-ready server. For most use cases, TGI, vLLM, and SGLang will be equivalently good options.</p> <p><strong>Core features</strong>:</p> <ul><li><strong>PagedAttention for memory efficiency</strong></li> <li><strong>Continuous batching</strong></li> <li><strong>Optimized CUDA/HIP execution</strong></li> <li><strong>Speculative decoding & chunked prefill</strong></li> <li><strong>Multi-backend and hardware support</strong>: Runs across NVIDIA, AMD, and AWS Neuron to name a few</li></ul> <!--[1--><h2 class="relative group"><a id="configuration" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#configuration"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Configuration</span></h2><!--]--><!----> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/vllm/vllm_config.png" alt="config"/></p> <ul><li><strong>Max Number of Sequences</strong>: The maximum number of sequences (requests) that can be processed together in a single batch. Controls | |
| the batch size by sequence count, affecting throughput and memory usage. For example, if max_num_seqs=8, up to 8 different prompts can | |
| be handled at once, regardless of their individual lengths, as long as the total token count also fits within the Max Number of Batched Tokens.</li> <li><strong>Max Number of Batched Tokens</strong>: The maximum total number of tokens (summed across all sequences) that can be processed in a single | |
| batch. Limits batch size by token count, balancing throughput and GPU memory allocation.</li> <li><strong>Tensor Parallel Size</strong>: The number of GPUs across which model weights are split within each layer. Increasing this allows larger | |
| models to run and frees up GPU memory for KV cache, but may introduce synchronization overhead.</li> <li><strong>KV Cache DType</strong>: the data type used for storing the key-value cache during generation. Options include “auto”, “fp8”, “fp8_e5m2”, | |
| and “fp8_e4m3”. Using lower precision types can reduce memory usage but may slightly impact generation quality.</li></ul> <p>For more advanced configuration you can pass any of the <a href="https://docs.vllm.ai/en/stable/api/vllm/engine/arg_utils.html#vllm.engine.arg_utils.EngineArgs" rel="nofollow">Engine Arguments that vLLM supports</a> as container arguments. For example changing the <code>enable_lora</code> to <code>true</code> would look like this:</p> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/vllm/vllm-advanced.png" alt="vllm-advanced"/></p> <!--[1--><h2 class="relative group"><a id="supported-models" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#supported-models"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Supported models</span></h2><!--]--><!----> <p>vLLM has wide support for large language models and embedding models. We recommend reading the <a href="https://docs.vllm.ai/en/stable/models/supported_models.html?h=supported+models" rel="nofollow">supported models</a> section in the vLLM documentation for a full list.</p> <p>vLLM also supports model implementations that are available in Transformers. Currently not all models work but support is planned for most | |
| decoder language models are supported, and vision language models.</p> <!--[1--><h2 class="relative group"><a id="parallelism-and-scaling" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#parallelism-and-scaling"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Parallelism and Scaling</span></h2><!--]--><!----> <p>vLLM supports several parallelism strategies for distributed inference. The two most common ones are <strong>Tensor Parallelism (TP)</strong> and <strong>Data Parallelism (DP)</strong>. Understanding when and how to use each is essential for optimal performance.</p> <!--[2--><h3 class="relative group"><a id="default-behavior-on-inference-endpoints" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#default-behavior-on-inference-endpoints"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Default Behavior on Inference Endpoints</span></h3><!--]--><!----> <p>When you create an endpoint, after you’ve selected an instance type (e.g., 4 × A10G, 8 × H100). The defaults are:</p> <ul><li><strong><code>tensor_parallel_size</code></strong> = number of GPUs on the instance (shards the model across all GPUs)</li> <li><strong><code>data_parallel_size</code></strong> = 1 (single copy of the model)</li></ul> <p>This default configuration prioritizes fitting larger models by using all available GPU memory. However, you might want to tweak these settings if your model fits on fewer GPUs than your instance has and you want <strong>higher throughput</strong> by running multiple copies of the model.</p> <!--[2--><h3 class="relative group"><a id="tensor-parallelism-tp" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#tensor-parallelism-tp"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Tensor Parallelism (TP)</span></h3><!--]--><!----> <p>Tensor parallelism splits the model’s weights across multiple GPUs within each layer. Each GPU holds a slice of the model and computes its portion of the output, then synchronizes with other GPUs.</p> <p><strong>When to use:</strong> Your model is too large to fit on a single GPU. You must set <code>tensor_parallel_size</code> to at least the number of GPUs required to hold the model in memory.</p> <p><strong>Example:</strong></p> <ul><li><strong>Llama 3 8B (FP16)</strong> requires ~16GB → fits on 1 GPU → <code>tensor_parallel_size=1</code></li> <li><strong>Llama 3 70B (FP16)</strong> requires ~140GB → needs 2 × 80GB GPUs → <code>tensor_parallel_size=2</code></li> <li><strong>Llama 3.1 405B (FP16)</strong> requires ~810GB → needs 8 × 80GB GPUs → <code>tensor_parallel_size=8</code></li></ul> <!--[2--><h3 class="relative group"><a id="data-parallelism-dp" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#data-parallelism-dp"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Data Parallelism (DP)</span></h3><!--]--><!----> <p>Data parallelism runs multiple independent copies of the model on different GPUs. Each copy handles different requests independently, increasing throughput.</p> <p><strong>When to use:</strong> You want higher throughput and your model fits on fewer GPUs than your instance provides.</p> <p><strong>Configuration:</strong> Set <code>data_parallel_size</code> to the number of copies you want.</p> <!--[2--><h3 class="relative group"><a id="combining-tp-and-dp" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#combining-tp-and-dp"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Combining TP and DP</span></h3><!--]--><!----> <p>On multi-GPU instances, you can combine both strategies. The key formula is <code>tensor_parallel_size × data_parallel_size = total GPUs on instance</code>.</p> <!--[3--><h4 class="relative group"><a id="optimizing-for-throughput" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#optimizing-for-throughput"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Optimizing for Throughput</span></h4><!--]--><!----> <p>If your model fits on a single GPU but you want high throughput, lower TP and increase DP to run multiple copies of the model.</p> <p><strong>Example:</strong> Serving Llama 3 8B (~16GB) on a 4 × A100 80GB instance:</p> <table><thead><tr><th>Configuration</th><th>TP</th><th>DP</th><th>Copies</th><th>Behavior</th></tr></thead><tbody><tr><td>Default</td><td>4</td><td>1</td><td>1</td><td>Model sharded across all 4 GPUs</td></tr><tr><td>Balanced</td><td>2</td><td>2</td><td>2</td><td>2 copies, each sharded across 2 GPUs</td></tr><tr><td>Max throughput</td><td>1</td><td>4</td><td>4</td><td>4 independent copies</td></tr></tbody></table> <p>There’s always a trade-off:</p> <ul><li><strong>Higher DP</strong> (more copies) → higher throughput, but each copy has less memory for KV cache (shorter context)</li> <li><strong>Higher TP</strong> (fewer copies) → more memory per copy for KV cache (longer context), but lower throughput</li></ul> <p>For example, <code>tensor_parallel_size=2</code> and <code>data_parallel_size=2</code> gives you 2 copies that can each handle longer contexts than the max throughput configuration, while still doubling your request capacity compared to the default.</p> <!--[2--><h3 class="relative group"><a id="choosing-the-right-configuration" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#choosing-the-right-configuration"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Choosing the Right Configuration</span></h3><!--]--><!----> <ol><li><strong>Calculate minimum TP:</strong> How many GPUs are needed to fit your model in memory?</li> <li><strong>Set TP to that minimum</strong></li> <li><strong>Set DP</strong> = (total instance GPUs) ÷ TP</li></ol> <p><strong>Example:</strong> You want to deploy Llama 3 70B on 8 × H100 80GB.</p> <ul><li>Model needs ~140GB → minimum 2 × 80GB GPUs → <code>tensor_parallel_size=2</code></li> <li>Instance has 8 GPUs → <code>data_parallel_size=8÷2=4</code></li> <li>Result: 4 copies, each on 2 GPUs</li></ul> <!--[2--><h3 class="relative group"><a id="common-mistakes" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#common-mistakes"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Common Mistakes</span></h3><!--]--><!----> <table><thead><tr><th>Configuration</th><th>Problem</th><th>Solution</th></tr></thead><tbody><tr><td>TP=1, DP=1 for 7B on 4 × A10G</td><td>3 GPUs sitting idle</td><td>Increase <code>data_parallel_size=4</code></td></tr><tr><td>TP=1 for 70B on single 80GB GPU</td><td>Out of memory</td><td>Use an instance with at least 2 × 80GB GPU and make sure <code>tensor_parallel_size=2</code></td></tr><tr><td>TP=2, DP=4 on 4 × A10G</td><td>Fails since 2 × 4 = 8 GPUs required, but only 4 available</td><td>Reduce to TP=2, DP=2 or TP=1, DP=4</td></tr><tr><td>TP=3, DP=1 on 4 × A10G</td><td>1 GPU sits completely idle</td><td>Use TP=4 or TP=2 with DP=2</td></tr></tbody></table> <!--[1--><h2 class="relative group"><a id="references" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#references"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>References</span></h2><!--]--><!----> <p>We also recommend reading the <a href="https://docs.vllm.ai/en/stable/" rel="nofollow">vLLM documentation</a> for more in-depth information.</p> <a class="!text-gray-400 !no-underline text-sm flex items-center not-prose mt-4" href="https://github.com/huggingface/hf-endpoints-documentation/blob/main/docs/source/engines/vllm.md" target="_blank"><svg class="mr-1" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M31,16l-7,7l-1.41-1.41L28.17,16l-5.58-5.59L24,9l7,7z"></path><path d="M1,16l7-7l1.41,1.41L3.83,16l5.58,5.59L8,23l-7-7z"></path><path d="M12.419,25.484L17.639,6.552l1.932,0.518L14.351,26.002z"></path></svg><!----> <span><span class="underline">Update</span> on GitHub</span></a><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]--> | |
| <script> | |
| { | |
| __sveltekit_sljs6z = { | |
| base: "/docs/inference-endpoints/pr_170/en", | |
| assets: "/docs/inference-endpoints/pr_170/en" | |
| }; | |
| const element = document.currentScript.parentElement; | |
| Promise.all([ | |
| import("/docs/inference-endpoints/pr_170/en/_app/immutable/entry/start.D6CMpMkr.js"), | |
| import("/docs/inference-endpoints/pr_170/en/_app/immutable/entry/app.CK3kxEa6.js") | |
| ]).then(([kit, app]) => { | |
| kit.start(app, element, { | |
| node_ids: [0, 10], | |
| data: [null,null], | |
| form: null, | |
| error: null | |
| }); | |
| }); | |
| } | |
| </script> | |
Xet Storage Details
- Size:
- 29.3 kB
- Xet hash:
- 5bd9e7caa7412b5632917057cacbd783ce068df7130fd2f7a2e16a545149461f
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.