Buckets:
| <meta charset="utf-8" /><meta name="hf:doc:metadata" content="{"title":"Deep Learning Containers","local":"deep-learning-containers","sections":[{"title":"Features & benefits","local":"features--benefits","sections":[],"depth":2},{"title":"Available DLCs","local":"available-dlcs","sections":[{"title":"Transformers","local":"transformers","sections":[{"title":"Training","local":"training","sections":[],"depth":4},{"title":"Inference","local":"inference","sections":[],"depth":4}],"depth":3},{"title":"vLLM","local":"vllm","sections":[{"title":"vLLM Omni","local":"vllm-omni","sections":[],"depth":4}],"depth":3},{"title":"SGLang","local":"sglang","sections":[],"depth":3},{"title":"Llama.cpp","local":"llamacpp","sections":[],"depth":3},{"title":"Text Embeddings Inference","local":"text-embeddings-inference","sections":[],"depth":3}],"depth":2},{"title":"FAQ","local":"faq","sections":[],"depth":2}],"depth":1}"/> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/entry/start.D0qXhpFD.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/D1ekGocw.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/BAlBQgxm.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/entry/app.BJw2fAxO.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/CShA6w_M.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/Ds9Xs-bq.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/BdEMyugS.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/nodes/0.BLk_T2Pd.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/BchC4-7E.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/nodes/7.BSQiWFzV.js" rel="modulepreload"> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/chunks/NrhFkieU.js" rel="modulepreload"> | |
| <!--8kua14--><meta name="hf:doc:metadata" content="{"title":"Deep Learning Containers","local":"deep-learning-containers","sections":[{"title":"Features & benefits","local":"features--benefits","sections":[],"depth":2},{"title":"Available DLCs","local":"available-dlcs","sections":[{"title":"Transformers","local":"transformers","sections":[{"title":"Training","local":"training","sections":[],"depth":4},{"title":"Inference","local":"inference","sections":[],"depth":4}],"depth":3},{"title":"vLLM","local":"vllm","sections":[{"title":"vLLM Omni","local":"vllm-omni","sections":[],"depth":4}],"depth":3},{"title":"SGLang","local":"sglang","sections":[],"depth":3},{"title":"Llama.cpp","local":"llamacpp","sections":[],"depth":3},{"title":"Text Embeddings Inference","local":"text-embeddings-inference","sections":[],"depth":3}],"depth":2},{"title":"FAQ","local":"faq","sections":[],"depth":2}],"depth":1}"/><!----> | |
| <link href="/docs/sagemaker/pr_2709/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg><!----></button></div> <!--[-1--><!--]--></div><!----> <!--[0--><h1 class="relative group"><a id="deep-learning-containers" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#deep-learning-containers"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Deep Learning Containers</span></h1><!--]--><!----> <p>Hugging Face, together with Amazon Web Services, builds and maintains Deep Learning Containers (DLCs) so you can run your machine learning workloads in an optimized environment with no configuration or maintenance on your part. These are Docker images pre-installed with popular frameworks and libraries such as 🤗 Transformers, 🤗 Datasets, and 🤗 Tokenizers, alongside high-performance serving engines. The DLCs let you serve and train models directly, skipping the complex process of building and optimizing your own environments from scratch.</p> <p>The containers are publicly maintained, updated, and released periodically by Hugging Face and the AWS team, and are available to all AWS customers in the <a href="https://aws.github.io/deep-learning-containers/reference/available_images/#huggingface-vllm-inference" rel="nofollow">Amazon Elastic Container Registry (ECR)</a>. You can use them in <strong>Amazon SageMaker AI</strong>: a fully managed platform to build, train, and deploy ML models into a production-ready hosted environment.</p> <p>Hugging Face DLCs are open source and licensed under Apache 2.0. Browse the full list of images and versions in the <a href="#available-dlcs">Available DLCs</a> section below, and feel free to reach out on our <a href="https://discuss.huggingface.co/c/sagemaker/17" rel="nofollow">community forum</a> if you have any questions.</p> <div class="grid grid-cols-2 gap-3 sm:grid-cols-3 lg:grid-cols-5 my-6 not-prose"><a class="group rounded-xl border border-gray-200 px-4 py-3 no-underline! dark:border-gray-800" href="#vllm"><div class="font-semibold text-gray-900 group-hover:text-orange-600 dark:text-white dark:group-hover:text-orange-300">vLLM</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">LLM serving on GPU and Neuron</p></a> <a class="group rounded-xl border border-gray-200 px-4 py-3 no-underline! dark:border-gray-800" href="#sglang"><div class="font-semibold text-gray-900 group-hover:text-orange-600 dark:text-white dark:group-hover:text-orange-300">SGLang</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">Fast serving on GPU</p></a> <a class="group rounded-xl border border-gray-200 px-4 py-3 no-underline! dark:border-gray-800" href="#llamacpp"><div class="font-semibold text-gray-900 group-hover:text-orange-600 dark:text-white dark:group-hover:text-orange-300">llama.cpp</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">Lightweight GGUF serving</p></a> <a class="group rounded-xl border border-gray-200 px-4 py-3 no-underline! dark:border-gray-800" href="#text-embeddings-inference"><div class="font-semibold text-gray-900 group-hover:text-orange-600 dark:text-white dark:group-hover:text-orange-300">TEI</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">Embeddings and reranking</p></a> <a class="group rounded-xl border border-gray-200 px-4 py-3 no-underline! dark:border-gray-800" href="#transformers"><div class="font-semibold text-gray-900 group-hover:text-orange-600 dark:text-white dark:group-hover:text-orange-300">Transformers</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">Training and general inference</p></a></div> <!--[1--><h2 class="relative group"><a id="features--benefits" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#features--benefits"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Features & benefits</span></h2><!--]--><!----> <div class="grid grid-cols-1 gap-4 sm:grid-cols-2 my-6 not-prose"><div class="rounded-xl border border-gray-200 px-5 py-4 dark:border-gray-800"><div class="font-semibold text-gray-900 dark:text-white">One command to train</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">The training DLCs ship with everything needed to run a single command — for example the <a href="https://huggingface.co/docs/trl/en/clis">TRL CLI</a> — to fine-tune LLMs from single-GPU to multi-node multi-GPU.</p></div> <div class="rounded-xl border border-gray-200 px-5 py-4 dark:border-gray-800"><div class="font-semibold text-gray-900 dark:text-white">Production serving engines</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">Dedicated DLCs built around <a href="https://docs.vllm.ai/">vLLM</a>, <a href="https://docs.sglang.ai/">SGLang</a>, and <a href="https://github.com/ggml-org/llama.cpp">llama.cpp</a> serve most Hub text-generation architectures with OpenAI-compatible APIs and direct Amazon S3 model loading.</p></div> <div class="rounded-xl border border-gray-200 px-5 py-4 dark:border-gray-800"><div class="font-semibold text-gray-900 dark:text-white">Embeddings and reranking</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">The <a href="https://huggingface.co/docs/text-embeddings-inference">Text Embeddings Inference (TEI)</a> DLC serves embedding, re-ranking, and sequence-classification models on CPU and GPU, including the thousands of <a href="https://huggingface.co/models?other=text-embeddings-inference">supported models</a> on the Hub.</p></div> <div class="rounded-xl border border-gray-200 px-5 py-4 dark:border-gray-800"><div class="font-semibold text-gray-900 dark:text-white">Built-in performance</div> <p class="mt-1 text-sm text-gray-600 dark:text-gray-400">Tested, optimized environments with production-ready endpoints that scale with your AWS environment, so you can pick infrastructure by price/performance target.</p></div></div> <!--[1--><h2 class="relative group"><a id="available-dlcs" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#available-dlcs"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Available DLCs</span></h2><!--]--><!----> <p>Below you can find a listing of our latest Deep Learning Containers (DLCs) available on AWS.</p> <p>For each supported combination of use-case (training, inference), accelerator type (CPU, GPU, Neuron), and framework (PyTorch, vLLM, SGLang, llama.cpp, TEI) containers are created. The URIs below use <code>us-east-1</code> or <code>us-west-2</code>; replace the region as needed, or <a href="#faq">retrieve the URI programmatically</a>.</p> <p>Neuron DLCs for training and inference on AWS Trainium and AWS Inferentia instances can be found in the <a href="https://huggingface.co/docs/optimum-neuron/en/containers" rel="nofollow">Optimum Neuron documentation</a>. To keep track of all our available DLCs, check the <a href="https://aws.github.io/deep-learning-containers/reference/available_images#huggingface-pytorch-training" rel="nofollow">AWS Deep Learning Containers releases</a> page.</p> <!--[2--><h3 class="relative group"><a id="transformers" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#transformers"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Transformers</span></h3><!--]--><!----> <!--[3--><h4 class="relative group"><a id="training" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#training"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Training</span></h4><!--]--><!----> <p>For training, the DLCs are available for PyTorch via Transformers. They include GPUs and AWS AI chips support, with libraries such as TRL, Sentence Transformers, or Diffusers. You can also keep track of the latest PyTorch Training DLC releases <a href="https://github.com/aws/deep-learning-containers/releases?q=huggingface-training+AND+NOT+neuronx&expanded=true" rel="nofollow">here</a>.</p> <div class="overflow-x-auto my-4 not-prose"><table class="w-full text-sm"><thead><tr class="border-b border-gray-200 dark:border-gray-800"><th class="py-2 pr-4 text-left font-semibold">Accelerator</th><th class="py-2 text-left font-semibold">Container URI</th></tr></thead><tbody><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">GPU</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-east-1.amazonaws.com/huggingface-pytorch-training:2.9.0-transformers5.3.0-gpu-py312-cu130-ubuntu22.04</code></td></tr><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">Neuron</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-west-2.amazonaws.com/huggingface-pytorch-training-neuronx:2.8.0-transformers4.55.4-neuronx-py310-sdk2.26.0-ubuntu22.04</code></td></tr></tbody></table></div> <!--[3--><h4 class="relative group"><a id="inference" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#inference"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Inference</span></h4><!--]--><!----> <p>For inference, the general-purpose PyTorch inference DLC serves models trained with any of those frameworks on CPU, GPU, and AWS AI chips.</p> <div class="overflow-x-auto my-4 not-prose"><table class="w-full text-sm"><thead><tr class="border-b border-gray-200 dark:border-gray-800"><th class="py-2 pr-4 text-left font-semibold">Accelerator</th><th class="py-2 text-left font-semibold">Container URI</th></tr></thead><tbody><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">CPU</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-east-1.amazonaws.com/huggingface-pytorch-inference:2.6.0-transformers4.51.3-cpu-py312-ubuntu22.04</code></td></tr><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">GPU</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-east-1.amazonaws.com/huggingface-pytorch-inference:2.6.0-transformers4.51.3-gpu-py312-cu124-ubuntu22.04</code></td></tr><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">Neuron</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-west-2.amazonaws.com/huggingface-pytorch-inference-neuronx:2.8.0-transformers4.55.4-neuronx-py310-sdk2.26.0-ubuntu22.04</code></td></tr></tbody></table></div> <!--[2--><h3 class="relative group"><a id="vllm" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#vllm"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>vLLM</span></h3><!--]--><!----> <p>For serving text generation models with <a href="https://docs.vllm.ai/" rel="nofollow">vLLM</a>, there are specific DLCs available for GPU and AWS AI chips.</p> <div class="overflow-x-auto my-4 not-prose"><table class="w-full text-sm"><thead><tr class="border-b border-gray-200 dark:border-gray-800"><th class="py-2 pr-4 text-left font-semibold">Accelerator</th><th class="py-2 pr-4 text-left font-semibold">Version</th><th class="py-2 text-left font-semibold">Container URI</th></tr></thead><tbody><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">GPU</td><td class="py-2 pr-4 whitespace-nowrap">0.28.0</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-east-1.amazonaws.com/huggingface-vllm:0.28.0-transformers5.15.0-gpu-py312-cu130-ubuntu24.04</code></td></tr><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">Neuron</td><td class="py-2 pr-4 whitespace-nowrap">0.11.0</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-west-2.amazonaws.com/huggingface-vllm-inference-neuronx:0.11.0-optimum0.4.5-neuronx-py310-sdk2.26.1-ubuntu22.04</code></td></tr></tbody></table></div> <!--[3--><h4 class="relative group"><a id="vllm-omni" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#vllm-omni"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>vLLM Omni</span></h4><!--]--><!----> <p>You can also use vLLM Omni for serving multimodal models with vLLM on GPUs.</p> <div class="overflow-x-auto my-4 not-prose"><table class="w-full text-sm"><thead><tr class="border-b border-gray-200 dark:border-gray-800"><th class="py-2 pr-4 text-left font-semibold">Accelerator</th><th class="py-2 pr-4 text-left font-semibold">Version</th><th class="py-2 text-left font-semibold">Container URI</th></tr></thead><tbody><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">GPU</td><td class="py-2 pr-4 whitespace-nowrap">0.20.0</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-west-2.amazonaws.com/huggingface-vllm-omni:0.20.0-transformers5.8.1-gpu-py312-cu130-amzn2023</code></td></tr></tbody></table></div> <!--[2--><h3 class="relative group"><a id="sglang" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#sglang"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>SGLang</span></h3><!--]--><!----> <p>There is also a specific DLC for serving models with <a href="https://docs.sglang.ai/" rel="nofollow">SGLang</a> on GPU.</p> <div class="overflow-x-auto my-4 not-prose"><table class="w-full text-sm"><thead><tr class="border-b border-gray-200 dark:border-gray-800"><th class="py-2 pr-4 text-left font-semibold">Accelerator</th><th class="py-2 pr-4 text-left font-semibold">Version</th><th class="py-2 text-left font-semibold">Container URI</th></tr></thead><tbody><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">GPU</td><td class="py-2 pr-4 whitespace-nowrap">0.5.12</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-west-2.amazonaws.com/huggingface-sglang:0.5.12-transformers5.6.0-gpu-py312-cu130-ubuntu24.04</code></td></tr></tbody></table></div> <!--[2--><h3 class="relative group"><a id="llamacpp" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#llamacpp"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Llama.cpp</span></h3><!--]--><!----> <p>For lightweight inference serving, there is a specific DLC for serving models with <a href="https://github.com/ggml-org/llama.cpp" rel="nofollow">llama.cpp</a> on both CPU and GPU.</p> <div class="overflow-x-auto my-4 not-prose"><table class="w-full text-sm"><thead><tr class="border-b border-gray-200 dark:border-gray-800"><th class="py-2 pr-4 text-left font-semibold">Accelerator</th><th class="py-2 pr-4 text-left font-semibold">Version</th><th class="py-2 text-left font-semibold">Container URI</th></tr></thead><tbody><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">GPU</td><td class="py-2 pr-4 whitespace-nowrap">b9522</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-west-2.amazonaws.com/huggingface-llama.cpp:b9522-gpu-cu130-ubuntu24.04</code></td></tr><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">CPU</td><td class="py-2 pr-4 whitespace-nowrap">b9522</td><td class="py-2"><code class="text-xs break-all">763104351884.dkr.ecr.us-west-2.amazonaws.com/huggingface-llama.cpp:b9522-cpu-ubuntu24.04</code></td></tr></tbody></table></div> <!--[2--><h3 class="relative group"><a id="text-embeddings-inference" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#text-embeddings-inference"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Text Embeddings Inference</span></h3><!--]--><!----> <p>Finally, the <a href="https://huggingface.co/docs/text-embeddings-inference" rel="nofollow">Text Embeddings Inference (TEI)</a> DLC provides high-performance serving of embedding models on CPU and GPU.</p> <div class="overflow-x-auto my-4 not-prose"><table class="w-full text-sm"><thead><tr class="border-b border-gray-200 dark:border-gray-800"><th class="py-2 pr-4 text-left font-semibold">Accelerator</th><th class="py-2 text-left font-semibold">Container URI</th></tr></thead><tbody><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">CPU</td><td class="py-2"><code class="text-xs break-all">683313688378.dkr.ecr.us-east-1.amazonaws.com/tei-cpu:2.0.1-tei1.9.3-cpu-py310-ubuntu24.04</code></td></tr><tr class="border-b border-gray-100 dark:border-gray-850"><td class="py-2 pr-4 whitespace-nowrap">GPU</td><td class="py-2"><code class="text-xs break-all">683313688378.dkr.ecr.us-east-1.amazonaws.com/tei:2.0.1-tei1.9.3-gpu-py310-cu129-ubuntu24.04</code></td></tr></tbody></table></div> <!--[1--><h2 class="relative group"><a id="faq" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#faq"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>FAQ</span></h2><!--]--><!----> <details class="my-3 rounded-xl border border-gray-200 px-5 py-4 dark:border-gray-800"><summary class="cursor-pointer font-semibold text-gray-900 dark:text-white">How do I find the URI of my container?</summary> <p>The SageMaker SDK provides a utility function to get the URI of a container programmatically:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> sagemaker.core <span class="hljs-keyword">import</span> image_uris | |
| AVAILABLE_FRAMEWORKS = [ | |
| <span class="hljs-string">"huggingface"</span>, | |
| <span class="hljs-string">"huggingface-tei"</span>, | |
| <span class="hljs-string">"huggingface-llamacpp"</span>, | |
| <span class="hljs-string">"huggingface-vllm"</span>, | |
| <span class="hljs-string">"huggingface-vllm-omni"</span>, | |
| <span class="hljs-string">"huggingface-sglang"</span>, | |
| ] | |
| <span class="hljs-comment"># use image_scope="training" for training containers</span> | |
| image_uris.retrieve( | |
| <span class="hljs-string">"huggingface-vllm"</span>, | |
| region=<span class="hljs-string">"us-east-1"</span>, | |
| image_scope=<span class="hljs-string">"inference"</span>, | |
| instance_type=<span class="hljs-string">"ml.g5.2xlarge"</span>, | |
| )<!----></pre></div><!----></details> <details class="my-3 rounded-xl border border-gray-200 px-5 py-4 dark:border-gray-800"><summary class="cursor-pointer font-semibold text-gray-900 dark:text-white">Can the SDK choose the container for me?</summary> <p>If you just want the default container for a given model, you can rely on the SageMaker SDK <code>ModelBuilder</code>, which automatically chooses the container for you:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-python "><!----><span class="hljs-keyword">from</span> sagemaker.serve <span class="hljs-keyword">import</span> ModelBuilder | |
| builder = ModelBuilder( | |
| model=<span class="hljs-string">"google/gemma-4-E2B-it"</span>, | |
| instance_type=<span class="hljs-string">"ml.g5.2xlarge"</span>, | |
| role_arn=role, | |
| )<!----></pre></div><!----> <blockquote class="note"><p>The SDK may not always be up to date or may choose the wrong container for your use case. When in doubt, compare the container URI returned by the SDK with the ones listed on this page.</p></blockquote></details> <a class="!text-gray-400 !no-underline text-sm flex items-center not-prose mt-4" href="https://github.com/huggingface/hub-docs/blob/main/docs/sagemaker/source/get-started/dlcs.md" target="_blank"><svg class="mr-1" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M31,16l-7,7l-1.41-1.41L28.17,16l-5.58-5.59L24,9l7,7z"></path><path d="M1,16l7-7l1.41,1.41L3.83,16l5.58,5.59L8,23l-7-7z"></path><path d="M12.419,25.484L17.639,6.552l1.932,0.518L14.351,26.002z"></path></svg><!----> <span><span class="underline">Update</span> on GitHub</span></a><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]--> | |
| <script> | |
| { | |
| __sveltekit_18n89o7 = { | |
| base: "/docs/sagemaker/pr_2709/en", | |
| assets: "/docs/sagemaker/pr_2709/en" | |
| }; | |
| const element = document.currentScript.parentElement; | |
| Promise.all([ | |
| import("/docs/sagemaker/pr_2709/en/_app/immutable/entry/start.D0qXhpFD.js"), | |
| import("/docs/sagemaker/pr_2709/en/_app/immutable/entry/app.BJw2fAxO.js") | |
| ]).then(([kit, app]) => { | |
| kit.start(app, element, { | |
| node_ids: [0, 7], | |
| data: [null,null], | |
| form: null, | |
| error: null | |
| }); | |
| }); | |
| } | |
| </script> | |
Xet Storage Details
- Size:
- 39.1 kB
- Xet hash:
- edeedd4f99ea1e78554a2b3dbd085b3f84022aa1acd45a31296f7c6a61440f76
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.