Buckets:
| import"../chunks/DsnmJJEf.js";import{i as g,h as m,C as f,H as t,E as y,s as w}from"../chunks/DQX3K91K.js";import{p as b,o as k,s as e,f as v,a as l,b as I,c as d,n as _}from"../chunks/DkE5gPSR.js";const q='{"title":"Quick Start","local":"quick-start","sections":[{"title":"Create your endpoint","local":"create-your-endpoint","sections":[],"depth":2},{"title":"Test your Inference Endpoint","local":"test-your-inference-endpoint","sections":[],"depth":2}],"depth":1}';var E=d('<meta name="hf:doc:metadata"/>'),S=d(`<p></p> <!> <!> <p>In this guide you’ll deploy a production ready AI model using Inference Endpoints in only a few minutes. | |
| Make sure you’ve been able to log in the <a href="https://endpoints.huggingface.co" rel="nofollow">Inference Endpoints UI</a> with your Hugging Face account, and that you have a payment | |
| method set up. If not, it’s a quick add of valid payment method and credits in your <a href="https://huggingface.co/settings/billing" rel="nofollow">billing settings</a> - preferably with the automatic recharge enabled to ensure uninterrupted usage and the best possible experience.</p> <!> <p>Start by navigating to the Inference Endpoints UI, and once you have logged in, click the <strong>Catalog</strong> button.</p> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/quick_start/1-new-button.png" alt="new-button"/></p> <p>From there you’ll be directed to the catalog. The Model Catalog consists of popular models which have tuned configurations to work in one-click | |
| deploys. You can filter by name, task, hardware price, and much more.</p> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/quick_start/2-catalog.png" alt="catalog"/></p> <p>In this example let’s deploy the <a href="https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct" rel="nofollow">meta-llama/Llama-3.2-3B-Instruct</a> model. You can find | |
| it by searching for <code>llama-3.2-3b</code> in the search field and deploy it by clicking the card.</p> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/quick_start/3-llama.png" alt="llama"/></p> <p>Next we’ll choose which hardware and deployment settings we’ll go for. Since this is a catalog model, all of the pre-selected options are very good | |
| defaults. So in this case we don’t need to change anything. In case you want a deeper dive on what the different settings mean you can check out | |
| the <a href="./guides/configuration">configuration guide</a>.</p> <p>For this model the Nvidia L4 is the recommended choice. It will be perfect for our testing. Performant but still reasonably priced. Also note that by | |
| default the endpoint will scale down to zero, meaning it will become idle after 1h of inactivity.</p> <p>Now all you need to do is click click “Create Endpoint” 🚀</p> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/quick_start/4-config.png" alt="config"/></p> <p>Now our Inference Endpoint is initializing, which usually takes about 3-5 minutes. If you want to can allow browser notifications which will give you a | |
| ping once the endpoint reaches a running state.</p> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/quick_start/5-init.png" alt="init"/></p> <!> <p>And then once everything is up and running you’ll be able to see the <strong>Endpoint URL</strong> on the Overview tab — this is what you use to call your endpoint and send requests to it.</p> <p><img src="https://raw.githubusercontent.com/huggingface/hf-endpoints-documentation/main/assets/quick_start/6-done.png" alt="done"/></p> <p>Head over to the <strong>Playground</strong> tab for a quick visual way of testing that the model works. From the “API” section of the playground you can also copy + paste a code snippet for calling the model. By clicking <strong>API Token</strong> you can paste in an access token to be able to call the model. By default, all Inference Endpoints are created as private which require authentication and | |
| all data is encryped in transit using TLS/SSL.</p> <p>Congratulations, you just deployed a production ready AI model in Inference Endpoints 🔥</p> <p>Once you’re happy with the testing you can pause the Inference Endpoint, delete it. Or if you let it be, it will scale to zero after 1 hour.</p> <!> <p></p>`,1);function C(p,u){b(u,!1),k(()=>{new URLSearchParams(window.location.search).get("fw")}),g();var n=S();m("912imz",r=>{var c=E();w(c,"content",q),l(r,c)});var a=e(v(n),2);f(a,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var o=e(a,2);t(o,{title:"Quick Start",local:"quick-start",headingTag:"h1"});var i=e(o,4);t(i,{title:"Create your endpoint",local:"create-your-endpoint",headingTag:"h2"});var s=e(i,26);t(s,{title:"Test your Inference Endpoint",local:"test-your-inference-endpoint",headingTag:"h2"});var h=e(s,12);y(h,{source:"https://github.com/huggingface/hf-endpoints-documentation/blob/main/docs/source/quick_start.md"}),_(2),l(p,n),I()}export{C as component}; | |
Xet Storage Details
- Size:
- 5.17 kB
- Xet hash:
- c8791a96603264c716d662158afdbd9fa2af69e96a52d14aa5548c05f5c6f6a5
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.