Buckets:
| import{s as Ct,n as wt,o as Mt}from"../chunks/scheduler.56725da7.js";import{S as Ht,i as Pt,e as s,s as l,c as z,h as St,a as r,d as n,b as a,f as _t,g as I,j as o,k as kt,l as zt,m as i,n as A,t as B,o as j,p as E}from"../chunks/index.18a26576.js";import{C as It}from"../chunks/CopyLLMTxtMenu.a1f2bcd7.js";import{H as rt}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.9f98faf7.js";function At(ot){let m,q,P,R,p,W,f,F,u,mt="How fast is Llama-3.1-8b on Inferentia2? Let’s figure out!",G,h,pt="For this benchmark we will use the following configurations:",N,d,ft="<thead><tr><th>Model type</th> <th>batch_size</th> <th>sequence_length</th></tr></thead> <tbody><tr><td>Llama3.1 8b BS1</td> <td>1</td> <td>4096</td></tr> <tr><td>Llama3.1 8b BS4</td> <td>4</td> <td>4096</td></tr> <tr><td>Llama3.1 8b BS8</td> <td>8</td> <td>4096</td></tr> <tr><td>Llama3.1 8b BS16</td> <td>16</td> <td>4096</td></tr> <tr><td>Llama3.1 8b BS32</td> <td>32</td> <td>4096</td></tr> <tr><td>Llama3.1 8b BS48</td> <td>48</td> <td>4096</td></tr></tbody>",U,c,ut="<em>Note: all models are compiled to use 4 devices corresponding to 8 cores on the <code>inf2.48xlarge</code> instance.</em>",Q,g,ht='<em>Note: please refer to the <a href="https://aws.amazon.com/ec2/instance-types/inf2/" rel="nofollow">inferentia2 product page</a> for details on the available instances.</em>',D,v,J,$,dt=`The time to first token is the time required to process the input tokens and generate the first output token. | |
| It is a very important metric, as it corresponds to the latency directly perceived by the user when streaming generated tokens.`,K,x,ct="We test the time to first token for increasing context sizes, from a typical Q/A usage, to heavy Retrieval Augmented Generation (RAG) use-cases.",O,b,gt="Time to first token is expressed in <strong>seconds</strong>.",V,T,vt='<img src="https://raw.githubusercontent.com/huggingface/optimum-neuron/main/docs/assets/benchmarks/inferentia-llama3.1-8b/ttft.png" alt="Llama3.1 8b inferentia2 TTFT" title="Time to first token"/>',X,L,Y,y,$t="The inter-token latency corresponds to the average time elapsed between two generated tokens.",Z,_,xt="It is expressed in <strong>milliseconds</strong>.",tt,k,bt='<img src="https://raw.githubusercontent.com/huggingface/optimum-neuron/main/docs/assets/benchmarks/inferentia-llama3.1-8b/latency.png" alt="Llama3.1 8b inferentia2 inter-token latency" title="Inter-token latency"/>',et,C,nt,w,Tt=`Unlike some other benchmarks, we evaluate the throughput using generated tokens only, by dividing their number | |
| by the end-to-end latency.`,it,M,Lt="Throughput is expressed in <strong>tokens/second</strong>.",lt,H,yt='<img src="https://raw.githubusercontent.com/huggingface/optimum-neuron/main/docs/assets/benchmarks/inferentia-llama3.1-8b/throughput.png" alt="Llama3.1 8b inferentia2 throughput" title="Throughput"/>',at,S,st;return p=new It({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),f=new rt({props:{title:"Llama-3.1-8b performance on AWS Inferentia2 (Latency & Throughput)",local:"llama-31-8b-performance-on-aws-inferentia2-latency--throughput",headingTag:"h1"}}),v=new rt({props:{title:"Time to first token",local:"time-to-first-token",headingTag:"h2"}}),L=new rt({props:{title:"Inter-token Latency",local:"inter-token-latency",headingTag:"h2"}}),C=new rt({props:{title:"Throughput",local:"throughput",headingTag:"h3"}}),{c(){m=s("meta"),q=l(),P=s("p"),R=l(),z(p.$$.fragment),W=l(),z(f.$$.fragment),F=l(),u=s("p"),u.textContent=mt,G=l(),h=s("p"),h.textContent=pt,N=l(),d=s("table"),d.innerHTML=ft,U=l(),c=s("p"),c.innerHTML=ut,Q=l(),g=s("p"),g.innerHTML=ht,D=l(),z(v.$$.fragment),J=l(),$=s("p"),$.textContent=dt,K=l(),x=s("p"),x.textContent=ct,O=l(),b=s("p"),b.innerHTML=gt,V=l(),T=s("p"),T.innerHTML=vt,X=l(),z(L.$$.fragment),Y=l(),y=s("p"),y.textContent=$t,Z=l(),_=s("p"),_.innerHTML=xt,tt=l(),k=s("p"),k.innerHTML=bt,et=l(),z(C.$$.fragment),nt=l(),w=s("p"),w.textContent=Tt,it=l(),M=s("p"),M.innerHTML=Lt,lt=l(),H=s("p"),H.innerHTML=yt,at=l(),S=s("p"),this.h()},l(t){const e=St("svelte-u9bgzb",document.head);m=r(e,"META",{name:!0,content:!0}),e.forEach(n),q=a(t),P=r(t,"P",{}),_t(P).forEach(n),R=a(t),I(p.$$.fragment,t),W=a(t),I(f.$$.fragment,t),F=a(t),u=r(t,"P",{"data-svelte-h":!0}),o(u)!=="svelte-1r6vka8"&&(u.textContent=mt),G=a(t),h=r(t,"P",{"data-svelte-h":!0}),o(h)!=="svelte-18zzjby"&&(h.textContent=pt),N=a(t),d=r(t,"TABLE",{"data-svelte-h":!0}),o(d)!=="svelte-1i0agy3"&&(d.innerHTML=ft),U=a(t),c=r(t,"P",{"data-svelte-h":!0}),o(c)!=="svelte-tktwzm"&&(c.innerHTML=ut),Q=a(t),g=r(t,"P",{"data-svelte-h":!0}),o(g)!=="svelte-1zgafe"&&(g.innerHTML=ht),D=a(t),I(v.$$.fragment,t),J=a(t),$=r(t,"P",{"data-svelte-h":!0}),o($)!=="svelte-16f7nw9"&&($.textContent=dt),K=a(t),x=r(t,"P",{"data-svelte-h":!0}),o(x)!=="svelte-e39hxg"&&(x.textContent=ct),O=a(t),b=r(t,"P",{"data-svelte-h":!0}),o(b)!=="svelte-1et04dj"&&(b.innerHTML=gt),V=a(t),T=r(t,"P",{"data-svelte-h":!0}),o(T)!=="svelte-a4whjy"&&(T.innerHTML=vt),X=a(t),I(L.$$.fragment,t),Y=a(t),y=r(t,"P",{"data-svelte-h":!0}),o(y)!=="svelte-13h9pgw"&&(y.textContent=$t),Z=a(t),_=r(t,"P",{"data-svelte-h":!0}),o(_)!=="svelte-1ahl198"&&(_.innerHTML=xt),tt=a(t),k=r(t,"P",{"data-svelte-h":!0}),o(k)!=="svelte-ffe8jz"&&(k.innerHTML=bt),et=a(t),I(C.$$.fragment,t),nt=a(t),w=r(t,"P",{"data-svelte-h":!0}),o(w)!=="svelte-1a309vq"&&(w.textContent=Tt),it=a(t),M=r(t,"P",{"data-svelte-h":!0}),o(M)!=="svelte-14m9i5e"&&(M.innerHTML=Lt),lt=a(t),H=r(t,"P",{"data-svelte-h":!0}),o(H)!=="svelte-140gjuh"&&(H.innerHTML=yt),at=a(t),S=r(t,"P",{}),_t(S).forEach(n),this.h()},h(){kt(m,"name","hf:doc:metadata"),kt(m,"content",Bt)},m(t,e){zt(document.head,m),i(t,q,e),i(t,P,e),i(t,R,e),A(p,t,e),i(t,W,e),A(f,t,e),i(t,F,e),i(t,u,e),i(t,G,e),i(t,h,e),i(t,N,e),i(t,d,e),i(t,U,e),i(t,c,e),i(t,Q,e),i(t,g,e),i(t,D,e),A(v,t,e),i(t,J,e),i(t,$,e),i(t,K,e),i(t,x,e),i(t,O,e),i(t,b,e),i(t,V,e),i(t,T,e),i(t,X,e),A(L,t,e),i(t,Y,e),i(t,y,e),i(t,Z,e),i(t,_,e),i(t,tt,e),i(t,k,e),i(t,et,e),A(C,t,e),i(t,nt,e),i(t,w,e),i(t,it,e),i(t,M,e),i(t,lt,e),i(t,H,e),i(t,at,e),i(t,S,e),st=!0},p:wt,i(t){st||(B(p.$$.fragment,t),B(f.$$.fragment,t),B(v.$$.fragment,t),B(L.$$.fragment,t),B(C.$$.fragment,t),st=!0)},o(t){j(p.$$.fragment,t),j(f.$$.fragment,t),j(v.$$.fragment,t),j(L.$$.fragment,t),j(C.$$.fragment,t),st=!1},d(t){t&&(n(q),n(P),n(R),n(W),n(F),n(u),n(G),n(h),n(N),n(d),n(U),n(c),n(Q),n(g),n(D),n(J),n($),n(K),n(x),n(O),n(b),n(V),n(T),n(X),n(Y),n(y),n(Z),n(_),n(tt),n(k),n(et),n(nt),n(w),n(it),n(M),n(lt),n(H),n(at),n(S)),n(m),E(p,t),E(f,t),E(v,t),E(L,t),E(C,t)}}}const Bt='{"title":"Llama-3.1-8b performance on AWS Inferentia2 (Latency & Throughput)","local":"llama-31-8b-performance-on-aws-inferentia2-latency--throughput","sections":[{"title":"Time to first token","local":"time-to-first-token","sections":[],"depth":2},{"title":"Inter-token Latency","local":"inter-token-latency","sections":[{"title":"Throughput","local":"throughput","sections":[],"depth":3}],"depth":2}],"depth":1}';function jt(ot){return Mt(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class Ft extends Ht{constructor(m){super(),Pt(this,m,jt,At,Ct,{})}}export{Ft as component}; | |
Xet Storage Details
- Size:
- 7.21 kB
- Xet hash:
- 204eb43c19a3daec392e1522a1783a5183c56d230fe9de57fc22975b8db737c0
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.