Buckets:
| import{s as kt,n as wt,o as Mt}from"../chunks/scheduler.56725da7.js";import{S as Ht,i as Pt,e as s,s as l,c as S,h as zt,a as r,d as n,b as a,f as _t,g as I,j as o,k as Ct,l as St,m as i,n as A,t as q,o as E,p as B}from"../chunks/index.18a26576.js";import{C as It}from"../chunks/CopyLLMTxtMenu.c5feff19.js";import{H as rt}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.0f5f04c9.js";function At(ot){let m,j,P,R,p,W,f,F,u,mt="How fast is Llama-3.3-70b on Inferentia2? Let’s figure out!",G,h,pt="For this benchmark we will use the following configurations:",N,c,ft="<thead><tr><th>Model type</th> <th>batch_size</th> <th>sequence_length</th></tr></thead> <tbody><tr><td>Llama3.3 70b BS1</td> <td>1</td> <td>4096</td></tr> <tr><td>Llama3.3 70b BS4</td> <td>4</td> <td>4096</td></tr> <tr><td>Llama3.3 70b BS8</td> <td>8</td> <td>4096</td></tr></tbody>",U,d,ut="<em>Note: all models are compiled to use 12 devices corresponding to 24 cores on the <code>inf2.48xlarge</code> instance.</em>",Q,g,ht='<em>Note: please refer to the <a href="https://aws.amazon.com/ec2/instance-types/inf2/" rel="nofollow">inferentia2 product page</a> for details on the available instances.</em>',D,v,J,x,ct=`The time to first token is the time required to process the input tokens and generate the first output token. | |
| It is a very important metric, as it corresponds to the latency directly perceived by the user when streaming generated tokens.`,K,$,dt="We test the time to first token for increasing context sizes, from a typical Q/A usage, to heavy Retrieval Augmented Generation (RAG) use-cases.",O,T,gt="Time to first token is expressed in <strong>seconds</strong>.",V,b,vt='<img src="https://raw.githubusercontent.com/huggingface/optimum-neuron/main/docs/assets/benchmarks/inferentia-llama3.3-70b/ttft.png" alt="Llama3.3 70b inferentia2 TTFT" title="Time to first token"/>',X,L,Y,y,xt="The inter-token latency corresponds to the average time elapsed between two generated tokens.",Z,_,$t="It is expressed in <strong>milliseconds</strong>.",tt,C,Tt='<img src="https://raw.githubusercontent.com/huggingface/optimum-neuron/main/docs/assets/benchmarks/inferentia-llama3.3-70b/latency.png" alt="Llama3.3 70b inferentia2 inter-token latency" title="Inter-token latency"/>',et,k,nt,w,bt=`Unlike some other benchmarks, we evaluate the throughput using generated tokens only, by dividing their number | |
| by the end-to-end latency.`,it,M,Lt="Throughput is expressed in <strong>tokens/second</strong>.",lt,H,yt='<img src="https://raw.githubusercontent.com/huggingface/optimum-neuron/main/docs/assets/benchmarks/inferentia-llama3.3-70b/throughput.png" alt="Llama3.3 70b inferentia2 throughput" title="Throughput"/>',at,z,st;return p=new It({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),f=new rt({props:{title:"Llama-3.3-70b performance on AWS Inferentia2 (Latency & Throughput)",local:"llama-33-70b-performance-on-aws-inferentia2-latency--throughput",headingTag:"h1"}}),v=new rt({props:{title:"Time to first token",local:"time-to-first-token",headingTag:"h2"}}),L=new rt({props:{title:"Inter-token Latency",local:"inter-token-latency",headingTag:"h2"}}),k=new rt({props:{title:"Throughput",local:"throughput",headingTag:"h3"}}),{c(){m=s("meta"),j=l(),P=s("p"),R=l(),S(p.$$.fragment),W=l(),S(f.$$.fragment),F=l(),u=s("p"),u.textContent=mt,G=l(),h=s("p"),h.textContent=pt,N=l(),c=s("table"),c.innerHTML=ft,U=l(),d=s("p"),d.innerHTML=ut,Q=l(),g=s("p"),g.innerHTML=ht,D=l(),S(v.$$.fragment),J=l(),x=s("p"),x.textContent=ct,K=l(),$=s("p"),$.textContent=dt,O=l(),T=s("p"),T.innerHTML=gt,V=l(),b=s("p"),b.innerHTML=vt,X=l(),S(L.$$.fragment),Y=l(),y=s("p"),y.textContent=xt,Z=l(),_=s("p"),_.innerHTML=$t,tt=l(),C=s("p"),C.innerHTML=Tt,et=l(),S(k.$$.fragment),nt=l(),w=s("p"),w.textContent=bt,it=l(),M=s("p"),M.innerHTML=Lt,lt=l(),H=s("p"),H.innerHTML=yt,at=l(),z=s("p"),this.h()},l(t){const e=zt("svelte-u9bgzb",document.head);m=r(e,"META",{name:!0,content:!0}),e.forEach(n),j=a(t),P=r(t,"P",{}),_t(P).forEach(n),R=a(t),I(p.$$.fragment,t),W=a(t),I(f.$$.fragment,t),F=a(t),u=r(t,"P",{"data-svelte-h":!0}),o(u)!=="svelte-h7x8hp"&&(u.textContent=mt),G=a(t),h=r(t,"P",{"data-svelte-h":!0}),o(h)!=="svelte-18zzjby"&&(h.textContent=pt),N=a(t),c=r(t,"TABLE",{"data-svelte-h":!0}),o(c)!=="svelte-q4dz7x"&&(c.innerHTML=ft),U=a(t),d=r(t,"P",{"data-svelte-h":!0}),o(d)!=="svelte-1c2tu05"&&(d.innerHTML=ut),Q=a(t),g=r(t,"P",{"data-svelte-h":!0}),o(g)!=="svelte-1zgafe"&&(g.innerHTML=ht),D=a(t),I(v.$$.fragment,t),J=a(t),x=r(t,"P",{"data-svelte-h":!0}),o(x)!=="svelte-16f7nw9"&&(x.textContent=ct),K=a(t),$=r(t,"P",{"data-svelte-h":!0}),o($)!=="svelte-e39hxg"&&($.textContent=dt),O=a(t),T=r(t,"P",{"data-svelte-h":!0}),o(T)!=="svelte-1et04dj"&&(T.innerHTML=gt),V=a(t),b=r(t,"P",{"data-svelte-h":!0}),o(b)!=="svelte-cna4o8"&&(b.innerHTML=vt),X=a(t),I(L.$$.fragment,t),Y=a(t),y=r(t,"P",{"data-svelte-h":!0}),o(y)!=="svelte-13h9pgw"&&(y.textContent=xt),Z=a(t),_=r(t,"P",{"data-svelte-h":!0}),o(_)!=="svelte-1ahl198"&&(_.innerHTML=$t),tt=a(t),C=r(t,"P",{"data-svelte-h":!0}),o(C)!=="svelte-vtizkf"&&(C.innerHTML=Tt),et=a(t),I(k.$$.fragment,t),nt=a(t),w=r(t,"P",{"data-svelte-h":!0}),o(w)!=="svelte-1a309vq"&&(w.textContent=bt),it=a(t),M=r(t,"P",{"data-svelte-h":!0}),o(M)!=="svelte-14m9i5e"&&(M.innerHTML=Lt),lt=a(t),H=r(t,"P",{"data-svelte-h":!0}),o(H)!=="svelte-1r7vce3"&&(H.innerHTML=yt),at=a(t),z=r(t,"P",{}),_t(z).forEach(n),this.h()},h(){Ct(m,"name","hf:doc:metadata"),Ct(m,"content",qt)},m(t,e){St(document.head,m),i(t,j,e),i(t,P,e),i(t,R,e),A(p,t,e),i(t,W,e),A(f,t,e),i(t,F,e),i(t,u,e),i(t,G,e),i(t,h,e),i(t,N,e),i(t,c,e),i(t,U,e),i(t,d,e),i(t,Q,e),i(t,g,e),i(t,D,e),A(v,t,e),i(t,J,e),i(t,x,e),i(t,K,e),i(t,$,e),i(t,O,e),i(t,T,e),i(t,V,e),i(t,b,e),i(t,X,e),A(L,t,e),i(t,Y,e),i(t,y,e),i(t,Z,e),i(t,_,e),i(t,tt,e),i(t,C,e),i(t,et,e),A(k,t,e),i(t,nt,e),i(t,w,e),i(t,it,e),i(t,M,e),i(t,lt,e),i(t,H,e),i(t,at,e),i(t,z,e),st=!0},p:wt,i(t){st||(q(p.$$.fragment,t),q(f.$$.fragment,t),q(v.$$.fragment,t),q(L.$$.fragment,t),q(k.$$.fragment,t),st=!0)},o(t){E(p.$$.fragment,t),E(f.$$.fragment,t),E(v.$$.fragment,t),E(L.$$.fragment,t),E(k.$$.fragment,t),st=!1},d(t){t&&(n(j),n(P),n(R),n(W),n(F),n(u),n(G),n(h),n(N),n(c),n(U),n(d),n(Q),n(g),n(D),n(J),n(x),n(K),n($),n(O),n(T),n(V),n(b),n(X),n(Y),n(y),n(Z),n(_),n(tt),n(C),n(et),n(nt),n(w),n(it),n(M),n(lt),n(H),n(at),n(z)),n(m),B(p,t),B(f,t),B(v,t),B(L,t),B(k,t)}}}const qt='{"title":"Llama-3.3-70b performance on AWS Inferentia2 (Latency & Throughput)","local":"llama-33-70b-performance-on-aws-inferentia2-latency--throughput","sections":[{"title":"Time to first token","local":"time-to-first-token","sections":[],"depth":2},{"title":"Inter-token Latency","local":"inter-token-latency","sections":[{"title":"Throughput","local":"throughput","sections":[],"depth":3}],"depth":2}],"depth":1}';function Et(ot){return Mt(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class Ft extends Ht{constructor(m){super(),Pt(this,m,Et,At,kt,{})}}export{Ft as component}; | |
Xet Storage Details
- Size:
- 7.04 kB
- Xet hash:
- fa0369a3e705ae6532300c29312b259e59551358869c5371aca164512a02c3a6
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.