Buckets:

rtrm's picture
download
raw
2.94 kB
import{s as K,n as N,o as O}from"../chunks/scheduler.893fe8c9.js";import{S as R,i as W,e as h,s,c as T,h as D,a as $,d as n,b as o,f as G,g as E,j as S,k as F,l as I,m as a,n as L,t as M,o as B,p as H}from"../chunks/index.b1df2166.js";import{C as J,H as Q,E as V}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.0a26397a.js";import{C as X}from"../chunks/CourseFloatingBanner.c1c08878.js";function Y(U){let i,g,d,x,l,_,r,k,m,w,p,j="Great job finishing this chapter!",v,f,q="After this deep dive into tokenizers, you should:",y,u,A="<li>Be able to train a new tokenizer using an old one as a template</li> <li>Understand how to use offsets to map tokens’ positions to their original span of text</li> <li>Know the differences between BPE, WordPiece, and Unigram</li> <li>Be able to mix and match the blocks provided by the 🤗 Tokenizers library to build your own tokenizer</li> <li>Be able to use that tokenizer inside the 🤗 Transformers library</li>",z,c,C,b,P;return l=new J({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),r=new Q({props:{title:"Tokenizers, check!",local:"tokenizers-check",headingTag:"h1"}}),m=new X({props:{chapter:6,classNames:"absolute z-10 right-0 top-0"}}),c=new V({props:{source:"https://github.com/huggingface/course/blob/main/chapters/en/chapter6/9.mdx"}}),{c(){i=h("meta"),g=s(),d=h("p"),x=s(),T(l.$$.fragment),_=s(),T(r.$$.fragment),k=s(),T(m.$$.fragment),w=s(),p=h("p"),p.textContent=j,v=s(),f=h("p"),f.textContent=q,y=s(),u=h("ul"),u.innerHTML=A,z=s(),T(c.$$.fragment),C=s(),b=h("p"),this.h()},l(e){const t=D("svelte-u9bgzb",document.head);i=$(t,"META",{name:!0,content:!0}),t.forEach(n),g=o(e),d=$(e,"P",{}),G(d).forEach(n),x=o(e),E(l.$$.fragment,e),_=o(e),E(r.$$.fragment,e),k=o(e),E(m.$$.fragment,e),w=o(e),p=$(e,"P",{"data-svelte-h":!0}),S(p)!=="svelte-qrdqcf"&&(p.textContent=j),v=o(e),f=$(e,"P",{"data-svelte-h":!0}),S(f)!=="svelte-ziaxv6"&&(f.textContent=q),y=o(e),u=$(e,"UL",{"data-svelte-h":!0}),S(u)!=="svelte-jl1wny"&&(u.innerHTML=A),z=o(e),E(c.$$.fragment,e),C=o(e),b=$(e,"P",{}),G(b).forEach(n),this.h()},h(){F(i,"name","hf:doc:metadata"),F(i,"content",Z)},m(e,t){I(document.head,i),a(e,g,t),a(e,d,t),a(e,x,t),L(l,e,t),a(e,_,t),L(r,e,t),a(e,k,t),L(m,e,t),a(e,w,t),a(e,p,t),a(e,v,t),a(e,f,t),a(e,y,t),a(e,u,t),a(e,z,t),L(c,e,t),a(e,C,t),a(e,b,t),P=!0},p:N,i(e){P||(M(l.$$.fragment,e),M(r.$$.fragment,e),M(m.$$.fragment,e),M(c.$$.fragment,e),P=!0)},o(e){B(l.$$.fragment,e),B(r.$$.fragment,e),B(m.$$.fragment,e),B(c.$$.fragment,e),P=!1},d(e){e&&(n(g),n(d),n(x),n(_),n(k),n(w),n(p),n(v),n(f),n(y),n(u),n(z),n(C),n(b)),n(i),H(l,e),H(r,e),H(m,e),H(c,e)}}}const Z='{"title":"Tokenizers, check!","local":"tokenizers-check","sections":[],"depth":1}';function ee(U){return O(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class se extends R{constructor(i){super(),W(this,i,ee,Y,K,{})}}export{se as component};

Xet Storage Details

Size:
2.94 kB
·
Xet hash:
b3cb1503f4e1810bc9fc15e39896f5b3d91926b7c9028ac24e2b71e2e2d5e1ab

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.