Buckets:
| import{s as Z,n as tt,o as et}from"../chunks/scheduler.893fe8c9.js";import{S as nt,i as it,e as a,s as r,c as L,h as rt,a as l,d as n,b as o,f as W,g as v,j as b,k as X,l as ot,m as i,n as P,t as _,o as x,p as w}from"../chunks/index.b1df2166.js";import{C as st,H as Y,E as at}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.e71060cd.js";import{C as lt}from"../chunks/CourseFloatingBanner.c1c08878.js";function gt(V){let s,C,y,M,g,H,u,S,m,A,p,j='<a href="/course/chapter3">Chapter 3</a> မှာ၊ သတ်မှတ်ထားတဲ့ task တစ်ခုပေါ်မှာ model တစ်ခုကို fine-tune လုပ်နည်းကို ကျွန်တော်တို့ ကြည့်ခဲ့ပါတယ်။ အဲဒီလိုလုပ်တဲ့အခါ၊ model ကို pretrained လုပ်ခဲ့တဲ့ tokenizer တူတူကို ကျွန်တော်တို့ အသုံးပြုပါတယ်။ — ဒါပေမယ့် model တစ်ခုကို အစကနေ train လုပ်ချင်တဲ့အခါ ဘာလုပ်ရမလဲ။ ဒီလိုအခြေအနေတွေမှာ၊ အခြား domain ဒါမှမဟုတ် language က corpus တစ်ခုပေါ်မှာ pretrained လုပ်ထားတဲ့ tokenizer ကို အသုံးပြုတာက ပုံမှန်အားဖြင့် suboptimal ဖြစ်ပါတယ်။ ဥပမာ၊ English corpus တစ်ခုပေါ်မှာ train ထားတဲ့ tokenizer က Japanese texts corpus ပေါ်မှာ ကောင်းကောင်းအလုပ်လုပ်မှာ မဟုတ်ပါဘူး။ ဘာလို့လဲဆိုတော့ spaces နဲ့ punctuation အသုံးပြုမှုက ဘာသာစကားနှစ်ခုမှာ အလွန်ကွာခြားလို့ပါပဲ။',E,f,B='ဒီအခန်းမှာ၊ စာသား corpus တစ်ခုပေါ်မှာ tokenizer အသစ်တစ်ခုကို ဘယ်လို train လုပ်ရမယ်ဆိုတာ သင်ယူရမှာဖြစ်ပြီး၊ အဲဒါကို language model တစ်ခုကို pretrain လုပ်ဖို့ အသုံးပြုနိုင်ပါလိမ့်မယ်။ ဒါတွေအားလုံးကို <a href="https://github.com/huggingface/tokenizers" rel="nofollow">🤗 Tokenizers</a> library ရဲ့ အကူအညီနဲ့ လုပ်ဆောင်သွားမှာပါ။ အဲဒီ library က <a href="https://github.com/huggingface/transformers" rel="nofollow">🤗 Transformers</a> library မှာ “fast” tokenizers တွေကို ပံ့ပိုးပေးပါတယ်။ ဒီ library က ပံ့ပိုးပေးတဲ့ features တွေကို အနီးကပ်ကြည့်ရှုပြီး fast tokenizers တွေက “slow” versions တွေနဲ့ ဘယ်လိုကွာခြားလဲဆိုတာ လေ့လာသွားမှာပါ။',F,c,D="ကျွန်တော်တို့ ဖော်ပြမယ့် ခေါင်းစဉ်တွေကတော့-",I,h,J="<li>ပေးထားတဲ့ checkpoint တစ်ခုက အသုံးပြုတဲ့ tokenizer နဲ့ ဆင်တူတဲ့ tokenizer အသစ်တစ်ခုကို texts corpus အသစ်တစ်ခုပေါ်မှာ ဘယ်လို train လုပ်ရမလဲ။</li> <li>fast tokenizers တွေရဲ့ သီးခြား features တွေ။</li> <li>ဒီနေ့ခေတ် NLP မှာ အသုံးပြုနေတဲ့ subword tokenization algorithms သုံးခုကြားက ကွာခြားချက်တွေ။</li> <li>🤗 Tokenizers library နဲ့ tokenizer တစ်ခုကို အစကနေ ဘယ်လိုတည်ဆောက်ပြီး data အချို့ပေါ်မှာ train လုပ်ရမလဲ။</li>",N,$,K='ဒီအခန်းမှာ မိတ်ဆက်ပေးမယ့် နည်းလမ်းတွေက <a href="/course/chapter7/6">Chapter 7</a> မှာ Python source code အတွက် language model တစ်ခု ဖန်တီးတာကို ကြည့်ရှုမယ့် အပိုင်းအတွက် သင့်ကို ပြင်ဆင်ပေးပါလိမ့်မယ်။ ပထမဆုံး tokenizer ကို “train” လုပ်တယ်ဆိုတာ ဘာကိုဆိုလိုသလဲဆိုတာ ကြည့်ခြင်းဖြင့် စတင်ကြရအောင်။',q,d,G,k,Q="<li><strong>Fine-tune</strong>: ကြိုတင်လေ့ကျင့်ထားပြီးသား (pre-trained) မော်ဒယ်တစ်ခုကို သီးခြားလုပ်ငန်းတစ်ခု (specific task) အတွက် အနည်းငယ်သော ဒေတာနဲ့ ထပ်မံလေ့ကျင့်ပေးခြင်းကို ဆိုလိုပါတယ်။</li> <li><strong>Model</strong>: Artificial Intelligence (AI) နယ်ပယ်တွင် အချက်အလက်များကို လေ့လာပြီး ခန့်မှန်းချက်များ ပြုလုပ်ရန် ဒီဇိုင်းထုတ်ထားသော သင်္ချာဆိုင်ရာဖွဲ့စည်းပုံများ။</li> <li><strong>Tokenizer</strong>: စာသား (သို့မဟုတ် အခြားဒေတာ) ကို AI မော်ဒယ်များ စီမံဆောင်ရွက်နိုင်ရန် tokens တွေအဖြစ် ပိုင်းခြားပေးသည့် ကိရိယာ သို့မဟုတ် လုပ်ငန်းစဉ်။</li> <li><strong>Pretrained</strong>: Model တစ်ခုကို အကြီးစားဒေတာများဖြင့် အစောပိုင်းကတည်းက လေ့ကျင့်ထားခြင်း။</li> <li><strong>From Scratch</strong>: Model (သို့မဟုတ် tokenizer) တစ်ခုကို မည်သည့် အစောပိုင်းလေ့ကျင့်မှုမျှ မရှိဘဲ လုံးဝအသစ်ကနေ စတင်တည်ဆောက်ခြင်းနှင့် လေ့ကျင့်ခြင်း။</li> <li><strong>Corpus</strong>: စာသား (သို့မဟုတ် အခြားဒေတာ) အစုအဝေးကြီးတစ်ခု။</li> <li><strong>Domain</strong>: သီးခြားနယ်ပယ် (ဥပမာ- ဆေးပညာ domain, ဘဏ္ဍာရေး domain)။</li> <li><strong>Language</strong>: ဘာသာစကား။</li> <li><strong>Suboptimal</strong>: အကောင်းဆုံး မဟုတ်ဘဲ စွမ်းဆောင်ရည် နည်းပါးခြင်း။</li> <li><strong>Punctuation</strong>: စာသားများတွင် အသုံးပြုသော သတ်ပုံအမှတ်အသားများ (ဥပမာ- comma, period, question mark)။</li> <li><strong>Language Model</strong>: လူသားဘာသာစကား၏ ဖြန့်ဝေမှုကို နားလည်ရန် လေ့ကျင့်ထားသော AI မော်ဒယ်တစ်ခု။ ၎င်းသည် စာသားထုတ်လုပ်ခြင်း၊ ဘာသာပြန်ခြင်း စသည့်လုပ်ငန်းများတွင် အသုံးပြုနိုင်သည်။</li> <li><strong>🤗 Tokenizers Library</strong>: Rust ဘာသာနဲ့ ရေးသားထားတဲ့ Hugging Face library တစ်ခုဖြစ်ပြီး မြန်ဆန်ထိရောက်တဲ့ tokenization ကို လုပ်ဆောင်ပေးသည်။ 🤗 Transformers library အတွက် “fast” tokenizers တွေကို ပံ့ပိုးပေးသည်။</li> <li><strong>🤗 Transformers Library</strong>: Hugging Face က ထုတ်လုပ်ထားတဲ့ library တစ်ခုဖြစ်ပြီး Transformer မော်ဒယ်တွေကို အသုံးပြုပြီး Natural Language Processing (NLP), computer vision, audio processing စတဲ့ နယ်ပယ်တွေမှာ အဆင့်မြင့် AI မော်ဒယ်တွေကို တည်ဆောက်ပြီး အသုံးပြုနိုင်စေပါတယ်။</li> <li><strong>Fast Tokenizers</strong>: Rust ဘာသာစကားဖြင့် အကောင်အထည်ဖော်ထားသော tokenizers များဖြစ်ပြီး Python-based “slow” tokenizers များထက် အလွန်မြန်ဆန်သည်။</li> <li><strong>Slow Versions (Tokenizers)</strong>: Python ဘာသာစကားဖြင့် အကောင်အထည်ဖော်ထားသော tokenizers များ။</li> <li><strong>Checkpoint</strong>: မော်ဒယ်၏ weights များနှင့် အခြားဖွဲ့စည်းပုံများ (configuration) ကို သတ်မှတ်ထားသော အချိန်တစ်ခုတွင် သိမ်းဆည်းထားခြင်း။</li> <li><strong>Subword Tokenization Algorithms</strong>: စကားလုံးများကို သေးငယ်သော subword units (ဥပမာ- word pieces, byte-pair encodings) များအဖြစ် ပိုင်းခြားသော tokenization နည်းလမ်းများ။ ၎င်းသည် vocabulary အရွယ်အစားကို ထိန်းချုပ်ရန်နှင့် out-of-vocabulary (OOV) ပြဿနာများကို ဖြေရှင်းရန် ကူညီပေးသည်။</li> <li><strong>NLP (Natural Language Processing)</strong>: ကွန်ပျူတာတွေ လူသားဘာသာစကားကို နားလည်၊ အဓိပ္ပာယ်ဖော်ပြီး၊ ဖန်တီးနိုင်အောင် လုပ်ဆောင်ပေးတဲ့ Artificial Intelligence (AI) ရဲ့ နယ်ပယ်ခွဲတစ်ခုပါ။</li> <li><strong>From Scratch (Tokenizer)</strong>: မည်သည့် ကြိုတင်လေ့ကျင့်မှုမျှ မရှိဘဲ လုံးဝအသစ်ကနေ စတင်တည်ဆောက်ခြင်းနှင့် လေ့ကျင့်ခြင်း (tokenizer အတွက်)။</li> <li><strong>Python Source Code</strong>: Python programming language ဖြင့် ရေးသားထားသော code များ။</li>",O,z,R,T,U;return g=new st({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),u=new Y({props:{title:"နိဒါန်း",local:"introduction",headingTag:"h1"}}),m=new lt({props:{chapter:6,classNames:"absolute z-10 right-0 top-0"}}),d=new Y({props:{title:"ဝေါဟာရ ရှင်းလင်းချက် (Glossary)",local:"ဝဟရ-ရငလငခက-glossary",headingTag:"h2"}}),z=new at({props:{source:"https://github.com/huggingface/course/blob/main/chapters/my/chapter6/1.mdx"}}),{c(){s=a("meta"),C=r(),y=a("p"),M=r(),L(g.$$.fragment),H=r(),L(u.$$.fragment),S=r(),L(m.$$.fragment),A=r(),p=a("p"),p.innerHTML=j,E=r(),f=a("p"),f.innerHTML=B,F=r(),c=a("p"),c.textContent=D,I=r(),h=a("ul"),h.innerHTML=J,N=r(),$=a("p"),$.innerHTML=K,q=r(),L(d.$$.fragment),G=r(),k=a("ul"),k.innerHTML=Q,O=r(),L(z.$$.fragment),R=r(),T=a("p"),this.h()},l(t){const e=rt("svelte-u9bgzb",document.head);s=l(e,"META",{name:!0,content:!0}),e.forEach(n),C=o(t),y=l(t,"P",{}),W(y).forEach(n),M=o(t),v(g.$$.fragment,t),H=o(t),v(u.$$.fragment,t),S=o(t),v(m.$$.fragment,t),A=o(t),p=l(t,"P",{"data-svelte-h":!0}),b(p)!=="svelte-3n7sua"&&(p.innerHTML=j),E=o(t),f=l(t,"P",{"data-svelte-h":!0}),b(f)!=="svelte-g1c92z"&&(f.innerHTML=B),F=o(t),c=l(t,"P",{"data-svelte-h":!0}),b(c)!=="svelte-1kxz5lq"&&(c.textContent=D),I=o(t),h=l(t,"UL",{"data-svelte-h":!0}),b(h)!=="svelte-1k29h19"&&(h.innerHTML=J),N=o(t),$=l(t,"P",{"data-svelte-h":!0}),b($)!=="svelte-1tic8e8"&&($.innerHTML=K),q=o(t),v(d.$$.fragment,t),G=o(t),k=l(t,"UL",{"data-svelte-h":!0}),b(k)!=="svelte-vvnd3q"&&(k.innerHTML=Q),O=o(t),v(z.$$.fragment,t),R=o(t),T=l(t,"P",{}),W(T).forEach(n),this.h()},h(){X(s,"name","hf:doc:metadata"),X(s,"content",ut)},m(t,e){ot(document.head,s),i(t,C,e),i(t,y,e),i(t,M,e),P(g,t,e),i(t,H,e),P(u,t,e),i(t,S,e),P(m,t,e),i(t,A,e),i(t,p,e),i(t,E,e),i(t,f,e),i(t,F,e),i(t,c,e),i(t,I,e),i(t,h,e),i(t,N,e),i(t,$,e),i(t,q,e),P(d,t,e),i(t,G,e),i(t,k,e),i(t,O,e),P(z,t,e),i(t,R,e),i(t,T,e),U=!0},p:tt,i(t){U||(_(g.$$.fragment,t),_(u.$$.fragment,t),_(m.$$.fragment,t),_(d.$$.fragment,t),_(z.$$.fragment,t),U=!0)},o(t){x(g.$$.fragment,t),x(u.$$.fragment,t),x(m.$$.fragment,t),x(d.$$.fragment,t),x(z.$$.fragment,t),U=!1},d(t){t&&(n(C),n(y),n(M),n(H),n(S),n(A),n(p),n(E),n(f),n(F),n(c),n(I),n(h),n(N),n($),n(q),n(G),n(k),n(O),n(R),n(T)),n(s),w(g,t),w(u,t),w(m,t),w(d,t),w(z,t)}}}const ut='{"title":"နိဒါန်း","local":"introduction","sections":[{"title":"ဝေါဟာရ ရှင်းလင်းချက် (Glossary)","local":"ဝဟရ-ရငလငခက-glossary","sections":[],"depth":2}],"depth":1}';function mt(V){return et(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class $t extends nt{constructor(s){super(),it(this,s,mt,gt,Z,{})}}export{$t as component}; | |
Xet Storage Details
- Size:
- 14 kB
- Xet hash:
- 45bb793e73dd3a2f587b03ec4d9bc09cf1d6a511483fb147dfffff30d4d53404
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.