Buckets:
| import{s as Ws,o as js,n as ee}from"../chunks/scheduler.bdbef820.js";import{S as Us,i as Js,g as r,s as o,r as u,A as Ns,h as s,f as i,c as n,j as x,u as h,x as c,k as y,y as e,a as v,v as f,d as _,t as g,w as k}from"../chunks/index.33f81d56.js";import{T as wo}from"../chunks/Tip.34194030.js";import{D as w}from"../chunks/Docstring.64554317.js";import{C as Xo}from"../chunks/CodeBlock.362b34a4.js";import{E as Go}from"../chunks/ExampleCodeBlock.4f2252c6.js";import{H as Ho,E as Fs}from"../chunks/EditOnGithub.a9246e21.js";function Zs(B){let a,z="This method is deprecated, <code>__call__</code> should be used instead.";return{c(){a=r("p"),a.innerHTML=z},l(m){a=s(m,"P",{"data-svelte-h":!0}),c(a)!=="svelte-1phrc72"&&(a.innerHTML=z)},m(m,T){v(m,a,T)},p:ee,d(m){m&&i(a)}}}function Vs(B){let a,z="This method is deprecated, <code>__call__</code> should be used instead.";return{c(){a=r("p"),a.innerHTML=z},l(m){a=s(m,"P",{"data-svelte-h":!0}),c(a)!=="svelte-1phrc72"&&(a.innerHTML=z)},m(m,T){v(m,a,T)},p:ee,d(m){m&&i(a)}}}function Ds(B){let a,z="Passing <code>token=True</code> is required when you want to use a private model.";return{c(){a=r("p"),a.innerHTML=z},l(m){a=s(m,"P",{"data-svelte-h":!0}),c(a)!=="svelte-15auxyb"&&(a.innerHTML=z)},m(m,T){v(m,a,T)},p:ee,d(m){m&&i(a)}}}function Ss(B){let a,z="Examples:",m,T,$;return T=new Xo({props:{code:"JTIzJTIwV2UlMjBjYW4ndCUyMGluc3RhbnRpYXRlJTIwZGlyZWN0bHklMjB0aGUlMjBiYXNlJTIwY2xhc3MlMjAqUHJlVHJhaW5lZFRva2VuaXplckJhc2UqJTIwc28lMjBsZXQncyUyMHNob3clMjBvdXIlMjBleGFtcGxlcyUyMG9uJTIwYSUyMGRlcml2ZWQlMjBjbGFzcyUzQSUyMEJlcnRUb2tlbml6ZXIlMEElMjMlMjBEb3dubG9hZCUyMHZvY2FidWxhcnklMjBmcm9tJTIwaHVnZ2luZ2ZhY2UuY28lMjBhbmQlMjBjYWNoZS4lMEF0b2tlbml6ZXIlMjAlM0QlMjBCZXJ0VG9rZW5pemVyLmZyb21fcHJldHJhaW5lZCglMjJnb29nbGUtYmVydCUyRmJlcnQtYmFzZS11bmNhc2VkJTIyKSUwQSUwQSUyMyUyMERvd25sb2FkJTIwdm9jYWJ1bGFyeSUyMGZyb20lMjBodWdnaW5nZmFjZS5jbyUyMCh1c2VyLXVwbG9hZGVkKSUyMGFuZCUyMGNhY2hlLiUwQXRva2VuaXplciUyMCUzRCUyMEJlcnRUb2tlbml6ZXIuZnJvbV9wcmV0cmFpbmVkKCUyMmRibWR6JTJGYmVydC1iYXNlLWdlcm1hbi1jYXNlZCUyMiklMEElMEElMjMlMjBJZiUyMHZvY2FidWxhcnklMjBmaWxlcyUyMGFyZSUyMGluJTIwYSUyMGRpcmVjdG9yeSUyMChlLmcuJTIwdG9rZW5pemVyJTIwd2FzJTIwc2F2ZWQlMjB1c2luZyUyMCpzYXZlX3ByZXRyYWluZWQoJy4lMkZ0ZXN0JTJGc2F2ZWRfbW9kZWwlMkYnKSopJTBBdG9rZW5pemVyJTIwJTNEJTIwQmVydFRva2VuaXplci5mcm9tX3ByZXRyYWluZWQoJTIyLiUyRnRlc3QlMkZzYXZlZF9tb2RlbCUyRiUyMiklMEElMEElMjMlMjBJZiUyMHRoZSUyMHRva2VuaXplciUyMHVzZXMlMjBhJTIwc2luZ2xlJTIwdm9jYWJ1bGFyeSUyMGZpbGUlMkMlMjB5b3UlMjBjYW4lMjBwb2ludCUyMGRpcmVjdGx5JTIwdG8lMjB0aGlzJTIwZmlsZSUwQXRva2VuaXplciUyMCUzRCUyMEJlcnRUb2tlbml6ZXIuZnJvbV9wcmV0cmFpbmVkKCUyMi4lMkZ0ZXN0JTJGc2F2ZWRfbW9kZWwlMkZteV92b2NhYi50eHQlMjIpJTBBJTBBJTIzJTIwWW91JTIwY2FuJTIwbGluayUyMHRva2VucyUyMHRvJTIwc3BlY2lhbCUyMHZvY2FidWxhcnklMjB3aGVuJTIwaW5zdGFudGlhdGluZyUwQXRva2VuaXplciUyMCUzRCUyMEJlcnRUb2tlbml6ZXIuZnJvbV9wcmV0cmFpbmVkKCUyMmdvb2dsZS1iZXJ0JTJGYmVydC1iYXNlLXVuY2FzZWQlMjIlMkMlMjB1bmtfdG9rZW4lM0QlMjIlM0N1bmslM0UlMjIpJTBBJTIzJTIwWW91JTIwc2hvdWxkJTIwYmUlMjBzdXJlJTIwJyUzQ3VuayUzRSclMjBpcyUyMGluJTIwdGhlJTIwdm9jYWJ1bGFyeSUyMHdoZW4lMjBkb2luZyUyMHRoYXQuJTBBJTIzJTIwT3RoZXJ3aXNlJTIwdXNlJTIwdG9rZW5pemVyLmFkZF9zcGVjaWFsX3Rva2VucyglN0IndW5rX3Rva2VuJyUzQSUyMCclM0N1bmslM0UnJTdEKSUyMGluc3RlYWQpJTBBYXNzZXJ0JTIwdG9rZW5pemVyLnVua190b2tlbiUyMCUzRCUzRCUyMCUyMiUzQ3VuayUzRSUyMg==",highlighted:`<span class="hljs-comment"># We can't instantiate directly the base class *PreTrainedTokenizerBase* so let's show our examples on a derived class: BertTokenizer</span> | |
| <span class="hljs-comment"># Download vocabulary from huggingface.co and cache.</span> | |
| tokenizer = BertTokenizer.from_pretrained(<span class="hljs-string">"google-bert/bert-base-uncased"</span>) | |
| <span class="hljs-comment"># Download vocabulary from huggingface.co (user-uploaded) and cache.</span> | |
| tokenizer = BertTokenizer.from_pretrained(<span class="hljs-string">"dbmdz/bert-base-german-cased"</span>) | |
| <span class="hljs-comment"># If vocabulary files are in a directory (e.g. tokenizer was saved using *save_pretrained('./test/saved_model/')*)</span> | |
| tokenizer = BertTokenizer.from_pretrained(<span class="hljs-string">"./test/saved_model/"</span>) | |
| <span class="hljs-comment"># If the tokenizer uses a single vocabulary file, you can point directly to this file</span> | |
| tokenizer = BertTokenizer.from_pretrained(<span class="hljs-string">"./test/saved_model/my_vocab.txt"</span>) | |
| <span class="hljs-comment"># You can link tokens to special vocabulary when instantiating</span> | |
| tokenizer = BertTokenizer.from_pretrained(<span class="hljs-string">"google-bert/bert-base-uncased"</span>, unk_token=<span class="hljs-string">"<unk>"</span>) | |
| <span class="hljs-comment"># You should be sure '<unk>' is in the vocabulary when doing that.</span> | |
| <span class="hljs-comment"># Otherwise use tokenizer.add_special_tokens({'unk_token': '<unk>'}) instead)</span> | |
| <span class="hljs-keyword">assert</span> tokenizer.unk_token == <span class="hljs-string">"<unk>"</span>`,wrap:!1}}),{c(){a=r("p"),a.textContent=z,m=o(),u(T.$$.fragment)},l(l){a=s(l,"P",{"data-svelte-h":!0}),c(a)!=="svelte-kvfsh7"&&(a.textContent=z),m=n(l),h(T.$$.fragment,l)},m(l,P){v(l,a,P),v(l,m,P),f(T,l,P),$=!0},p:ee,i(l){$||(_(T.$$.fragment,l),$=!0)},o(l){g(T.$$.fragment,l),$=!1},d(l){l&&(i(a),i(m)),k(T,l)}}}function Rs(B){let a,z=`If the <code>encoded_inputs</code> passed are dictionary of numpy arrays, PyTorch tensors or TensorFlow tensors, the | |
| result will use the same type unless you provide a different tensor type with <code>return_tensors</code>. In the case of | |
| PyTorch tensors, you will lose the specific device of your tensors however.`;return{c(){a=r("p"),a.innerHTML=z},l(m){a=s(m,"P",{"data-svelte-h":!0}),c(a)!=="svelte-ppz3re"&&(a.innerHTML=z)},m(m,T){v(m,a,T)},p:ee,d(m){m&&i(a)}}}function As(B){let a,z="Examples:",m,T,$;return T=new Xo({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMEElMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZCglMjJnb29nbGUtYmVydCUyRmJlcnQtYmFzZS1jYXNlZCUyMiklMEElMEElMjMlMjBQdXNoJTIwdGhlJTIwdG9rZW5pemVyJTIwdG8lMjB5b3VyJTIwbmFtZXNwYWNlJTIwd2l0aCUyMHRoZSUyMG5hbWUlMjAlMjJteS1maW5ldHVuZWQtYmVydCUyMi4lMEF0b2tlbml6ZXIucHVzaF90b19odWIoJTIybXktZmluZXR1bmVkLWJlcnQlMjIpJTBBJTBBJTIzJTIwUHVzaCUyMHRoZSUyMHRva2VuaXplciUyMHRvJTIwYW4lMjBvcmdhbml6YXRpb24lMjB3aXRoJTIwdGhlJTIwbmFtZSUyMCUyMm15LWZpbmV0dW5lZC1iZXJ0JTIyLiUwQXRva2VuaXplci5wdXNoX3RvX2h1YiglMjJodWdnaW5nZmFjZSUyRm15LWZpbmV0dW5lZC1iZXJ0JTIyKQ==",highlighted:`<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer | |
| tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"google-bert/bert-base-cased"</span>) | |
| <span class="hljs-comment"># Push the tokenizer to your namespace with the name "my-finetuned-bert".</span> | |
| tokenizer.push_to_hub(<span class="hljs-string">"my-finetuned-bert"</span>) | |
| <span class="hljs-comment"># Push the tokenizer to an organization with the name "my-finetuned-bert".</span> | |
| tokenizer.push_to_hub(<span class="hljs-string">"huggingface/my-finetuned-bert"</span>)`,wrap:!1}}),{c(){a=r("p"),a.textContent=z,m=o(),u(T.$$.fragment)},l(l){a=s(l,"P",{"data-svelte-h":!0}),c(a)!=="svelte-kvfsh7"&&(a.textContent=z),m=n(l),h(T.$$.fragment,l)},m(l,P){v(l,a,P),v(l,m,P),f(T,l,P),$=!0},p:ee,i(l){$||(_(T.$$.fragment,l),$=!0)},o(l){g(T.$$.fragment,l),$=!1},d(l){l&&(i(a),i(m)),k(T,l)}}}function Es(B){let a,z="This API is experimental and may have some slight breaking changes in the next releases.";return{c(){a=r("p"),a.textContent=z},l(m){a=s(m,"P",{"data-svelte-h":!0}),c(a)!=="svelte-15rpg4"&&(a.textContent=z)},m(m,T){v(m,a,T)},p:ee,d(m){m&&i(a)}}}function Gs(B){let a,z="Examples:",m,T,$;return T=new Xo({props:{code:"JTIzJTIwTGV0J3MlMjBzZWUlMjBob3clMjB0byUyMGFkZCUyMGElMjBuZXclMjBjbGFzc2lmaWNhdGlvbiUyMHRva2VuJTIwdG8lMjBHUFQtMiUwQXRva2VuaXplciUyMCUzRCUyMEdQVDJUb2tlbml6ZXIuZnJvbV9wcmV0cmFpbmVkKCUyMm9wZW5haS1jb21tdW5pdHklMkZncHQyJTIyKSUwQW1vZGVsJTIwJTNEJTIwR1BUMk1vZGVsLmZyb21fcHJldHJhaW5lZCglMjJvcGVuYWktY29tbXVuaXR5JTJGZ3B0MiUyMiklMEElMEFzcGVjaWFsX3Rva2Vuc19kaWN0JTIwJTNEJTIwJTdCJTIyY2xzX3Rva2VuJTIyJTNBJTIwJTIyJTNDQ0xTJTNFJTIyJTdEJTBBJTBBbnVtX2FkZGVkX3Rva3MlMjAlM0QlMjB0b2tlbml6ZXIuYWRkX3NwZWNpYWxfdG9rZW5zKHNwZWNpYWxfdG9rZW5zX2RpY3QpJTBBcHJpbnQoJTIyV2UlMjBoYXZlJTIwYWRkZWQlMjIlMkMlMjBudW1fYWRkZWRfdG9rcyUyQyUyMCUyMnRva2VucyUyMiklMEElMjMlMjBOb3RpY2UlM0ElMjByZXNpemVfdG9rZW5fZW1iZWRkaW5ncyUyMGV4cGVjdCUyMHRvJTIwcmVjZWl2ZSUyMHRoZSUyMGZ1bGwlMjBzaXplJTIwb2YlMjB0aGUlMjBuZXclMjB2b2NhYnVsYXJ5JTJDJTIwaS5lLiUyQyUyMHRoZSUyMGxlbmd0aCUyMG9mJTIwdGhlJTIwdG9rZW5pemVyLiUwQW1vZGVsLnJlc2l6ZV90b2tlbl9lbWJlZGRpbmdzKGxlbih0b2tlbml6ZXIpKSUwQSUwQWFzc2VydCUyMHRva2VuaXplci5jbHNfdG9rZW4lMjAlM0QlM0QlMjAlMjIlM0NDTFMlM0UlMjI=",highlighted:`<span class="hljs-comment"># Let's see how to add a new classification token to GPT-2</span> | |
| tokenizer = GPT2Tokenizer.from_pretrained(<span class="hljs-string">"openai-community/gpt2"</span>) | |
| model = GPT2Model.from_pretrained(<span class="hljs-string">"openai-community/gpt2"</span>) | |
| special_tokens_dict = {<span class="hljs-string">"cls_token"</span>: <span class="hljs-string">"<CLS>"</span>} | |
| num_added_toks = tokenizer.add_special_tokens(special_tokens_dict) | |
| <span class="hljs-built_in">print</span>(<span class="hljs-string">"We have added"</span>, num_added_toks, <span class="hljs-string">"tokens"</span>) | |
| <span class="hljs-comment"># Notice: resize_token_embeddings expect to receive the full size of the new vocabulary, i.e., the length of the tokenizer.</span> | |
| model.resize_token_embeddings(<span class="hljs-built_in">len</span>(tokenizer)) | |
| <span class="hljs-keyword">assert</span> tokenizer.cls_token == <span class="hljs-string">"<CLS>"</span>`,wrap:!1}}),{c(){a=r("p"),a.textContent=z,m=o(),u(T.$$.fragment)},l(l){a=s(l,"P",{"data-svelte-h":!0}),c(a)!=="svelte-kvfsh7"&&(a.textContent=z),m=n(l),h(T.$$.fragment,l)},m(l,P){v(l,a,P),v(l,m,P),f(T,l,P),$=!0},p:ee,i(l){$||(_(T.$$.fragment,l),$=!0)},o(l){g(T.$$.fragment,l),$=!1},d(l){l&&(i(a),i(m)),k(T,l)}}}function Hs(B){let a,z="Examples:",m,T,$;return T=new Xo({props:{code:"JTIzJTIwTGV0J3MlMjBzZWUlMjBob3clMjB0byUyMGluY3JlYXNlJTIwdGhlJTIwdm9jYWJ1bGFyeSUyMG9mJTIwQmVydCUyMG1vZGVsJTIwYW5kJTIwdG9rZW5pemVyJTBBdG9rZW5pemVyJTIwJTNEJTIwQmVydFRva2VuaXplckZhc3QuZnJvbV9wcmV0cmFpbmVkKCUyMmdvb2dsZS1iZXJ0JTJGYmVydC1iYXNlLXVuY2FzZWQlMjIpJTBBbW9kZWwlMjAlM0QlMjBCZXJ0TW9kZWwuZnJvbV9wcmV0cmFpbmVkKCUyMmdvb2dsZS1iZXJ0JTJGYmVydC1iYXNlLXVuY2FzZWQlMjIpJTBBJTBBbnVtX2FkZGVkX3Rva3MlMjAlM0QlMjB0b2tlbml6ZXIuYWRkX3Rva2VucyglNUIlMjJuZXdfdG9rMSUyMiUyQyUyMCUyMm15X25ldy10b2syJTIyJTVEKSUwQXByaW50KCUyMldlJTIwaGF2ZSUyMGFkZGVkJTIyJTJDJTIwbnVtX2FkZGVkX3Rva3MlMkMlMjAlMjJ0b2tlbnMlMjIpJTBBJTIzJTIwTm90aWNlJTNBJTIwcmVzaXplX3Rva2VuX2VtYmVkZGluZ3MlMjBleHBlY3QlMjB0byUyMHJlY2VpdmUlMjB0aGUlMjBmdWxsJTIwc2l6ZSUyMG9mJTIwdGhlJTIwbmV3JTIwdm9jYWJ1bGFyeSUyQyUyMGkuZS4lMkMlMjB0aGUlMjBsZW5ndGglMjBvZiUyMHRoZSUyMHRva2VuaXplci4lMEFtb2RlbC5yZXNpemVfdG9rZW5fZW1iZWRkaW5ncyhsZW4odG9rZW5pemVyKSk=",highlighted:`<span class="hljs-comment"># Let's see how to increase the vocabulary of Bert model and tokenizer</span> | |
| tokenizer = BertTokenizerFast.from_pretrained(<span class="hljs-string">"google-bert/bert-base-uncased"</span>) | |
| model = BertModel.from_pretrained(<span class="hljs-string">"google-bert/bert-base-uncased"</span>) | |
| num_added_toks = tokenizer.add_tokens([<span class="hljs-string">"new_tok1"</span>, <span class="hljs-string">"my_new-tok2"</span>]) | |
| <span class="hljs-built_in">print</span>(<span class="hljs-string">"We have added"</span>, num_added_toks, <span class="hljs-string">"tokens"</span>) | |
| <span class="hljs-comment"># Notice: resize_token_embeddings expect to receive the full size of the new vocabulary, i.e., the length of the tokenizer.</span> | |
| model.resize_token_embeddings(<span class="hljs-built_in">len</span>(tokenizer))`,wrap:!1}}),{c(){a=r("p"),a.textContent=z,m=o(),u(T.$$.fragment)},l(l){a=s(l,"P",{"data-svelte-h":!0}),c(a)!=="svelte-kvfsh7"&&(a.textContent=z),m=n(l),h(T.$$.fragment,l)},m(l,P){v(l,a,P),v(l,m,P),f(T,l,P),$=!0},p:ee,i(l){$||(_(T.$$.fragment,l),$=!0)},o(l){g(T.$$.fragment,l),$=!1},d(l){l&&(i(a),i(m)),k(T,l)}}}function Xs(B){let a,z,m,T,$,l,P,Cr='이 페이지는 토크나이저에서 사용되는 모든 유틸리티 함수들을 나열하며, 주로 <code>PreTrainedTokenizer</code>와 <code>PreTrainedTokenizerFast</code> 사이의 공통 메소드를 구현하는 <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.PreTrainedTokenizerBase">PreTrainedTokenizerBase</a> 클래스와 <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.SpecialTokensMixin">SpecialTokensMixin</a>을 다룹니다.',zo,ze,Wr="이 함수들 대부분은 라이브러리의 토크나이저 코드를 연구할 때만 유용합니다.",$o,$e,Po,d,Pe,Yo,ft,jr="Base class for <code>PreTrainedTokenizer</code> and <code>PreTrainedTokenizerFast</code>.",Oo,_t,Ur="Handles shared (mostly boiler plate) methods for those two classes.",Qo,gt,Jr="Class attributes (overridden by derived classes)",Ko,kt,Nr=`<li><strong>vocab_files_names</strong> (<code>Dict[str, str]</code>) — A dictionary with, as keys, the <code>__init__</code> keyword name of each | |
| vocabulary file required by the model, and as associated values, the filename for saving the associated file | |
| (string).</li> <li><strong>pretrained_vocab_files_map</strong> (<code>Dict[str, Dict[str, str]]</code>) — A dictionary of dictionaries, with the | |
| high-level keys being the <code>__init__</code> keyword name of each vocabulary file required by the model, the | |
| low-level being the <code>short-cut-names</code> of the pretrained models with, as associated values, the <code>url</code> to the | |
| associated pretrained vocabulary file.</li> <li><strong>model_input_names</strong> (<code>List[str]</code>) — A list of inputs expected in the forward pass of the model.</li> <li><strong>padding_side</strong> (<code>str</code>) — The default value for the side on which the model should have padding applied. | |
| Should be <code>'right'</code> or <code>'left'</code>.</li> <li><strong>truncation_side</strong> (<code>str</code>) — The default value for the side on which the model should have truncation | |
| applied. Should be <code>'right'</code> or <code>'left'</code>.</li>`,en,te,Be,tn,bt,Fr=`Main method to tokenize and prepare for the model one or several sequence(s) or one or several pair(s) of | |
| sequences.`,on,oe,Me,nn,Tt,Zr=`Converts a list of dictionaries with <code>"role"</code> and <code>"content"</code> keys to a list of token | |
| ids. This method is intended for use with chat models, and will read the tokenizer’s chat_template attribute to | |
| determine the format and control tokens to use when converting.`,rn,ne,qe,sn,vt,Vr=`Temporarily sets the tokenizer for encoding the targets. Useful for tokenizer associated to | |
| sequence-to-sequence models that need a slightly different processing for the labels.`,an,re,Ie,dn,xt,Dr="Convert a list of lists of token ids into a list of strings by calling decode.",ln,U,Le,cn,yt,Sr="Tokenize and prepare for the model a list of sequences or a list of pairs of sequences.",pn,se,mn,J,Ce,un,wt,Rr=`Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and | |
| adding special tokens.`,hn,zt,Ar="This implementation does not add special tokens and this method should be overridden in a subclass.",fn,ae,We,_n,$t,Er="Clean up a list of simple English tokenization artifacts like spaces before punctuations and abbreviated forms.",gn,ie,je,kn,Pt,Gr=`Converts a sequence of tokens in a single string. The most simple way to do it is <code>" ".join(tokens)</code> but we | |
| often want to remove sub-word tokenization artifacts at the same time.`,bn,N,Ue,Tn,Bt,Hr=`Create the token type IDs corresponding to the sequences passed. <a href="../glossary#token-type-ids">What are token type | |
| IDs?</a>`,vn,Mt,Xr="Should be overridden in a subclass if the model has a special way of building those.",xn,F,Je,yn,qt,Yr=`Converts a sequence of ids in a string, using the tokenizer and vocabulary with options to remove special | |
| tokens and clean up tokenization spaces.`,wn,It,Or="Similar to doing <code>self.convert_tokens_to_string(self.convert_ids_to_tokens(token_ids))</code>.",zn,Z,Ne,$n,Lt,Qr="Converts a string to a sequence of ids (integer), using the tokenizer and vocabulary.",Pn,Ct,Kr="Same as doing <code>self.convert_tokens_to_ids(self.tokenize(text))</code>.",Bn,V,Fe,Mn,Wt,es="Tokenize and prepare for the model a sequence or a pair of sequences.",qn,de,In,W,Ze,Ln,jt,ts=`Instantiate a <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.PreTrainedTokenizerBase">PreTrainedTokenizerBase</a> (or a derived class) from a predefined | |
| tokenizer.`,Cn,le,Wn,ce,jn,pe,Ve,Un,Ut,os=`Retrieve the chat template string used for tokenizing chat messages. This template is used | |
| internally by the <code>apply_chat_template</code> method and can also be used externally to retrieve the model’s chat | |
| template for better generation tracking.`,Jn,me,De,Nn,Jt,ns=`Retrieves sequence ids from a token list that has no special tokens added. This method is called when adding | |
| special tokens using the tokenizer <code>prepare_for_model</code> or <code>encode_plus</code> methods.`,Fn,D,Se,Zn,Nt,rs="Returns the vocabulary as a dictionary of token to index.",Vn,Ft,ss=`<code>tokenizer.get_vocab()[token]</code> is equivalent to <code>tokenizer.convert_tokens_to_ids(token)</code> when <code>token</code> is in the | |
| vocab.`,Dn,L,Re,Sn,Zt,as=`Pad a single encoded input or a batch of encoded inputs up to predefined length or to the max sequence length | |
| in the batch.`,Rn,Vt,is=`Padding side (left/right) padding token ids are defined at the tokenizer level (with <code>self.padding_side</code>, | |
| <code>self.pad_token_id</code> and <code>self.pad_token_type_id</code>).`,An,Dt,ds=`Please note that with a fast tokenizer, using the <code>__call__</code> method is faster than using a method to encode the | |
| text followed by a call to the <code>pad</code> method to get a padded encoding.`,En,ue,Gn,he,Ae,Hn,St,ls=`Prepares a sequence of input id, or a pair of sequences of inputs ids so that it can be used by the model. It | |
| adds special tokens, truncates sequences if overflowing while taking into account the special tokens and | |
| manages a moving window (with user defined stride) for overflowing tokens. Please Note, for <em>pair_ids</em> | |
| different than <code>None</code> and <em>truncation_strategy = longest_first</em> or <code>True</code>, it is not possible to return | |
| overflowing tokens. Such a combination of arguments will raise an error.`,Xn,fe,Ee,Yn,Rt,cs="Prepare model inputs for translation. For best performance, translate one sentence at a time.",On,S,Ge,Qn,At,ps="Upload the tokenizer files to the 🤗 Model Hub.",Kn,_e,er,R,He,tr,Et,ms=`Register this class with a given auto class. This should only be used for custom tokenizers as the ones in the | |
| library are already mapped with <code>AutoTokenizer</code>.`,or,ge,nr,j,Xe,rr,Gt,us="Save the full tokenizer state.",sr,Ht,hs=`This method make sure the full tokenizer can then be re-loaded using the | |
| <code>~tokenization_utils_base.PreTrainedTokenizer.from_pretrained</code> class method..`,ar,Xt,fs=`Warning,None This won’t save modifications you may have applied to the tokenizer after the instantiation (for | |
| instance, modifying <code>tokenizer.do_lower_case</code> after creation).`,ir,A,Ye,dr,Yt,_s="Save only the vocabulary of the tokenizer (vocabulary + added tokens).",lr,Ot,gs=`This method won’t save the configuration and special token mappings of the tokenizer. Use | |
| <code>_save_pretrained()</code> to save the whole state of the tokenizer.`,cr,ke,Oe,pr,Qt,ks="Converts a string into a sequence of tokens, replacing unknown tokens with the <code>unk_token</code>.",mr,be,Qe,ur,Kt,bs="Truncates a sequence pair in-place following the strategy.",Bo,Ke,Mo,I,et,hr,eo,Ts=`A mixin derived by <code>PreTrainedTokenizer</code> and <code>PreTrainedTokenizerFast</code> to handle specific behaviors related to | |
| special tokens. In particular, this class hold the attributes which can be used to directly access these special | |
| tokens in a model-independent manner and allow to set and update the special tokens.`,fr,M,tt,_r,to,vs=`Add a dictionary of special tokens (eos, pad, cls, etc.) to the encoder and link them to class attributes. If | |
| special tokens are NOT in the vocabulary, they are added to it (indexed starting from the last index of the | |
| current vocabulary).`,gr,oo,xs=`When adding new tokens to the vocabulary, you should make sure to also resize the token embedding matrix of the | |
| model so that its embedding matrix matches the tokenizer.`,kr,no,ys='In order to do that, please use the <a href="/docs/transformers/pr_34323/ko/main_classes/model#transformers.PreTrainedModel.resize_token_embeddings">resize_token_embeddings()</a> method.',br,ro,ws="Using <code>add_special_tokens</code> will ensure your special tokens can be used in several ways:",Tr,so,zs=`<li>Special tokens can be skipped when decoding using <code>skip_special_tokens = True</code>.</li> <li>Special tokens are carefully handled by the tokenizer (they are never split), similar to <code>AddedTokens</code>.</li> <li>You can easily refer to special tokens using tokenizer class attributes like <code>tokenizer.cls_token</code>. This | |
| makes it easy to develop model-agnostic training and fine-tuning scripts.</li>`,vr,ao,$s=`When possible, special tokens are already registered for provided pretrained models (for instance | |
| <code>BertTokenizer</code> <code>cls_token</code> is already registered to be :obj<em>’[CLS]’</em> and XLM’s one is also registered to be | |
| <code>'</s>'</code>).`,xr,Te,yr,C,ot,wr,io,Ps=`Add a list of new tokens to the tokenizer class. If the new tokens are not in the vocabulary, they are added to | |
| it with indices starting from length of the current vocabulary and and will be isolated before the tokenization | |
| algorithm is applied. Added tokens and tokens from the vocabulary of the tokenization algorithm are therefore | |
| not treated in the same way.`,zr,lo,Bs=`Note, when adding new tokens to the vocabulary, you should make sure to also resize the token embedding matrix | |
| of the model so that its embedding matrix matches the tokenizer.`,$r,co,Ms='In order to do that, please use the <a href="/docs/transformers/pr_34323/ko/main_classes/model#transformers.PreTrainedModel.resize_token_embeddings">resize_token_embeddings()</a> method.',Pr,ve,Br,xe,nt,Mr,po,qs=`The <code>sanitize_special_tokens</code> is now deprecated kept for backward compatibility and will be removed in | |
| transformers v5.`,qo,rt,Io,X,st,qr,mo,Is=`Possible values for the <code>truncation</code> argument in <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizerBase.<strong>call</strong>()</a>. Useful for tab-completion in | |
| an IDE.`,Lo,Y,at,Ir,uo,Ls="Character span in the original string.",Co,O,it,Lr,ho,Cs="Token span in an encoded string (list of tokens).",Wo,dt,jo,yo,Uo;return $=new Ho({props:{title:"토크나이저를 위한 유틸리티",local:"utilities-for-tokenizers",headingTag:"h1"}}),$e=new Ho({props:{title:"PreTrainedTokenizerBase",local:"transformers.PreTrainedTokenizerBase ][ transformers.PreTrainedTokenizerBase",headingTag:"h2"}}),Pe=new w({props:{name:"class transformers.PreTrainedTokenizerBase",anchor:"transformers.PreTrainedTokenizerBase",parameters:[{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.model_max_length",description:`<strong>model_max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| The maximum length (in number of tokens) for the inputs to the transformer model. When the tokenizer is | |
| loaded with <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.from_pretrained">from_pretrained()</a>, this will be set to the | |
| value stored for the associated model in <code>max_model_input_sizes</code> (see above). If no value is provided, will | |
| default to VERY_LARGE_INTEGER (<code>int(1e30)</code>).`,name:"model_max_length"},{anchor:"transformers.PreTrainedTokenizerBase.padding_side",description:`<strong>padding_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have padding applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"padding_side"},{anchor:"transformers.PreTrainedTokenizerBase.truncation_side",description:`<strong>truncation_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have truncation applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"truncation_side"},{anchor:"transformers.PreTrainedTokenizerBase.chat_template",description:`<strong>chat_template</strong> (<code>str</code>, <em>optional</em>) — | |
| A Jinja template string that will be used to format lists of chat messages. See | |
| <a href="https://huggingface.co/docs/transformers/chat_templating" rel="nofollow">https://huggingface.co/docs/transformers/chat_templating</a> for a full description.`,name:"chat_template"},{anchor:"transformers.PreTrainedTokenizerBase.model_input_names",description:`<strong>model_input_names</strong> (<code>List[string]</code>, <em>optional</em>) — | |
| The list of inputs accepted by the forward pass of the model (like <code>"token_type_ids"</code> or | |
| <code>"attention_mask"</code>). Default value is picked from the class attribute of the same name.`,name:"model_input_names"},{anchor:"transformers.PreTrainedTokenizerBase.bos_token",description:`<strong>bos_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing the beginning of a sentence. Will be associated to <code>self.bos_token</code> and | |
| <code>self.bos_token_id</code>.`,name:"bos_token"},{anchor:"transformers.PreTrainedTokenizerBase.eos_token",description:`<strong>eos_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing the end of a sentence. Will be associated to <code>self.eos_token</code> and | |
| <code>self.eos_token_id</code>.`,name:"eos_token"},{anchor:"transformers.PreTrainedTokenizerBase.unk_token",description:`<strong>unk_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing an out-of-vocabulary token. Will be associated to <code>self.unk_token</code> and | |
| <code>self.unk_token_id</code>.`,name:"unk_token"},{anchor:"transformers.PreTrainedTokenizerBase.sep_token",description:`<strong>sep_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token separating two different sentences in the same input (used by BERT for instance). Will be | |
| associated to <code>self.sep_token</code> and <code>self.sep_token_id</code>.`,name:"sep_token"},{anchor:"transformers.PreTrainedTokenizerBase.pad_token",description:`<strong>pad_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token used to make arrays of tokens the same size for batching purpose. Will then be ignored by | |
| attention mechanisms or loss computation. Will be associated to <code>self.pad_token</code> and <code>self.pad_token_id</code>.`,name:"pad_token"},{anchor:"transformers.PreTrainedTokenizerBase.cls_token",description:`<strong>cls_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing the class of the input (used by BERT for instance). Will be associated to | |
| <code>self.cls_token</code> and <code>self.cls_token_id</code>.`,name:"cls_token"},{anchor:"transformers.PreTrainedTokenizerBase.mask_token",description:`<strong>mask_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing a masked token (used by masked-language modeling pretraining objectives, like | |
| BERT). Will be associated to <code>self.mask_token</code> and <code>self.mask_token_id</code>.`,name:"mask_token"},{anchor:"transformers.PreTrainedTokenizerBase.additional_special_tokens",description:`<strong>additional_special_tokens</strong> (tuple or list of <code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A tuple or a list of additional special tokens. Add them here to ensure they are skipped when decoding with | |
| <code>skip_special_tokens</code> is set to True. If they are not part of the vocabulary, they will be added at the end | |
| of the vocabulary.`,name:"additional_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.clean_up_tokenization_spaces",description:`<strong>clean_up_tokenization_spaces</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not the model should cleanup the spaces that were added when splitting the input text during the | |
| tokenization process.`,name:"clean_up_tokenization_spaces"},{anchor:"transformers.PreTrainedTokenizerBase.split_special_tokens",description:`<strong>split_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the special tokens should be split during the tokenization process. Passing will affect the | |
| internal state of the tokenizer. The default behavior is to not split special tokens. This means that if | |
| <code><s></code> is the <code>bos_token</code>, then <code>tokenizer.tokenize("<s>") = ['<s></code>]. Otherwise, if | |
| <code>split_special_tokens=True</code>, then <code>tokenizer.tokenize("<s>")</code> will be give <code>['<','s', '>']</code>.`,name:"split_special_tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L1568"}}),Be=new w({props:{name:"__call__",anchor:"transformers.PreTrainedTokenizerBase.__call__",parameters:[{name:"text",val:": Union = None"},{name:"text_pair",val:": Union = None"},{name:"text_target",val:": Union = None"},{name:"text_pair_target",val:": Union = None"},{name:"add_special_tokens",val:": bool = True"},{name:"padding",val:": Union = False"},{name:"truncation",val:": Union = None"},{name:"max_length",val:": Optional = None"},{name:"stride",val:": int = 0"},{name:"is_split_into_words",val:": bool = False"},{name:"pad_to_multiple_of",val:": Optional = None"},{name:"padding_side",val:": Optional = None"},{name:"return_tensors",val:": Union = None"},{name:"return_token_type_ids",val:": Optional = None"},{name:"return_attention_mask",val:": Optional = None"},{name:"return_overflowing_tokens",val:": bool = False"},{name:"return_special_tokens_mask",val:": bool = False"},{name:"return_offsets_mapping",val:": bool = False"},{name:"return_length",val:": bool = False"},{name:"verbose",val:": bool = True"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.__call__.text",description:`<strong>text</strong> (<code>str</code>, <code>List[str]</code>, <code>List[List[str]]</code>, <em>optional</em>) — | |
| The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings | |
| (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set | |
| <code>is_split_into_words=True</code> (to lift the ambiguity with a batch of sequences).`,name:"text"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.text_pair",description:`<strong>text_pair</strong> (<code>str</code>, <code>List[str]</code>, <code>List[List[str]]</code>, <em>optional</em>) — | |
| The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings | |
| (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set | |
| <code>is_split_into_words=True</code> (to lift the ambiguity with a batch of sequences).`,name:"text_pair"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.text_target",description:`<strong>text_target</strong> (<code>str</code>, <code>List[str]</code>, <code>List[List[str]]</code>, <em>optional</em>) — | |
| The sequence or batch of sequences to be encoded as target texts. Each sequence can be a string or a | |
| list of strings (pretokenized string). If the sequences are provided as list of strings (pretokenized), | |
| you must set <code>is_split_into_words=True</code> (to lift the ambiguity with a batch of sequences).`,name:"text_target"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.text_pair_target",description:`<strong>text_pair_target</strong> (<code>str</code>, <code>List[str]</code>, <code>List[List[str]]</code>, <em>optional</em>) — | |
| The sequence or batch of sequences to be encoded as target texts. Each sequence can be a string or a | |
| list of strings (pretokenized string). If the sequences are provided as list of strings (pretokenized), | |
| you must set <code>is_split_into_words=True</code> (to lift the ambiguity with a batch of sequences).`,name:"text_pair_target"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.add_special_tokens",description:`<strong>add_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to add special tokens when encoding the sequences. This will use the underlying | |
| <code>PretrainedTokenizerBase.build_inputs_with_special_tokens</code> function, which defines which tokens are | |
| automatically added to the input ids. This is usefull if you want to add <code>bos</code> or <code>eos</code> tokens | |
| automatically.`,name:"add_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.padding",description:`<strong>padding</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.utils.PaddingStrategy">PaddingStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls padding. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest'</code>: Pad to the longest sequence in the batch (or no padding if only a single | |
| sequence if provided).</li> | |
| <li><code>'max_length'</code>: Pad to a maximum length specified with the argument <code>max_length</code> or to the maximum | |
| acceptable input length for the model if that argument is not provided.</li> | |
| <li><code>False</code> or <code>'do_not_pad'</code> (default): No padding (i.e., can output a batch with sequences of different | |
| lengths).</li> | |
| </ul>`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.truncation",description:`<strong>truncation</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.tokenization_utils_base.TruncationStrategy">TruncationStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls truncation. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or | |
| to the maximum acceptable input length for the model if that argument is not provided. This will | |
| truncate token by token, removing a token from the longest sequence in the pair if a pair of | |
| sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_second'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>False</code> or <code>'do_not_truncate'</code> (default): No truncation (i.e., can output batch with sequence lengths | |
| greater than the model maximum admissible input size).</li> | |
| </ul>`,name:"truncation"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Controls the maximum length to use by one of the truncation/padding parameters.</p> | |
| <p>If left unset or set to <code>None</code>, this will use the predefined model maximum length if a maximum length | |
| is required by one of the truncation/padding parameters. If the model has no specific maximum input | |
| length (like XLNet) truncation/padding to a maximum length will be deactivated.`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.stride",description:`<strong>stride</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| If set to a number along with <code>max_length</code>, the overflowing tokens returned when | |
| <code>return_overflowing_tokens=True</code> will contain some tokens from the end of the truncated sequence | |
| returned to provide some overlap between truncated and overflowing sequences. The value of this | |
| argument defines the number of overlapping tokens.`,name:"stride"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.is_split_into_words",description:`<strong>is_split_into_words</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the input is already pre-tokenized (e.g., split into words). If set to <code>True</code>, the | |
| tokenizer assumes the input is already split into words (for instance, by splitting it on whitespace) | |
| which it will tokenize. This is useful for NER or token classification.`,name:"is_split_into_words"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.pad_to_multiple_of",description:`<strong>pad_to_multiple_of</strong> (<code>int</code>, <em>optional</em>) — | |
| If set will pad the sequence to a multiple of the provided value. Requires <code>padding</code> to be activated. | |
| This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability | |
| <code>>= 7.5</code> (Volta).`,name:"pad_to_multiple_of"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.padding_side",description:`<strong>padding_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have padding applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"padding_side"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors instead of list of python integers. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.constant</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return Numpy <code>np.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.return_token_type_ids",description:`<strong>return_token_type_ids</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return token type IDs. If left to the default, will return the token type IDs according to | |
| the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a>`,name:"return_token_type_ids"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.return_attention_mask",description:`<strong>return_attention_mask</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return the attention mask. If left to the default, will return the attention mask according | |
| to the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"return_attention_mask"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.return_overflowing_tokens",description:`<strong>return_overflowing_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return overflowing token sequences. If a pair of sequences of input ids (or a batch | |
| of pairs) is provided with <code>truncation_strategy = longest_first</code> or <code>True</code>, an error is raised instead | |
| of returning overflowing tokens.`,name:"return_overflowing_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.return_special_tokens_mask",description:`<strong>return_special_tokens_mask</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return special tokens mask information.`,name:"return_special_tokens_mask"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.return_offsets_mapping",description:`<strong>return_offsets_mapping</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return <code>(char_start, char_end)</code> for each token.</p> | |
| <p>This is only available on fast tokenizers inheriting from <code>PreTrainedTokenizerFast</code>, if using | |
| Python’s tokenizer, this method will raise <code>NotImplementedError</code>.`,name:"return_offsets_mapping"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.return_length",description:`<strong>return_length</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return the lengths of the encoded inputs.`,name:"return_length"},{anchor:"transformers.PreTrainedTokenizerBase.__call__.verbose",description:`<strong>verbose</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to print more information and warnings. | |
| **kwargs — passed to the <code>self.tokenize()</code> method`,name:"verbose"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L2944",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>BatchEncoding</code> with the following fields:</p> | |
| <ul> | |
| <li> | |
| <p><strong>input_ids</strong> — List of token ids to be fed to a model.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>token_type_ids</strong> — List of token type ids to be fed to a model (when <code>return_token_type_ids=True</code> or | |
| if <em>“token_type_ids”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>attention_mask</strong> — List of indices specifying which tokens should be attended to by the model (when | |
| <code>return_attention_mask=True</code> or if <em>“attention_mask”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>overflowing_tokens</strong> — List of overflowing tokens sequences (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>num_truncated_tokens</strong> — Number of tokens truncated (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>special_tokens_mask</strong> — List of 0s and 1s, with 1 specifying added special tokens and 0 specifying | |
| regular sequence tokens (when <code>add_special_tokens=True</code> and <code>return_special_tokens_mask=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>length</strong> — The length of the inputs (when <code>return_length=True</code>)</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>BatchEncoding</code></p> | |
| `}}),Me=new w({props:{name:"apply_chat_template",anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template",parameters:[{name:"conversation",val:": Union"},{name:"tools",val:": Optional = None"},{name:"documents",val:": Optional = None"},{name:"chat_template",val:": Optional = None"},{name:"add_generation_prompt",val:": bool = False"},{name:"continue_final_message",val:": bool = False"},{name:"tokenize",val:": bool = True"},{name:"padding",val:": bool = False"},{name:"truncation",val:": bool = False"},{name:"max_length",val:": Optional = None"},{name:"return_tensors",val:": Union = None"},{name:"return_dict",val:": bool = False"},{name:"return_assistant_tokens_mask",val:": bool = False"},{name:"tokenizer_kwargs",val:": Optional = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.conversation",description:`<strong>conversation</strong> (Union[List[Dict[str, str]], List[List[Dict[str, str]]]]) — A list of dicts | |
| with “role” and “content” keys, representing the chat history so far.`,name:"conversation"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.tools",description:`<strong>tools</strong> (<code>List[Dict]</code>, <em>optional</em>) — | |
| A list of tools (callable functions) that will be accessible to the model. If the template does not | |
| support function calling, this argument will have no effect. Each tool should be passed as a JSON Schema, | |
| giving the name, description and argument types for the tool. See our | |
| <a href="https://huggingface.co/docs/transformers/main/en/chat_templating#automated-function-conversion-for-tool-use" rel="nofollow">chat templating guide</a> | |
| for more information.`,name:"tools"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.documents",description:`<strong>documents</strong> (<code>List[Dict[str, str]]</code>, <em>optional</em>) — | |
| A list of dicts representing documents that will be accessible to the model if it is performing RAG | |
| (retrieval-augmented generation). If the template does not support RAG, this argument will have no | |
| effect. We recommend that each document should be a dict containing “title” and “text” keys. Please | |
| see the RAG section of the <a href="https://huggingface.co/docs/transformers/main/en/chat_templating#arguments-for-RAG" rel="nofollow">chat templating guide</a> | |
| for examples of passing documents with chat templates.`,name:"documents"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.chat_template",description:`<strong>chat_template</strong> (<code>str</code>, <em>optional</em>) — | |
| A Jinja template to use for this conversion. It is usually not necessary to pass anything to this | |
| argument, as the model’s template will be used by default.`,name:"chat_template"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.add_generation_prompt",description:`<strong>add_generation_prompt</strong> (bool, <em>optional</em>) — | |
| If this is set, a prompt with the token(s) that indicate | |
| the start of an assistant message will be appended to the formatted output. This is useful when you want to generate a response from the model. | |
| Note that this argument will be passed to the chat template, and so it must be supported in the | |
| template for this argument to have any effect.`,name:"add_generation_prompt"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.continue_final_message",description:`<strong>continue_final_message</strong> (bool, <em>optional</em>) — | |
| If this is set, the chat will be formatted so that the final | |
| message in the chat is open-ended, without any EOS tokens. The model will continue this message | |
| rather than starting a new one. This allows you to “prefill” part of | |
| the model’s response for it. Cannot be used at the same time as <code>add_generation_prompt</code>.`,name:"continue_final_message"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.tokenize",description:`<strong>tokenize</strong> (<code>bool</code>, defaults to <code>True</code>) — | |
| Whether to tokenize the output. If <code>False</code>, the output will be a string.`,name:"tokenize"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.padding",description:`<strong>padding</strong> (<code>bool</code>, defaults to <code>False</code>) — | |
| Whether to pad sequences to the maximum length. Has no effect if tokenize is <code>False</code>.`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.truncation",description:`<strong>truncation</strong> (<code>bool</code>, defaults to <code>False</code>) — | |
| Whether to truncate sequences at the maximum length. Has no effect if tokenize is <code>False</code>.`,name:"truncation"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Maximum length (in tokens) to use for padding or truncation. Has no effect if tokenize is <code>False</code>. If | |
| not specified, the tokenizer’s <code>max_length</code> attribute will be used as a default.`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors of a particular framework. Has no effect if tokenize is <code>False</code>. Acceptable | |
| values are:<ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.Tensor</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return NumPy <code>np.ndarray</code> objects.</li> | |
| <li><code>'jax'</code>: Return JAX <code>jnp.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, defaults to <code>False</code>) — | |
| Whether to return a dictionary with named outputs. Has no effect if tokenize is <code>False</code>.`,name:"return_dict"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.tokenizer_kwargs",description:"<strong>tokenizer_kwargs</strong> (<code>Dict[str -- Any]</code>, <em>optional</em>): Additional kwargs to pass to the tokenizer.",name:"tokenizer_kwargs"},{anchor:"transformers.PreTrainedTokenizerBase.apply_chat_template.return_assistant_tokens_mask",description:`<strong>return_assistant_tokens_mask</strong> (<code>bool</code>, defaults to <code>False</code>) — | |
| Whether to return a mask of the assistant generated tokens. For tokens generated by the assistant, | |
| the mask will contain 1. For user and system tokens, the mask will contain 0. | |
| This functionality is only available for chat templates that support it via the <code>{% generation %}</code> keyword. | |
| **kwargs — Additional kwargs to pass to the template renderer. Will be accessible by the chat template.`,name:"return_assistant_tokens_mask"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L1709",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of token ids representing the tokenized chat so far, including control tokens. This | |
| output is ready to pass to the model, either directly or via methods like <code>generate()</code>. If <code>return_dict</code> is | |
| set, will return a dict of tokenizer outputs instead.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>Union[List[int], Dict]</code></p> | |
| `}}),qe=new w({props:{name:"as_target_tokenizer",anchor:"transformers.PreTrainedTokenizerBase.as_target_tokenizer",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L4108"}}),Ie=new w({props:{name:"batch_decode",anchor:"transformers.PreTrainedTokenizerBase.batch_decode",parameters:[{name:"sequences",val:": Union"},{name:"skip_special_tokens",val:": bool = False"},{name:"clean_up_tokenization_spaces",val:": bool = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.batch_decode.sequences",description:`<strong>sequences</strong> (<code>Union[List[int], List[List[int]], np.ndarray, torch.Tensor, tf.Tensor]</code>) — | |
| List of tokenized input ids. Can be obtained using the <code>__call__</code> method.`,name:"sequences"},{anchor:"transformers.PreTrainedTokenizerBase.batch_decode.skip_special_tokens",description:`<strong>skip_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to remove special tokens in the decoding.`,name:"skip_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.batch_decode.clean_up_tokenization_spaces",description:`<strong>clean_up_tokenization_spaces</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to clean up the tokenization spaces. If <code>None</code>, will default to | |
| <code>self.clean_up_tokenization_spaces</code>.`,name:"clean_up_tokenization_spaces"},{anchor:"transformers.PreTrainedTokenizerBase.batch_decode.kwargs",description:`<strong>kwargs</strong> (additional keyword arguments, <em>optional</em>) — | |
| Will be passed to the underlying model specific decode method.`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3940",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The list of decoded sentences.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[str]</code></p> | |
| `}}),Le=new w({props:{name:"batch_encode_plus",anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus",parameters:[{name:"batch_text_or_text_pairs",val:": Union"},{name:"add_special_tokens",val:": bool = True"},{name:"padding",val:": Union = False"},{name:"truncation",val:": Union = None"},{name:"max_length",val:": Optional = None"},{name:"stride",val:": int = 0"},{name:"is_split_into_words",val:": bool = False"},{name:"pad_to_multiple_of",val:": Optional = None"},{name:"padding_side",val:": Optional = None"},{name:"return_tensors",val:": Union = None"},{name:"return_token_type_ids",val:": Optional = None"},{name:"return_attention_mask",val:": Optional = None"},{name:"return_overflowing_tokens",val:": bool = False"},{name:"return_special_tokens_mask",val:": bool = False"},{name:"return_offsets_mapping",val:": bool = False"},{name:"return_length",val:": bool = False"},{name:"verbose",val:": bool = True"},{name:"split_special_tokens",val:": bool = False"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.batch_text_or_text_pairs",description:`<strong>batch_text_or_text_pairs</strong> (<code>List[str]</code>, <code>List[Tuple[str, str]]</code>, <code>List[List[str]]</code>, <code>List[Tuple[List[str], List[str]]]</code>, and for not-fast tokenizers, also <code>List[List[int]]</code>, <code>List[Tuple[List[int], List[int]]]</code>) — | |
| Batch of sequences or pair of sequences to be encoded. This can be a list of | |
| string/string-sequences/int-sequences or a list of pair of string/string-sequences/int-sequence (see | |
| details in <code>encode_plus</code>).`,name:"batch_text_or_text_pairs"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.add_special_tokens",description:`<strong>add_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to add special tokens when encoding the sequences. This will use the underlying | |
| <code>PretrainedTokenizerBase.build_inputs_with_special_tokens</code> function, which defines which tokens are | |
| automatically added to the input ids. This is usefull if you want to add <code>bos</code> or <code>eos</code> tokens | |
| automatically.`,name:"add_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.padding",description:`<strong>padding</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.utils.PaddingStrategy">PaddingStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls padding. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest'</code>: Pad to the longest sequence in the batch (or no padding if only a single | |
| sequence if provided).</li> | |
| <li><code>'max_length'</code>: Pad to a maximum length specified with the argument <code>max_length</code> or to the maximum | |
| acceptable input length for the model if that argument is not provided.</li> | |
| <li><code>False</code> or <code>'do_not_pad'</code> (default): No padding (i.e., can output a batch with sequences of different | |
| lengths).</li> | |
| </ul>`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.truncation",description:`<strong>truncation</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.tokenization_utils_base.TruncationStrategy">TruncationStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls truncation. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or | |
| to the maximum acceptable input length for the model if that argument is not provided. This will | |
| truncate token by token, removing a token from the longest sequence in the pair if a pair of | |
| sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_second'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>False</code> or <code>'do_not_truncate'</code> (default): No truncation (i.e., can output batch with sequence lengths | |
| greater than the model maximum admissible input size).</li> | |
| </ul>`,name:"truncation"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Controls the maximum length to use by one of the truncation/padding parameters.</p> | |
| <p>If left unset or set to <code>None</code>, this will use the predefined model maximum length if a maximum length | |
| is required by one of the truncation/padding parameters. If the model has no specific maximum input | |
| length (like XLNet) truncation/padding to a maximum length will be deactivated.`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.stride",description:`<strong>stride</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| If set to a number along with <code>max_length</code>, the overflowing tokens returned when | |
| <code>return_overflowing_tokens=True</code> will contain some tokens from the end of the truncated sequence | |
| returned to provide some overlap between truncated and overflowing sequences. The value of this | |
| argument defines the number of overlapping tokens.`,name:"stride"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.is_split_into_words",description:`<strong>is_split_into_words</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the input is already pre-tokenized (e.g., split into words). If set to <code>True</code>, the | |
| tokenizer assumes the input is already split into words (for instance, by splitting it on whitespace) | |
| which it will tokenize. This is useful for NER or token classification.`,name:"is_split_into_words"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.pad_to_multiple_of",description:`<strong>pad_to_multiple_of</strong> (<code>int</code>, <em>optional</em>) — | |
| If set will pad the sequence to a multiple of the provided value. Requires <code>padding</code> to be activated. | |
| This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability | |
| <code>>= 7.5</code> (Volta).`,name:"pad_to_multiple_of"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.padding_side",description:`<strong>padding_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have padding applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"padding_side"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors instead of list of python integers. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.constant</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return Numpy <code>np.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.return_token_type_ids",description:`<strong>return_token_type_ids</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return token type IDs. If left to the default, will return the token type IDs according to | |
| the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a>`,name:"return_token_type_ids"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.return_attention_mask",description:`<strong>return_attention_mask</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return the attention mask. If left to the default, will return the attention mask according | |
| to the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"return_attention_mask"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.return_overflowing_tokens",description:`<strong>return_overflowing_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return overflowing token sequences. If a pair of sequences of input ids (or a batch | |
| of pairs) is provided with <code>truncation_strategy = longest_first</code> or <code>True</code>, an error is raised instead | |
| of returning overflowing tokens.`,name:"return_overflowing_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.return_special_tokens_mask",description:`<strong>return_special_tokens_mask</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return special tokens mask information.`,name:"return_special_tokens_mask"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.return_offsets_mapping",description:`<strong>return_offsets_mapping</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return <code>(char_start, char_end)</code> for each token.</p> | |
| <p>This is only available on fast tokenizers inheriting from <code>PreTrainedTokenizerFast</code>, if using | |
| Python’s tokenizer, this method will raise <code>NotImplementedError</code>.`,name:"return_offsets_mapping"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.return_length",description:`<strong>return_length</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return the lengths of the encoded inputs.`,name:"return_length"},{anchor:"transformers.PreTrainedTokenizerBase.batch_encode_plus.verbose",description:`<strong>verbose</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to print more information and warnings. | |
| **kwargs — passed to the <code>self.tokenize()</code> method`,name:"verbose"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3255",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>BatchEncoding</code> with the following fields:</p> | |
| <ul> | |
| <li> | |
| <p><strong>input_ids</strong> — List of token ids to be fed to a model.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>token_type_ids</strong> — List of token type ids to be fed to a model (when <code>return_token_type_ids=True</code> or | |
| if <em>“token_type_ids”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>attention_mask</strong> — List of indices specifying which tokens should be attended to by the model (when | |
| <code>return_attention_mask=True</code> or if <em>“attention_mask”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>overflowing_tokens</strong> — List of overflowing tokens sequences (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>num_truncated_tokens</strong> — Number of tokens truncated (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>special_tokens_mask</strong> — List of 0s and 1s, with 1 specifying added special tokens and 0 specifying | |
| regular sequence tokens (when <code>add_special_tokens=True</code> and <code>return_special_tokens_mask=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>length</strong> — The length of the inputs (when <code>return_length=True</code>)</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>BatchEncoding</code></p> | |
| `}}),se=new wo({props:{warning:!0,$$slots:{default:[Zs]},$$scope:{ctx:B}}}),Ce=new w({props:{name:"build_inputs_with_special_tokens",anchor:"transformers.PreTrainedTokenizerBase.build_inputs_with_special_tokens",parameters:[{name:"token_ids_0",val:": List"},{name:"token_ids_1",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.build_inputs_with_special_tokens.token_ids_0",description:"<strong>token_ids_0</strong> (<code>List[int]</code>) — The first tokenized sequence.",name:"token_ids_0"},{anchor:"transformers.PreTrainedTokenizerBase.build_inputs_with_special_tokens.token_ids_1",description:"<strong>token_ids_1</strong> (<code>List[int]</code>, <em>optional</em>) — The second tokenized sequence.",name:"token_ids_1"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3563",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The model input with special tokens.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[int]</code></p> | |
| `}}),We=new w({props:{name:"clean_up_tokenization",anchor:"transformers.PreTrainedTokenizerBase.clean_up_tokenization",parameters:[{name:"out_string",val:": str"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.clean_up_tokenization.out_string",description:"<strong>out_string</strong> (<code>str</code>) — The text to clean up.",name:"out_string"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L4051",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The cleaned-up string.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>str</code></p> | |
| `}}),je=new w({props:{name:"convert_tokens_to_string",anchor:"transformers.PreTrainedTokenizerBase.convert_tokens_to_string",parameters:[{name:"tokens",val:": List"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.convert_tokens_to_string.tokens",description:"<strong>tokens</strong> (<code>List[str]</code>) — The token to join in a string.",name:"tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3927",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The joined tokens.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>str</code></p> | |
| `}}),Ue=new w({props:{name:"create_token_type_ids_from_sequences",anchor:"transformers.PreTrainedTokenizerBase.create_token_type_ids_from_sequences",parameters:[{name:"token_ids_0",val:": List"},{name:"token_ids_1",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.create_token_type_ids_from_sequences.token_ids_0",description:"<strong>token_ids_0</strong> (<code>List[int]</code>) — The first tokenized sequence.",name:"token_ids_0"},{anchor:"transformers.PreTrainedTokenizerBase.create_token_type_ids_from_sequences.token_ids_1",description:"<strong>token_ids_1</strong> (<code>List[int]</code>, <em>optional</em>) — The second tokenized sequence.",name:"token_ids_1"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3543",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The token type ids.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[int]</code></p> | |
| `}}),Je=new w({props:{name:"decode",anchor:"transformers.PreTrainedTokenizerBase.decode",parameters:[{name:"token_ids",val:": Union"},{name:"skip_special_tokens",val:": bool = False"},{name:"clean_up_tokenization_spaces",val:": bool = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.decode.token_ids",description:`<strong>token_ids</strong> (<code>Union[int, List[int], np.ndarray, torch.Tensor, tf.Tensor]</code>) — | |
| List of tokenized input ids. Can be obtained using the <code>__call__</code> method.`,name:"token_ids"},{anchor:"transformers.PreTrainedTokenizerBase.decode.skip_special_tokens",description:`<strong>skip_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to remove special tokens in the decoding.`,name:"skip_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.decode.clean_up_tokenization_spaces",description:`<strong>clean_up_tokenization_spaces</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to clean up the tokenization spaces. If <code>None</code>, will default to | |
| <code>self.clean_up_tokenization_spaces</code>.`,name:"clean_up_tokenization_spaces"},{anchor:"transformers.PreTrainedTokenizerBase.decode.kwargs",description:`<strong>kwargs</strong> (additional keyword arguments, <em>optional</em>) — | |
| Will be passed to the underlying model specific decode method.`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3974",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The decoded sentence.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>str</code></p> | |
| `}}),Ne=new w({props:{name:"encode",anchor:"transformers.PreTrainedTokenizerBase.encode",parameters:[{name:"text",val:": Union"},{name:"text_pair",val:": Union = None"},{name:"add_special_tokens",val:": bool = True"},{name:"padding",val:": Union = False"},{name:"truncation",val:": Union = None"},{name:"max_length",val:": Optional = None"},{name:"stride",val:": int = 0"},{name:"padding_side",val:": Optional = None"},{name:"return_tensors",val:": Union = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.encode.text",description:`<strong>text</strong> (<code>str</code>, <code>List[str]</code> or <code>List[int]</code>) — | |
| The first sequence to be encoded. This can be a string, a list of strings (tokenized string using the | |
| <code>tokenize</code> method) or a list of integers (tokenized string ids using the <code>convert_tokens_to_ids</code> | |
| method).`,name:"text"},{anchor:"transformers.PreTrainedTokenizerBase.encode.text_pair",description:`<strong>text_pair</strong> (<code>str</code>, <code>List[str]</code> or <code>List[int]</code>, <em>optional</em>) — | |
| Optional second sequence to be encoded. This can be a string, a list of strings (tokenized string using | |
| the <code>tokenize</code> method) or a list of integers (tokenized string ids using the <code>convert_tokens_to_ids</code> | |
| method).`,name:"text_pair"},{anchor:"transformers.PreTrainedTokenizerBase.encode.add_special_tokens",description:`<strong>add_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to add special tokens when encoding the sequences. This will use the underlying | |
| <code>PretrainedTokenizerBase.build_inputs_with_special_tokens</code> function, which defines which tokens are | |
| automatically added to the input ids. This is usefull if you want to add <code>bos</code> or <code>eos</code> tokens | |
| automatically.`,name:"add_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.encode.padding",description:`<strong>padding</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.utils.PaddingStrategy">PaddingStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls padding. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest'</code>: Pad to the longest sequence in the batch (or no padding if only a single | |
| sequence if provided).</li> | |
| <li><code>'max_length'</code>: Pad to a maximum length specified with the argument <code>max_length</code> or to the maximum | |
| acceptable input length for the model if that argument is not provided.</li> | |
| <li><code>False</code> or <code>'do_not_pad'</code> (default): No padding (i.e., can output a batch with sequences of different | |
| lengths).</li> | |
| </ul>`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.encode.truncation",description:`<strong>truncation</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.tokenization_utils_base.TruncationStrategy">TruncationStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls truncation. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or | |
| to the maximum acceptable input length for the model if that argument is not provided. This will | |
| truncate token by token, removing a token from the longest sequence in the pair if a pair of | |
| sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_second'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>False</code> or <code>'do_not_truncate'</code> (default): No truncation (i.e., can output batch with sequence lengths | |
| greater than the model maximum admissible input size).</li> | |
| </ul>`,name:"truncation"},{anchor:"transformers.PreTrainedTokenizerBase.encode.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Controls the maximum length to use by one of the truncation/padding parameters.</p> | |
| <p>If left unset or set to <code>None</code>, this will use the predefined model maximum length if a maximum length | |
| is required by one of the truncation/padding parameters. If the model has no specific maximum input | |
| length (like XLNet) truncation/padding to a maximum length will be deactivated.`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.encode.stride",description:`<strong>stride</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| If set to a number along with <code>max_length</code>, the overflowing tokens returned when | |
| <code>return_overflowing_tokens=True</code> will contain some tokens from the end of the truncated sequence | |
| returned to provide some overlap between truncated and overflowing sequences. The value of this | |
| argument defines the number of overlapping tokens.`,name:"stride"},{anchor:"transformers.PreTrainedTokenizerBase.encode.is_split_into_words",description:`<strong>is_split_into_words</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the input is already pre-tokenized (e.g., split into words). If set to <code>True</code>, the | |
| tokenizer assumes the input is already split into words (for instance, by splitting it on whitespace) | |
| which it will tokenize. This is useful for NER or token classification.`,name:"is_split_into_words"},{anchor:"transformers.PreTrainedTokenizerBase.encode.pad_to_multiple_of",description:`<strong>pad_to_multiple_of</strong> (<code>int</code>, <em>optional</em>) — | |
| If set will pad the sequence to a multiple of the provided value. Requires <code>padding</code> to be activated. | |
| This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability | |
| <code>>= 7.5</code> (Volta).`,name:"pad_to_multiple_of"},{anchor:"transformers.PreTrainedTokenizerBase.encode.padding_side",description:`<strong>padding_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have padding applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"padding_side"},{anchor:"transformers.PreTrainedTokenizerBase.encode.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors instead of list of python integers. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.constant</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return Numpy <code>np.ndarray</code> objects.</li> | |
| </ul> | |
| <p>**kwargs — Passed along to the <code>.tokenize()</code> method.`,name:"return_tensors"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L2750",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The tokenized ids of the text.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[int]</code>, <code>torch.Tensor</code>, <code>tf.Tensor</code> or <code>np.ndarray</code></p> | |
| `}}),Fe=new w({props:{name:"encode_plus",anchor:"transformers.PreTrainedTokenizerBase.encode_plus",parameters:[{name:"text",val:": Union"},{name:"text_pair",val:": Union = None"},{name:"add_special_tokens",val:": bool = True"},{name:"padding",val:": Union = False"},{name:"truncation",val:": Union = None"},{name:"max_length",val:": Optional = None"},{name:"stride",val:": int = 0"},{name:"is_split_into_words",val:": bool = False"},{name:"pad_to_multiple_of",val:": Optional = None"},{name:"padding_side",val:": Optional = None"},{name:"return_tensors",val:": Union = None"},{name:"return_token_type_ids",val:": Optional = None"},{name:"return_attention_mask",val:": Optional = None"},{name:"return_overflowing_tokens",val:": bool = False"},{name:"return_special_tokens_mask",val:": bool = False"},{name:"return_offsets_mapping",val:": bool = False"},{name:"return_length",val:": bool = False"},{name:"verbose",val:": bool = True"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.text",description:`<strong>text</strong> (<code>str</code>, <code>List[str]</code> or (for non-fast tokenizers) <code>List[int]</code>) — | |
| The first sequence to be encoded. This can be a string, a list of strings (tokenized string using the | |
| <code>tokenize</code> method) or a list of integers (tokenized string ids using the <code>convert_tokens_to_ids</code> | |
| method).`,name:"text"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.text_pair",description:`<strong>text_pair</strong> (<code>str</code>, <code>List[str]</code> or <code>List[int]</code>, <em>optional</em>) — | |
| Optional second sequence to be encoded. This can be a string, a list of strings (tokenized string using | |
| the <code>tokenize</code> method) or a list of integers (tokenized string ids using the <code>convert_tokens_to_ids</code> | |
| method).`,name:"text_pair"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.add_special_tokens",description:`<strong>add_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to add special tokens when encoding the sequences. This will use the underlying | |
| <code>PretrainedTokenizerBase.build_inputs_with_special_tokens</code> function, which defines which tokens are | |
| automatically added to the input ids. This is usefull if you want to add <code>bos</code> or <code>eos</code> tokens | |
| automatically.`,name:"add_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.padding",description:`<strong>padding</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.utils.PaddingStrategy">PaddingStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls padding. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest'</code>: Pad to the longest sequence in the batch (or no padding if only a single | |
| sequence if provided).</li> | |
| <li><code>'max_length'</code>: Pad to a maximum length specified with the argument <code>max_length</code> or to the maximum | |
| acceptable input length for the model if that argument is not provided.</li> | |
| <li><code>False</code> or <code>'do_not_pad'</code> (default): No padding (i.e., can output a batch with sequences of different | |
| lengths).</li> | |
| </ul>`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.truncation",description:`<strong>truncation</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.tokenization_utils_base.TruncationStrategy">TruncationStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls truncation. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or | |
| to the maximum acceptable input length for the model if that argument is not provided. This will | |
| truncate token by token, removing a token from the longest sequence in the pair if a pair of | |
| sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_second'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>False</code> or <code>'do_not_truncate'</code> (default): No truncation (i.e., can output batch with sequence lengths | |
| greater than the model maximum admissible input size).</li> | |
| </ul>`,name:"truncation"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Controls the maximum length to use by one of the truncation/padding parameters.</p> | |
| <p>If left unset or set to <code>None</code>, this will use the predefined model maximum length if a maximum length | |
| is required by one of the truncation/padding parameters. If the model has no specific maximum input | |
| length (like XLNet) truncation/padding to a maximum length will be deactivated.`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.stride",description:`<strong>stride</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| If set to a number along with <code>max_length</code>, the overflowing tokens returned when | |
| <code>return_overflowing_tokens=True</code> will contain some tokens from the end of the truncated sequence | |
| returned to provide some overlap between truncated and overflowing sequences. The value of this | |
| argument defines the number of overlapping tokens.`,name:"stride"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.is_split_into_words",description:`<strong>is_split_into_words</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the input is already pre-tokenized (e.g., split into words). If set to <code>True</code>, the | |
| tokenizer assumes the input is already split into words (for instance, by splitting it on whitespace) | |
| which it will tokenize. This is useful for NER or token classification.`,name:"is_split_into_words"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.pad_to_multiple_of",description:`<strong>pad_to_multiple_of</strong> (<code>int</code>, <em>optional</em>) — | |
| If set will pad the sequence to a multiple of the provided value. Requires <code>padding</code> to be activated. | |
| This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability | |
| <code>>= 7.5</code> (Volta).`,name:"pad_to_multiple_of"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.padding_side",description:`<strong>padding_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have padding applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"padding_side"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors instead of list of python integers. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.constant</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return Numpy <code>np.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.return_token_type_ids",description:`<strong>return_token_type_ids</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return token type IDs. If left to the default, will return the token type IDs according to | |
| the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a>`,name:"return_token_type_ids"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.return_attention_mask",description:`<strong>return_attention_mask</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return the attention mask. If left to the default, will return the attention mask according | |
| to the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"return_attention_mask"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.return_overflowing_tokens",description:`<strong>return_overflowing_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return overflowing token sequences. If a pair of sequences of input ids (or a batch | |
| of pairs) is provided with <code>truncation_strategy = longest_first</code> or <code>True</code>, an error is raised instead | |
| of returning overflowing tokens.`,name:"return_overflowing_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.return_special_tokens_mask",description:`<strong>return_special_tokens_mask</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return special tokens mask information.`,name:"return_special_tokens_mask"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.return_offsets_mapping",description:`<strong>return_offsets_mapping</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return <code>(char_start, char_end)</code> for each token.</p> | |
| <p>This is only available on fast tokenizers inheriting from <code>PreTrainedTokenizerFast</code>, if using | |
| Python’s tokenizer, this method will raise <code>NotImplementedError</code>.`,name:"return_offsets_mapping"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.return_length",description:`<strong>return_length</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return the lengths of the encoded inputs.`,name:"return_length"},{anchor:"transformers.PreTrainedTokenizerBase.encode_plus.verbose",description:`<strong>verbose</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to print more information and warnings. | |
| **kwargs — passed to the <code>self.tokenize()</code> method`,name:"verbose"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3154",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>BatchEncoding</code> with the following fields:</p> | |
| <ul> | |
| <li> | |
| <p><strong>input_ids</strong> — List of token ids to be fed to a model.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>token_type_ids</strong> — List of token type ids to be fed to a model (when <code>return_token_type_ids=True</code> or | |
| if <em>“token_type_ids”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>attention_mask</strong> — List of indices specifying which tokens should be attended to by the model (when | |
| <code>return_attention_mask=True</code> or if <em>“attention_mask”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>overflowing_tokens</strong> — List of overflowing tokens sequences (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>num_truncated_tokens</strong> — Number of tokens truncated (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>special_tokens_mask</strong> — List of 0s and 1s, with 1 specifying added special tokens and 0 specifying | |
| regular sequence tokens (when <code>add_special_tokens=True</code> and <code>return_special_tokens_mask=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>length</strong> — The length of the inputs (when <code>return_length=True</code>)</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>BatchEncoding</code></p> | |
| `}}),de=new wo({props:{warning:!0,$$slots:{default:[Vs]},$$scope:{ctx:B}}}),Ze=new w({props:{name:"from_pretrained",anchor:"transformers.PreTrainedTokenizerBase.from_pretrained",parameters:[{name:"pretrained_model_name_or_path",val:": Union"},{name:"*init_inputs",val:""},{name:"cache_dir",val:": Union = None"},{name:"force_download",val:": bool = False"},{name:"local_files_only",val:": bool = False"},{name:"token",val:": Union = None"},{name:"revision",val:": str = 'main'"},{name:"trust_remote_code",val:" = False"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.pretrained_model_name_or_path",description:`<strong>pretrained_model_name_or_path</strong> (<code>str</code> or <code>os.PathLike</code>) — | |
| Can be either:</p> | |
| <ul> | |
| <li>A string, the <em>model id</em> of a predefined tokenizer hosted inside a model repo on huggingface.co.</li> | |
| <li>A path to a <em>directory</em> containing vocabulary files required by the tokenizer, for instance saved | |
| using the <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.save_pretrained">save_pretrained()</a> method, e.g., | |
| <code>./my_model_directory/</code>.</li> | |
| <li>(<strong>Deprecated</strong>, not applicable to all derived classes) A path or url to a single saved vocabulary | |
| file (if and only if the tokenizer only requires a single vocabulary file like Bert or XLNet), e.g., | |
| <code>./my_model_directory/vocab.txt</code>.</li> | |
| </ul>`,name:"pretrained_model_name_or_path"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.cache_dir",description:`<strong>cache_dir</strong> (<code>str</code> or <code>os.PathLike</code>, <em>optional</em>) — | |
| Path to a directory in which a downloaded predefined tokenizer vocabulary files should be cached if the | |
| standard cache should not be used.`,name:"cache_dir"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.force_download",description:`<strong>force_download</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to force the (re-)download the vocabulary files and override the cached versions if they | |
| exist. | |
| resume_download — | |
| Deprecated and ignored. All downloads are now resumed by default when possible. | |
| Will be removed in v5 of Transformers.`,name:"force_download"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.proxies",description:`<strong>proxies</strong> (<code>Dict[str, str]</code>, <em>optional</em>) — | |
| A dictionary of proxy servers to use by protocol or endpoint, e.g., <code>{'http': 'foo.bar:3128', 'http://hostname': 'foo.bar:4012'}</code>. The proxies are used on each request.`,name:"proxies"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.token",description:`<strong>token</strong> (<code>str</code> or <em>bool</em>, <em>optional</em>) — | |
| The token to use as HTTP bearer authorization for remote files. If <code>True</code>, will use the token generated | |
| when running <code>huggingface-cli login</code> (stored in <code>~/.huggingface</code>).`,name:"token"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.local_files_only",description:`<strong>local_files_only</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to only rely on local files and not to attempt to download any files.`,name:"local_files_only"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.revision",description:`<strong>revision</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"main"</code>) — | |
| The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a | |
| git-based system for storing models and other artifacts on huggingface.co, so <code>revision</code> can be any | |
| identifier allowed by git.`,name:"revision"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.subfolder",description:`<strong>subfolder</strong> (<code>str</code>, <em>optional</em>) — | |
| In case the relevant files are located inside a subfolder of the model repo on huggingface.co (e.g. for | |
| facebook/rag-token-base), specify it here.`,name:"subfolder"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.inputs",description:`<strong>inputs</strong> (additional positional arguments, <em>optional</em>) — | |
| Will be passed along to the Tokenizer <code>__init__</code> method.`,name:"inputs"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.trust_remote_code",description:`<strong>trust_remote_code</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to allow for custom models defined on the Hub in their own modeling files. This option | |
| should only be set to <code>True</code> for repositories you trust and in which you have read the code, as it will | |
| execute code present on the Hub on your local machine.`,name:"trust_remote_code"},{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.kwargs",description:`<strong>kwargs</strong> (additional keyword arguments, <em>optional</em>) — | |
| Will be passed to the Tokenizer <code>__init__</code> method. Can be used to set special tokens like <code>bos_token</code>, | |
| <code>eos_token</code>, <code>unk_token</code>, <code>sep_token</code>, <code>pad_token</code>, <code>cls_token</code>, <code>mask_token</code>, | |
| <code>additional_special_tokens</code>. See parameters in the <code>__init__</code> for more details.`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L1976"}}),le=new wo({props:{$$slots:{default:[Ds]},$$scope:{ctx:B}}}),ce=new Go({props:{anchor:"transformers.PreTrainedTokenizerBase.from_pretrained.example",$$slots:{default:[Ss]},$$scope:{ctx:B}}}),Ve=new w({props:{name:"get_chat_template",anchor:"transformers.PreTrainedTokenizerBase.get_chat_template",parameters:[{name:"chat_template",val:": Optional = None"},{name:"tools",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.get_chat_template.chat_template",description:`<strong>chat_template</strong> (<code>str</code>, <em>optional</em>) — | |
| A Jinja template or the name of a template to use for this conversion. | |
| It is usually not necessary to pass anything to this argument, | |
| as the model’s template will be used by default.`,name:"chat_template"},{anchor:"transformers.PreTrainedTokenizerBase.get_chat_template.tools",description:`<strong>tools</strong> (<code>List[Dict]</code>, <em>optional</em>) — | |
| A list of tools (callable functions) that will be accessible to the model. If the template does not | |
| support function calling, this argument will have no effect. Each tool should be passed as a JSON Schema, | |
| giving the name, description and argument types for the tool. See our | |
| <a href="https://huggingface.co/docs/transformers/main/en/chat_templating#automated-function-conversion-for-tool-use" rel="nofollow">chat templating guide</a> | |
| for more information.`,name:"tools"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L1922",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The chat template string.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>str</code></p> | |
| `}}),De=new w({props:{name:"get_special_tokens_mask",anchor:"transformers.PreTrainedTokenizerBase.get_special_tokens_mask",parameters:[{name:"token_ids_0",val:": List"},{name:"token_ids_1",val:": Optional = None"},{name:"already_has_special_tokens",val:": bool = False"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.get_special_tokens_mask.token_ids_0",description:`<strong>token_ids_0</strong> (<code>List[int]</code>) — | |
| List of ids of the first sequence.`,name:"token_ids_0"},{anchor:"transformers.PreTrainedTokenizerBase.get_special_tokens_mask.token_ids_1",description:`<strong>token_ids_1</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| List of ids of the second sequence.`,name:"token_ids_1"},{anchor:"transformers.PreTrainedTokenizerBase.get_special_tokens_mask.already_has_special_tokens",description:`<strong>already_has_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the token list is already formatted with special tokens for the model.`,name:"already_has_special_tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L4020",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>1 for a special token, 0 for a sequence token.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of integers in the range [0, 1]</p> | |
| `}}),Se=new w({props:{name:"get_vocab",anchor:"transformers.PreTrainedTokenizerBase.get_vocab",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L1697",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The vocabulary.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>Dict[str, int]</code></p> | |
| `}}),Re=new w({props:{name:"pad",anchor:"transformers.PreTrainedTokenizerBase.pad",parameters:[{name:"encoded_inputs",val:": Union"},{name:"padding",val:": Union = True"},{name:"max_length",val:": Optional = None"},{name:"pad_to_multiple_of",val:": Optional = None"},{name:"padding_side",val:": Optional = None"},{name:"return_attention_mask",val:": Optional = None"},{name:"return_tensors",val:": Union = None"},{name:"verbose",val:": bool = True"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.pad.encoded_inputs",description:`<strong>encoded_inputs</strong> (<code>BatchEncoding</code>, list of <code>BatchEncoding</code>, <code>Dict[str, List[int]]</code>, <code>Dict[str, List[List[int]]</code> or <code>List[Dict[str, List[int]]]</code>) — | |
| Tokenized inputs. Can represent one input (<code>BatchEncoding</code> or <code>Dict[str, List[int]]</code>) or a batch of | |
| tokenized inputs (list of <code>BatchEncoding</code>, <em>Dict[str, List[List[int]]]</em> or <em>List[Dict[str, | |
| List[int]]]</em>) so you can use this method during preprocessing as well as in a PyTorch Dataloader | |
| collate function.</p> | |
| <p>Instead of <code>List[int]</code> you can have tensors (numpy arrays, PyTorch tensors or TensorFlow tensors), see | |
| the note above for the return type.`,name:"encoded_inputs"},{anchor:"transformers.PreTrainedTokenizerBase.pad.padding",description:`<strong>padding</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.utils.PaddingStrategy">PaddingStrategy</a>, <em>optional</em>, defaults to <code>True</code>) — | |
| Select a strategy to pad the returned sequences (according to the model’s padding side and padding | |
| index) among:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest'</code>: Pad to the longest sequence in the batch (or no padding if only a single | |
| sequence if provided).</li> | |
| <li><code>'max_length'</code>: Pad to a maximum length specified with the argument <code>max_length</code> or to the maximum | |
| acceptable input length for the model if that argument is not provided.</li> | |
| <li><code>False</code> or <code>'do_not_pad'</code> (default): No padding (i.e., can output a batch with sequences of different | |
| lengths).</li> | |
| </ul>`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.pad.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Maximum length of the returned list and optionally padding length (see above).`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.pad.pad_to_multiple_of",description:`<strong>pad_to_multiple_of</strong> (<code>int</code>, <em>optional</em>) — | |
| If set will pad the sequence to a multiple of the provided value.</p> | |
| <p>This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability | |
| <code>>= 7.5</code> (Volta).`,name:"pad_to_multiple_of"},{anchor:"transformers.PreTrainedTokenizerBase.pad.padding_side",description:`<strong>padding_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have padding applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"padding_side"},{anchor:"transformers.PreTrainedTokenizerBase.pad.return_attention_mask",description:`<strong>return_attention_mask</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return the attention mask. If left to the default, will return the attention mask according | |
| to the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"return_attention_mask"},{anchor:"transformers.PreTrainedTokenizerBase.pad.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors instead of list of python integers. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.constant</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return Numpy <code>np.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.PreTrainedTokenizerBase.pad.verbose",description:`<strong>verbose</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to print more information and warnings.`,name:"verbose"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3364"}}),ue=new wo({props:{$$slots:{default:[Rs]},$$scope:{ctx:B}}}),Ae=new w({props:{name:"prepare_for_model",anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model",parameters:[{name:"ids",val:": List"},{name:"pair_ids",val:": Optional = None"},{name:"add_special_tokens",val:": bool = True"},{name:"padding",val:": Union = False"},{name:"truncation",val:": Union = None"},{name:"max_length",val:": Optional = None"},{name:"stride",val:": int = 0"},{name:"pad_to_multiple_of",val:": Optional = None"},{name:"padding_side",val:": Optional = None"},{name:"return_tensors",val:": Union = None"},{name:"return_token_type_ids",val:": Optional = None"},{name:"return_attention_mask",val:": Optional = None"},{name:"return_overflowing_tokens",val:": bool = False"},{name:"return_special_tokens_mask",val:": bool = False"},{name:"return_offsets_mapping",val:": bool = False"},{name:"return_length",val:": bool = False"},{name:"verbose",val:": bool = True"},{name:"prepend_batch_axis",val:": bool = False"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.ids",description:`<strong>ids</strong> (<code>List[int]</code>) — | |
| Tokenized input ids of the first sequence. Can be obtained from a string by chaining the <code>tokenize</code> and | |
| <code>convert_tokens_to_ids</code> methods.`,name:"ids"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.pair_ids",description:`<strong>pair_ids</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| Tokenized input ids of the second sequence. Can be obtained from a string by chaining the <code>tokenize</code> | |
| and <code>convert_tokens_to_ids</code> methods.`,name:"pair_ids"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.add_special_tokens",description:`<strong>add_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to add special tokens when encoding the sequences. This will use the underlying | |
| <code>PretrainedTokenizerBase.build_inputs_with_special_tokens</code> function, which defines which tokens are | |
| automatically added to the input ids. This is usefull if you want to add <code>bos</code> or <code>eos</code> tokens | |
| automatically.`,name:"add_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.padding",description:`<strong>padding</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.utils.PaddingStrategy">PaddingStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls padding. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest'</code>: Pad to the longest sequence in the batch (or no padding if only a single | |
| sequence if provided).</li> | |
| <li><code>'max_length'</code>: Pad to a maximum length specified with the argument <code>max_length</code> or to the maximum | |
| acceptable input length for the model if that argument is not provided.</li> | |
| <li><code>False</code> or <code>'do_not_pad'</code> (default): No padding (i.e., can output a batch with sequences of different | |
| lengths).</li> | |
| </ul>`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.truncation",description:`<strong>truncation</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.tokenization_utils_base.TruncationStrategy">TruncationStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls truncation. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or | |
| to the maximum acceptable input length for the model if that argument is not provided. This will | |
| truncate token by token, removing a token from the longest sequence in the pair if a pair of | |
| sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_second'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>False</code> or <code>'do_not_truncate'</code> (default): No truncation (i.e., can output batch with sequence lengths | |
| greater than the model maximum admissible input size).</li> | |
| </ul>`,name:"truncation"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Controls the maximum length to use by one of the truncation/padding parameters.</p> | |
| <p>If left unset or set to <code>None</code>, this will use the predefined model maximum length if a maximum length | |
| is required by one of the truncation/padding parameters. If the model has no specific maximum input | |
| length (like XLNet) truncation/padding to a maximum length will be deactivated.`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.stride",description:`<strong>stride</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| If set to a number along with <code>max_length</code>, the overflowing tokens returned when | |
| <code>return_overflowing_tokens=True</code> will contain some tokens from the end of the truncated sequence | |
| returned to provide some overlap between truncated and overflowing sequences. The value of this | |
| argument defines the number of overlapping tokens.`,name:"stride"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.is_split_into_words",description:`<strong>is_split_into_words</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the input is already pre-tokenized (e.g., split into words). If set to <code>True</code>, the | |
| tokenizer assumes the input is already split into words (for instance, by splitting it on whitespace) | |
| which it will tokenize. This is useful for NER or token classification.`,name:"is_split_into_words"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.pad_to_multiple_of",description:`<strong>pad_to_multiple_of</strong> (<code>int</code>, <em>optional</em>) — | |
| If set will pad the sequence to a multiple of the provided value. Requires <code>padding</code> to be activated. | |
| This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability | |
| <code>>= 7.5</code> (Volta).`,name:"pad_to_multiple_of"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.padding_side",description:`<strong>padding_side</strong> (<code>str</code>, <em>optional</em>) — | |
| The side on which the model should have padding applied. Should be selected between [‘right’, ‘left’]. | |
| Default value is picked from the class attribute of the same name.`,name:"padding_side"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors instead of list of python integers. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.constant</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return Numpy <code>np.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.return_token_type_ids",description:`<strong>return_token_type_ids</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return token type IDs. If left to the default, will return the token type IDs according to | |
| the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a>`,name:"return_token_type_ids"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.return_attention_mask",description:`<strong>return_attention_mask</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to return the attention mask. If left to the default, will return the attention mask according | |
| to the specific tokenizer’s default, defined by the <code>return_outputs</code> attribute.</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"return_attention_mask"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.return_overflowing_tokens",description:`<strong>return_overflowing_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return overflowing token sequences. If a pair of sequences of input ids (or a batch | |
| of pairs) is provided with <code>truncation_strategy = longest_first</code> or <code>True</code>, an error is raised instead | |
| of returning overflowing tokens.`,name:"return_overflowing_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.return_special_tokens_mask",description:`<strong>return_special_tokens_mask</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return special tokens mask information.`,name:"return_special_tokens_mask"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.return_offsets_mapping",description:`<strong>return_offsets_mapping</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return <code>(char_start, char_end)</code> for each token.</p> | |
| <p>This is only available on fast tokenizers inheriting from <code>PreTrainedTokenizerFast</code>, if using | |
| Python’s tokenizer, this method will raise <code>NotImplementedError</code>.`,name:"return_offsets_mapping"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.return_length",description:`<strong>return_length</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to return the lengths of the encoded inputs.`,name:"return_length"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_for_model.verbose",description:`<strong>verbose</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to print more information and warnings. | |
| **kwargs — passed to the <code>self.tokenize()</code> method`,name:"verbose"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3583",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>BatchEncoding</code> with the following fields:</p> | |
| <ul> | |
| <li> | |
| <p><strong>input_ids</strong> — List of token ids to be fed to a model.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>token_type_ids</strong> — List of token type ids to be fed to a model (when <code>return_token_type_ids=True</code> or | |
| if <em>“token_type_ids”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#token-type-ids">What are token type IDs?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>attention_mask</strong> — List of indices specifying which tokens should be attended to by the model (when | |
| <code>return_attention_mask=True</code> or if <em>“attention_mask”</em> is in <code>self.model_input_names</code>).</p> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| </li> | |
| <li> | |
| <p><strong>overflowing_tokens</strong> — List of overflowing tokens sequences (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>num_truncated_tokens</strong> — Number of tokens truncated (when a <code>max_length</code> is specified and | |
| <code>return_overflowing_tokens=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>special_tokens_mask</strong> — List of 0s and 1s, with 1 specifying added special tokens and 0 specifying | |
| regular sequence tokens (when <code>add_special_tokens=True</code> and <code>return_special_tokens_mask=True</code>).</p> | |
| </li> | |
| <li> | |
| <p><strong>length</strong> — The length of the inputs (when <code>return_length=True</code>)</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>BatchEncoding</code></p> | |
| `}}),Ee=new w({props:{name:"prepare_seq2seq_batch",anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch",parameters:[{name:"src_texts",val:": List"},{name:"tgt_texts",val:": Optional = None"},{name:"max_length",val:": Optional = None"},{name:"max_target_length",val:": Optional = None"},{name:"padding",val:": str = 'longest'"},{name:"return_tensors",val:": str = None"},{name:"truncation",val:": bool = True"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch.src_texts",description:`<strong>src_texts</strong> (<code>List[str]</code>) — | |
| List of documents to summarize or source language texts.`,name:"src_texts"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch.tgt_texts",description:`<strong>tgt_texts</strong> (<code>list</code>, <em>optional</em>) — | |
| List of summaries or target language texts.`,name:"tgt_texts"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Controls the maximum length for encoder inputs (documents to summarize or source language texts) If | |
| left unset or set to <code>None</code>, this will use the predefined model maximum length if a maximum length is | |
| required by one of the truncation/padding parameters. If the model has no specific maximum input length | |
| (like XLNet) truncation/padding to a maximum length will be deactivated.`,name:"max_length"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch.max_target_length",description:`<strong>max_target_length</strong> (<code>int</code>, <em>optional</em>) — | |
| Controls the maximum length of decoder inputs (target language texts or summaries) If left unset or set | |
| to <code>None</code>, this will use the max_length value.`,name:"max_target_length"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch.padding",description:`<strong>padding</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.utils.PaddingStrategy">PaddingStrategy</a>, <em>optional</em>, defaults to <code>False</code>) — | |
| Activates and controls padding. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest'</code>: Pad to the longest sequence in the batch (or no padding if only a single | |
| sequence if provided).</li> | |
| <li><code>'max_length'</code>: Pad to a maximum length specified with the argument <code>max_length</code> or to the maximum | |
| acceptable input length for the model if that argument is not provided.</li> | |
| <li><code>False</code> or <code>'do_not_pad'</code> (default): No padding (i.e., can output a batch with sequences of different | |
| lengths).</li> | |
| </ul>`,name:"padding"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors instead of list of python integers. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'tf'</code>: Return TensorFlow <code>tf.constant</code> objects.</li> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return Numpy <code>np.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.PreTrainedTokenizerBase.prepare_seq2seq_batch.truncation",description:`<strong>truncation</strong> (<code>bool</code>, <code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.tokenization_utils_base.TruncationStrategy">TruncationStrategy</a>, <em>optional</em>, defaults to <code>True</code>) — | |
| Activates and controls truncation. Accepts the following values:</p> | |
| <ul> | |
| <li><code>True</code> or <code>'longest_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or | |
| to the maximum acceptable input length for the model if that argument is not provided. This will | |
| truncate token by token, removing a token from the longest sequence in the pair if a pair of | |
| sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_second'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>False</code> or <code>'do_not_truncate'</code> (default): No truncation (i.e., can output batch with sequence lengths | |
| greater than the model maximum admissible input size). | |
| **kwargs — | |
| Additional keyword arguments passed along to <code>self.__call__</code>.</li> | |
| </ul>`,name:"truncation"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L4151",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>BatchEncoding</code> with the following fields:</p> | |
| <ul> | |
| <li><strong>input_ids</strong> — List of token ids to be fed to the encoder.</li> | |
| <li><strong>attention_mask</strong> — List of indices specifying which tokens should be attended to by the model.</li> | |
| <li><strong>labels</strong> — List of token ids for tgt_texts.</li> | |
| </ul> | |
| <p>The full set of keys <code>[input_ids, attention_mask, labels]</code>, will only be returned if tgt_texts is passed. | |
| Otherwise, input_ids, attention_mask will be the only keys.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>BatchEncoding</code></p> | |
| `}}),Ge=new w({props:{name:"push_to_hub",anchor:"transformers.PreTrainedTokenizerBase.push_to_hub",parameters:[{name:"repo_id",val:": str"},{name:"use_temp_dir",val:": Optional = None"},{name:"commit_message",val:": Optional = None"},{name:"private",val:": Optional = None"},{name:"token",val:": Union = None"},{name:"max_shard_size",val:": Union = '5GB'"},{name:"create_pr",val:": bool = False"},{name:"safe_serialization",val:": bool = True"},{name:"revision",val:": str = None"},{name:"commit_description",val:": str = None"},{name:"tags",val:": Optional = None"},{name:"**deprecated_kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.repo_id",description:`<strong>repo_id</strong> (<code>str</code>) — | |
| The name of the repository you want to push your tokenizer to. It should contain your organization name | |
| when pushing to a given organization.`,name:"repo_id"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.use_temp_dir",description:`<strong>use_temp_dir</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to use a temporary directory to store the files saved before they are pushed to the Hub. | |
| Will default to <code>True</code> if there is no directory named like <code>repo_id</code>, <code>False</code> otherwise.`,name:"use_temp_dir"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.commit_message",description:`<strong>commit_message</strong> (<code>str</code>, <em>optional</em>) — | |
| Message to commit while pushing. Will default to <code>"Upload tokenizer"</code>.`,name:"commit_message"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.private",description:`<strong>private</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not the repository created should be private.`,name:"private"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.token",description:`<strong>token</strong> (<code>bool</code> or <code>str</code>, <em>optional</em>) — | |
| The token to use as HTTP bearer authorization for remote files. If <code>True</code>, will use the token generated | |
| when running <code>huggingface-cli login</code> (stored in <code>~/.huggingface</code>). Will default to <code>True</code> if <code>repo_url</code> | |
| is not specified.`,name:"token"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.max_shard_size",description:`<strong>max_shard_size</strong> (<code>int</code> or <code>str</code>, <em>optional</em>, defaults to <code>"5GB"</code>) — | |
| Only applicable for models. The maximum size for a checkpoint before being sharded. Checkpoints shard | |
| will then be each of size lower than this size. If expressed as a string, needs to be digits followed | |
| by a unit (like <code>"5MB"</code>). We default it to <code>"5GB"</code> so that users can easily load models on free-tier | |
| Google Colab instances without any CPU OOM issues.`,name:"max_shard_size"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.create_pr",description:`<strong>create_pr</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to create a PR with the uploaded files or directly commit.`,name:"create_pr"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.safe_serialization",description:`<strong>safe_serialization</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to convert the model weights in safetensors format for safer serialization.`,name:"safe_serialization"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.revision",description:`<strong>revision</strong> (<code>str</code>, <em>optional</em>) — | |
| Branch to push the uploaded files to.`,name:"revision"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.commit_description",description:`<strong>commit_description</strong> (<code>str</code>, <em>optional</em>) — | |
| The description of the commit that will be created`,name:"commit_description"},{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.tags",description:`<strong>tags</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| List of tags to push on the Hub.`,name:"tags"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/utils/hub.py#L828"}}),_e=new Go({props:{anchor:"transformers.PreTrainedTokenizerBase.push_to_hub.example",$$slots:{default:[As]},$$scope:{ctx:B}}}),He=new w({props:{name:"register_for_auto_class",anchor:"transformers.PreTrainedTokenizerBase.register_for_auto_class",parameters:[{name:"auto_class",val:" = 'AutoTokenizer'"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.register_for_auto_class.auto_class",description:`<strong>auto_class</strong> (<code>str</code> or <code>type</code>, <em>optional</em>, defaults to <code>"AutoTokenizer"</code>) — | |
| The auto class to register this new tokenizer with.`,name:"auto_class"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L4125"}}),ge=new wo({props:{warning:!0,$$slots:{default:[Es]},$$scope:{ctx:B}}}),Xe=new w({props:{name:"save_pretrained",anchor:"transformers.PreTrainedTokenizerBase.save_pretrained",parameters:[{name:"save_directory",val:": Union"},{name:"legacy_format",val:": Optional = None"},{name:"filename_prefix",val:": Optional = None"},{name:"push_to_hub",val:": bool = False"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.save_pretrained.save_directory",description:"<strong>save_directory</strong> (<code>str</code> or <code>os.PathLike</code>) — The path to a directory where the tokenizer will be saved.",name:"save_directory"},{anchor:"transformers.PreTrainedTokenizerBase.save_pretrained.legacy_format",description:`<strong>legacy_format</strong> (<code>bool</code>, <em>optional</em>) — | |
| Only applicable for a fast tokenizer. If unset (default), will save the tokenizer in the unified JSON | |
| format as well as in legacy format if it exists, i.e. with tokenizer specific vocabulary and a separate | |
| added_tokens files.</p> | |
| <p>If <code>False</code>, will only save the tokenizer in the unified JSON format. This format is incompatible with | |
| “slow” tokenizers (not powered by the <em>tokenizers</em> library), so the tokenizer will not be able to be | |
| loaded in the corresponding “slow” tokenizer.</p> | |
| <p>If <code>True</code>, will save the tokenizer in legacy format. If the “slow” tokenizer doesn’t exits, a value | |
| error is raised.`,name:"legacy_format"},{anchor:"transformers.PreTrainedTokenizerBase.save_pretrained.filename_prefix",description:`<strong>filename_prefix</strong> (<code>str</code>, <em>optional</em>) — | |
| A prefix to add to the names of the files saved by the tokenizer.`,name:"filename_prefix"},{anchor:"transformers.PreTrainedTokenizerBase.save_pretrained.push_to_hub",description:`<strong>push_to_hub</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to push your model to the Hugging Face model hub after saving it. You can specify the | |
| repository you want to push to with <code>repo_id</code> (will default to the name of <code>save_directory</code> in your | |
| namespace).`,name:"push_to_hub"},{anchor:"transformers.PreTrainedTokenizerBase.save_pretrained.kwargs",description:`<strong>kwargs</strong> (<code>Dict[str, Any]</code>, <em>optional</em>) — | |
| Additional key word arguments passed along to the <a href="/docs/transformers/pr_34323/ko/main_classes/model#transformers.utils.PushToHubMixin.push_to_hub">push_to_hub()</a> method.`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L2508",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The files saved.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A tuple of <code>str</code></p> | |
| `}}),Ye=new w({props:{name:"save_vocabulary",anchor:"transformers.PreTrainedTokenizerBase.save_vocabulary",parameters:[{name:"save_directory",val:": str"},{name:"filename_prefix",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.save_vocabulary.save_directory",description:`<strong>save_directory</strong> (<code>str</code>) — | |
| The directory in which to save the vocabulary.`,name:"save_directory"},{anchor:"transformers.PreTrainedTokenizerBase.save_vocabulary.filename_prefix",description:`<strong>filename_prefix</strong> (<code>str</code>, <em>optional</em>) — | |
| An optional prefix to add to the named of the saved files.`,name:"filename_prefix"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L2712",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Paths to the files saved.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>Tuple(str)</code></p> | |
| `}}),Oe=new w({props:{name:"tokenize",anchor:"transformers.PreTrainedTokenizerBase.tokenize",parameters:[{name:"text",val:": str"},{name:"pair",val:": Optional = None"},{name:"add_special_tokens",val:": bool = False"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.tokenize.text",description:`<strong>text</strong> (<code>str</code>) — | |
| The sequence to be encoded.`,name:"text"},{anchor:"transformers.PreTrainedTokenizerBase.tokenize.pair",description:`<strong>pair</strong> (<code>str</code>, <em>optional</em>) — | |
| A second sequence to be encoded with the first.`,name:"pair"},{anchor:"transformers.PreTrainedTokenizerBase.tokenize.add_special_tokens",description:`<strong>add_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to add the special tokens associated with the corresponding model.`,name:"add_special_tokens"},{anchor:"transformers.PreTrainedTokenizerBase.tokenize.kwargs",description:`<strong>kwargs</strong> (additional keyword arguments, <em>optional</em>) — | |
| Will be passed to the underlying model specific encode method. See details in | |
| <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__"><strong>call</strong>()</a>`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L2730",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The list of tokens.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[str]</code></p> | |
| `}}),Qe=new w({props:{name:"truncate_sequences",anchor:"transformers.PreTrainedTokenizerBase.truncate_sequences",parameters:[{name:"ids",val:": List"},{name:"pair_ids",val:": Optional = None"},{name:"num_tokens_to_remove",val:": int = 0"},{name:"truncation_strategy",val:": Union = 'longest_first'"},{name:"stride",val:": int = 0"}],parametersDescription:[{anchor:"transformers.PreTrainedTokenizerBase.truncate_sequences.ids",description:`<strong>ids</strong> (<code>List[int]</code>) — | |
| Tokenized input ids of the first sequence. Can be obtained from a string by chaining the <code>tokenize</code> and | |
| <code>convert_tokens_to_ids</code> methods.`,name:"ids"},{anchor:"transformers.PreTrainedTokenizerBase.truncate_sequences.pair_ids",description:`<strong>pair_ids</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| Tokenized input ids of the second sequence. Can be obtained from a string by chaining the <code>tokenize</code> | |
| and <code>convert_tokens_to_ids</code> methods.`,name:"pair_ids"},{anchor:"transformers.PreTrainedTokenizerBase.truncate_sequences.num_tokens_to_remove",description:`<strong>num_tokens_to_remove</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of tokens to remove using the truncation strategy.`,name:"num_tokens_to_remove"},{anchor:"transformers.PreTrainedTokenizerBase.truncate_sequences.truncation_strategy",description:`<strong>truncation_strategy</strong> (<code>str</code> or <a href="/docs/transformers/pr_34323/ko/internal/tokenization_utils#transformers.tokenization_utils_base.TruncationStrategy">TruncationStrategy</a>, <em>optional</em>, defaults to <code>'longest_first'</code>) — | |
| The strategy to follow for truncation. Can be:</p> | |
| <ul> | |
| <li><code>'longest_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will truncate | |
| token by token, removing a token from the longest sequence in the pair if a pair of sequences (or a | |
| batch of pairs) is provided.</li> | |
| <li><code>'only_first'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'only_second'</code>: Truncate to a maximum length specified with the argument <code>max_length</code> or to the | |
| maximum acceptable input length for the model if that argument is not provided. This will only | |
| truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.</li> | |
| <li><code>'do_not_truncate'</code> (default): No truncation (i.e., can output batch with sequence lengths greater | |
| than the model maximum admissible input size).</li> | |
| </ul>`,name:"truncation_strategy"},{anchor:"transformers.PreTrainedTokenizerBase.truncate_sequences.stride",description:`<strong>stride</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| If set to a positive number, the overflowing tokens returned will contain some tokens from the main | |
| sequence returned. The value of this argument defines the number of additional tokens.`,name:"stride"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L3721",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The truncated <code>ids</code>, the truncated <code>pair_ids</code> and the list of | |
| overflowing tokens. Note: The <em>longest_first</em> strategy returns empty list of overflowing tokens if a pair | |
| of sequences (or a batch of pairs) is provided.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>Tuple[List[int], List[int], List[int]]</code></p> | |
| `}}),Ke=new Ho({props:{title:"SpecialTokensMixin",local:"transformers.SpecialTokensMixin ][ transformers.SpecialTokensMixin",headingTag:"h2"}}),et=new w({props:{name:"class transformers.SpecialTokensMixin",anchor:"transformers.SpecialTokensMixin",parameters:[{name:"verbose",val:" = False"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.SpecialTokensMixin.bos_token",description:`<strong>bos_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing the beginning of a sentence.`,name:"bos_token"},{anchor:"transformers.SpecialTokensMixin.eos_token",description:`<strong>eos_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing the end of a sentence.`,name:"eos_token"},{anchor:"transformers.SpecialTokensMixin.unk_token",description:`<strong>unk_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing an out-of-vocabulary token.`,name:"unk_token"},{anchor:"transformers.SpecialTokensMixin.sep_token",description:`<strong>sep_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token separating two different sentences in the same input (used by BERT for instance).`,name:"sep_token"},{anchor:"transformers.SpecialTokensMixin.pad_token",description:`<strong>pad_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token used to make arrays of tokens the same size for batching purpose. Will then be ignored by | |
| attention mechanisms or loss computation.`,name:"pad_token"},{anchor:"transformers.SpecialTokensMixin.cls_token",description:`<strong>cls_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing the class of the input (used by BERT for instance).`,name:"cls_token"},{anchor:"transformers.SpecialTokensMixin.mask_token",description:`<strong>mask_token</strong> (<code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A special token representing a masked token (used by masked-language modeling pretraining objectives, like | |
| BERT).`,name:"mask_token"},{anchor:"transformers.SpecialTokensMixin.additional_special_tokens",description:`<strong>additional_special_tokens</strong> (tuple or list of <code>str</code> or <code>tokenizers.AddedToken</code>, <em>optional</em>) — | |
| A tuple or a list of additional tokens, which will be marked as <code>special</code>, meaning that they will be | |
| skipped when decoding if <code>skip_special_tokens</code> is set to <code>True</code>.`,name:"additional_special_tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L824"}}),tt=new w({props:{name:"add_special_tokens",anchor:"transformers.SpecialTokensMixin.add_special_tokens",parameters:[{name:"special_tokens_dict",val:": Dict"},{name:"replace_additional_special_tokens",val:" = True"}],parametersDescription:[{anchor:"transformers.SpecialTokensMixin.add_special_tokens.special_tokens_dict",description:`<strong>special_tokens_dict</strong> (dictionary <em>str</em> to <em>str</em> or <code>tokenizers.AddedToken</code>) — | |
| Keys should be in the list of predefined special attributes: [<code>bos_token</code>, <code>eos_token</code>, <code>unk_token</code>, | |
| <code>sep_token</code>, <code>pad_token</code>, <code>cls_token</code>, <code>mask_token</code>, <code>additional_special_tokens</code>].</p> | |
| <p>Tokens are only added if they are not already in the vocabulary (tested by checking if the tokenizer | |
| assign the index of the <code>unk_token</code> to them).`,name:"special_tokens_dict"},{anchor:"transformers.SpecialTokensMixin.add_special_tokens.replace_additional_special_tokens",description:`<strong>replace_additional_special_tokens</strong> (<code>bool</code>, <em>optional</em>,, defaults to <code>True</code>) — | |
| If <code>True</code>, the existing list of additional special tokens will be replaced by the list provided in | |
| <code>special_tokens_dict</code>. Otherwise, <code>self._additional_special_tokens</code> is just extended. In the former | |
| case, the tokens will NOT be removed from the tokenizer’s full vocabulary - they are only being flagged | |
| as non-special tokens. Remember, this only affects which tokens are skipped during decoding, not the | |
| <code>added_tokens_encoder</code> and <code>added_tokens_decoder</code>. This means that the previous | |
| <code>additional_special_tokens</code> are still added tokens, and will not be split by the model.`,name:"replace_additional_special_tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L902",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Number of tokens added to the vocabulary.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>int</code></p> | |
| `}}),Te=new Go({props:{anchor:"transformers.SpecialTokensMixin.add_special_tokens.example",$$slots:{default:[Gs]},$$scope:{ctx:B}}}),ot=new w({props:{name:"add_tokens",anchor:"transformers.SpecialTokensMixin.add_tokens",parameters:[{name:"new_tokens",val:": Union"},{name:"special_tokens",val:": bool = False"}],parametersDescription:[{anchor:"transformers.SpecialTokensMixin.add_tokens.new_tokens",description:`<strong>new_tokens</strong> (<code>str</code>, <code>tokenizers.AddedToken</code> or a list of <em>str</em> or <code>tokenizers.AddedToken</code>) — | |
| Tokens are only added if they are not already in the vocabulary. <code>tokenizers.AddedToken</code> wraps a string | |
| token to let you personalize its behavior: whether this token should only match against a single word, | |
| whether this token should strip all potential whitespaces on the left side, whether this token should | |
| strip all potential whitespaces on the right side, etc.`,name:"new_tokens"},{anchor:"transformers.SpecialTokensMixin.add_tokens.special_tokens",description:`<strong>special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Can be used to specify if the token is a special token. This mostly change the normalization behavior | |
| (special tokens like CLS or [MASK] are usually not lower-cased for instance).</p> | |
| <p>See details for <code>tokenizers.AddedToken</code> in HuggingFace tokenizers library.`,name:"special_tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L1004",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Number of tokens added to the vocabulary.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>int</code></p> | |
| `}}),ve=new Go({props:{anchor:"transformers.SpecialTokensMixin.add_tokens.example",$$slots:{default:[Hs]},$$scope:{ctx:B}}}),nt=new w({props:{name:"sanitize_special_tokens",anchor:"transformers.SpecialTokensMixin.sanitize_special_tokens",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L894"}}),rt=new Ho({props:{title:"Enums 및 namedtuples",local:"transformers.tokenization_utils_base.TruncationStrategy ][ transformers.tokenization_utils_base.TruncationStrategy",headingTag:"h2"}}),st=new w({props:{name:"class transformers.tokenization_utils_base.TruncationStrategy",anchor:"transformers.tokenization_utils_base.TruncationStrategy",parameters:[{name:"value",val:""},{name:"names",val:" = None"},{name:"module",val:" = None"},{name:"qualname",val:" = None"},{name:"type",val:" = None"},{name:"start",val:" = 1"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L154"}}),at=new w({props:{name:"class transformers.CharSpan",anchor:"transformers.CharSpan",parameters:[{name:"start",val:": int"},{name:"end",val:": int"}],parametersDescription:[{anchor:"transformers.CharSpan.start",description:"<strong>start</strong> (<code>int</code>) — Index of the first character in the original string.",name:"start"},{anchor:"transformers.CharSpan.end",description:"<strong>end</strong> (<code>int</code>) — Index of the character following the last character in the original string.",name:"end"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L166"}}),it=new w({props:{name:"class transformers.TokenSpan",anchor:"transformers.TokenSpan",parameters:[{name:"start",val:": int"},{name:"end",val:": int"}],parametersDescription:[{anchor:"transformers.TokenSpan.start",description:"<strong>start</strong> (<code>int</code>) — Index of the first token in the span.",name:"start"},{anchor:"transformers.TokenSpan.end",description:"<strong>end</strong> (<code>int</code>) — Index of the token following the last token in the span.",name:"end"}],source:"https://github.com/huggingface/transformers/blob/vr_34323/src/transformers/tokenization_utils_base.py#L179"}}),dt=new Fs({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/ko/internal/tokenization_utils.md"}}),{c(){a=r("meta"),z=o(),m=r("p"),T=o(),u($.$$.fragment),l=o(),P=r("p"),P.innerHTML=Cr,zo=o(),ze=r("p"),ze.textContent=Wr,$o=o(),u($e.$$.fragment),Po=o(),d=r("div"),u(Pe.$$.fragment),Yo=o(),ft=r("p"),ft.innerHTML=jr,Oo=o(),_t=r("p"),_t.textContent=Ur,Qo=o(),gt=r("p"),gt.textContent=Jr,Ko=o(),kt=r("ul"),kt.innerHTML=Nr,en=o(),te=r("div"),u(Be.$$.fragment),tn=o(),bt=r("p"),bt.textContent=Fr,on=o(),oe=r("div"),u(Me.$$.fragment),nn=o(),Tt=r("p"),Tt.innerHTML=Zr,rn=o(),ne=r("div"),u(qe.$$.fragment),sn=o(),vt=r("p"),vt.textContent=Vr,an=o(),re=r("div"),u(Ie.$$.fragment),dn=o(),xt=r("p"),xt.textContent=Dr,ln=o(),U=r("div"),u(Le.$$.fragment),cn=o(),yt=r("p"),yt.textContent=Sr,pn=o(),u(se.$$.fragment),mn=o(),J=r("div"),u(Ce.$$.fragment),un=o(),wt=r("p"),wt.textContent=Rr,hn=o(),zt=r("p"),zt.textContent=Ar,fn=o(),ae=r("div"),u(We.$$.fragment),_n=o(),$t=r("p"),$t.textContent=Er,gn=o(),ie=r("div"),u(je.$$.fragment),kn=o(),Pt=r("p"),Pt.innerHTML=Gr,bn=o(),N=r("div"),u(Ue.$$.fragment),Tn=o(),Bt=r("p"),Bt.innerHTML=Hr,vn=o(),Mt=r("p"),Mt.textContent=Xr,xn=o(),F=r("div"),u(Je.$$.fragment),yn=o(),qt=r("p"),qt.textContent=Yr,wn=o(),It=r("p"),It.innerHTML=Or,zn=o(),Z=r("div"),u(Ne.$$.fragment),$n=o(),Lt=r("p"),Lt.textContent=Qr,Pn=o(),Ct=r("p"),Ct.innerHTML=Kr,Bn=o(),V=r("div"),u(Fe.$$.fragment),Mn=o(),Wt=r("p"),Wt.textContent=es,qn=o(),u(de.$$.fragment),In=o(),W=r("div"),u(Ze.$$.fragment),Ln=o(),jt=r("p"),jt.innerHTML=ts,Cn=o(),u(le.$$.fragment),Wn=o(),u(ce.$$.fragment),jn=o(),pe=r("div"),u(Ve.$$.fragment),Un=o(),Ut=r("p"),Ut.innerHTML=os,Jn=o(),me=r("div"),u(De.$$.fragment),Nn=o(),Jt=r("p"),Jt.innerHTML=ns,Fn=o(),D=r("div"),u(Se.$$.fragment),Zn=o(),Nt=r("p"),Nt.textContent=rs,Vn=o(),Ft=r("p"),Ft.innerHTML=ss,Dn=o(),L=r("div"),u(Re.$$.fragment),Sn=o(),Zt=r("p"),Zt.textContent=as,Rn=o(),Vt=r("p"),Vt.innerHTML=is,An=o(),Dt=r("p"),Dt.innerHTML=ds,En=o(),u(ue.$$.fragment),Gn=o(),he=r("div"),u(Ae.$$.fragment),Hn=o(),St=r("p"),St.innerHTML=ls,Xn=o(),fe=r("div"),u(Ee.$$.fragment),Yn=o(),Rt=r("p"),Rt.textContent=cs,On=o(),S=r("div"),u(Ge.$$.fragment),Qn=o(),At=r("p"),At.textContent=ps,Kn=o(),u(_e.$$.fragment),er=o(),R=r("div"),u(He.$$.fragment),tr=o(),Et=r("p"),Et.innerHTML=ms,or=o(),u(ge.$$.fragment),nr=o(),j=r("div"),u(Xe.$$.fragment),rr=o(),Gt=r("p"),Gt.textContent=us,sr=o(),Ht=r("p"),Ht.innerHTML=hs,ar=o(),Xt=r("p"),Xt.innerHTML=fs,ir=o(),A=r("div"),u(Ye.$$.fragment),dr=o(),Yt=r("p"),Yt.textContent=_s,lr=o(),Ot=r("p"),Ot.innerHTML=gs,cr=o(),ke=r("div"),u(Oe.$$.fragment),pr=o(),Qt=r("p"),Qt.innerHTML=ks,mr=o(),be=r("div"),u(Qe.$$.fragment),ur=o(),Kt=r("p"),Kt.textContent=bs,Bo=o(),u(Ke.$$.fragment),Mo=o(),I=r("div"),u(et.$$.fragment),hr=o(),eo=r("p"),eo.innerHTML=Ts,fr=o(),M=r("div"),u(tt.$$.fragment),_r=o(),to=r("p"),to.textContent=vs,gr=o(),oo=r("p"),oo.textContent=xs,kr=o(),no=r("p"),no.innerHTML=ys,br=o(),ro=r("p"),ro.innerHTML=ws,Tr=o(),so=r("ul"),so.innerHTML=zs,vr=o(),ao=r("p"),ao.innerHTML=$s,xr=o(),u(Te.$$.fragment),yr=o(),C=r("div"),u(ot.$$.fragment),wr=o(),io=r("p"),io.textContent=Ps,zr=o(),lo=r("p"),lo.textContent=Bs,$r=o(),co=r("p"),co.innerHTML=Ms,Pr=o(),u(ve.$$.fragment),Br=o(),xe=r("div"),u(nt.$$.fragment),Mr=o(),po=r("p"),po.innerHTML=qs,qo=o(),u(rt.$$.fragment),Io=o(),X=r("div"),u(st.$$.fragment),qr=o(),mo=r("p"),mo.innerHTML=Is,Lo=o(),Y=r("div"),u(at.$$.fragment),Ir=o(),uo=r("p"),uo.textContent=Ls,Co=o(),O=r("div"),u(it.$$.fragment),Lr=o(),ho=r("p"),ho.textContent=Cs,Wo=o(),u(dt.$$.fragment),jo=o(),yo=r("p"),this.h()},l(t){const b=Ns("svelte-u9bgzb",document.head);a=s(b,"META",{name:!0,content:!0}),b.forEach(i),z=n(t),m=s(t,"P",{}),x(m).forEach(i),T=n(t),h($.$$.fragment,t),l=n(t),P=s(t,"P",{"data-svelte-h":!0}),c(P)!=="svelte-1qc92ms"&&(P.innerHTML=Cr),zo=n(t),ze=s(t,"P",{"data-svelte-h":!0}),c(ze)!=="svelte-1kgjbf8"&&(ze.textContent=Wr),$o=n(t),h($e.$$.fragment,t),Po=n(t),d=s(t,"DIV",{class:!0});var p=x(d);h(Pe.$$.fragment,p),Yo=n(p),ft=s(p,"P",{"data-svelte-h":!0}),c(ft)!=="svelte-bb1wer"&&(ft.innerHTML=jr),Oo=n(p),_t=s(p,"P",{"data-svelte-h":!0}),c(_t)!=="svelte-2oj7z8"&&(_t.textContent=Ur),Qo=n(p),gt=s(p,"P",{"data-svelte-h":!0}),c(gt)!=="svelte-1ixo79u"&&(gt.textContent=Jr),Ko=n(p),kt=s(p,"UL",{"data-svelte-h":!0}),c(kt)!=="svelte-1y0k6z5"&&(kt.innerHTML=Nr),en=n(p),te=s(p,"DIV",{class:!0});var lt=x(te);h(Be.$$.fragment,lt),tn=n(lt),bt=s(lt,"P",{"data-svelte-h":!0}),c(bt)!=="svelte-kpxj0c"&&(bt.textContent=Fr),lt.forEach(i),on=n(p),oe=s(p,"DIV",{class:!0});var ct=x(oe);h(Me.$$.fragment,ct),nn=n(ct),Tt=s(ct,"P",{"data-svelte-h":!0}),c(Tt)!=="svelte-j87b6t"&&(Tt.innerHTML=Zr),ct.forEach(i),rn=n(p),ne=s(p,"DIV",{class:!0});var pt=x(ne);h(qe.$$.fragment,pt),sn=n(pt),vt=s(pt,"P",{"data-svelte-h":!0}),c(vt)!=="svelte-hxrl9"&&(vt.textContent=Vr),pt.forEach(i),an=n(p),re=s(p,"DIV",{class:!0});var mt=x(re);h(Ie.$$.fragment,mt),dn=n(mt),xt=s(mt,"P",{"data-svelte-h":!0}),c(xt)!=="svelte-1deng2j"&&(xt.textContent=Dr),mt.forEach(i),ln=n(p),U=s(p,"DIV",{class:!0});var Q=x(U);h(Le.$$.fragment,Q),cn=n(Q),yt=s(Q,"P",{"data-svelte-h":!0}),c(yt)!=="svelte-6p21pf"&&(yt.textContent=Sr),pn=n(Q),h(se.$$.fragment,Q),Q.forEach(i),mn=n(p),J=s(p,"DIV",{class:!0});var K=x(J);h(Ce.$$.fragment,K),un=n(K),wt=s(K,"P",{"data-svelte-h":!0}),c(wt)!=="svelte-xip562"&&(wt.textContent=Rr),hn=n(K),zt=s(K,"P",{"data-svelte-h":!0}),c(zt)!=="svelte-1yvfiyo"&&(zt.textContent=Ar),K.forEach(i),fn=n(p),ae=s(p,"DIV",{class:!0});var ut=x(ae);h(We.$$.fragment,ut),_n=n(ut),$t=s(ut,"P",{"data-svelte-h":!0}),c($t)!=="svelte-6a62nd"&&($t.textContent=Er),ut.forEach(i),gn=n(p),ie=s(p,"DIV",{class:!0});var ht=x(ie);h(je.$$.fragment,ht),kn=n(ht),Pt=s(ht,"P",{"data-svelte-h":!0}),c(Pt)!=="svelte-sfkaj8"&&(Pt.innerHTML=Gr),ht.forEach(i),bn=n(p),N=s(p,"DIV",{class:!0});var fo=x(N);h(Ue.$$.fragment,fo),Tn=n(fo),Bt=s(fo,"P",{"data-svelte-h":!0}),c(Bt)!=="svelte-zj1vf1"&&(Bt.innerHTML=Hr),vn=n(fo),Mt=s(fo,"P",{"data-svelte-h":!0}),c(Mt)!=="svelte-9vptpw"&&(Mt.textContent=Xr),fo.forEach(i),xn=n(p),F=s(p,"DIV",{class:!0});var _o=x(F);h(Je.$$.fragment,_o),yn=n(_o),qt=s(_o,"P",{"data-svelte-h":!0}),c(qt)!=="svelte-vbfkpu"&&(qt.textContent=Yr),wn=n(_o),It=s(_o,"P",{"data-svelte-h":!0}),c(It)!=="svelte-125uxon"&&(It.innerHTML=Or),_o.forEach(i),zn=n(p),Z=s(p,"DIV",{class:!0});var go=x(Z);h(Ne.$$.fragment,go),$n=n(go),Lt=s(go,"P",{"data-svelte-h":!0}),c(Lt)!=="svelte-12b8hzo"&&(Lt.textContent=Qr),Pn=n(go),Ct=s(go,"P",{"data-svelte-h":!0}),c(Ct)!=="svelte-1kyhveh"&&(Ct.innerHTML=Kr),go.forEach(i),Bn=n(p),V=s(p,"DIV",{class:!0});var ko=x(V);h(Fe.$$.fragment,ko),Mn=n(ko),Wt=s(ko,"P",{"data-svelte-h":!0}),c(Wt)!=="svelte-ma945j"&&(Wt.textContent=es),qn=n(ko),h(de.$$.fragment,ko),ko.forEach(i),In=n(p),W=s(p,"DIV",{class:!0});var ye=x(W);h(Ze.$$.fragment,ye),Ln=n(ye),jt=s(ye,"P",{"data-svelte-h":!0}),c(jt)!=="svelte-1mppd76"&&(jt.innerHTML=ts),Cn=n(ye),h(le.$$.fragment,ye),Wn=n(ye),h(ce.$$.fragment,ye),ye.forEach(i),jn=n(p),pe=s(p,"DIV",{class:!0});var Jo=x(pe);h(Ve.$$.fragment,Jo),Un=n(Jo),Ut=s(Jo,"P",{"data-svelte-h":!0}),c(Ut)!=="svelte-1hrpjri"&&(Ut.innerHTML=os),Jo.forEach(i),Jn=n(p),me=s(p,"DIV",{class:!0});var No=x(me);h(De.$$.fragment,No),Nn=n(No),Jt=s(No,"P",{"data-svelte-h":!0}),c(Jt)!=="svelte-1wmjg8a"&&(Jt.innerHTML=ns),No.forEach(i),Fn=n(p),D=s(p,"DIV",{class:!0});var bo=x(D);h(Se.$$.fragment,bo),Zn=n(bo),Nt=s(bo,"P",{"data-svelte-h":!0}),c(Nt)!=="svelte-1gbatu6"&&(Nt.textContent=rs),Vn=n(bo),Ft=s(bo,"P",{"data-svelte-h":!0}),c(Ft)!=="svelte-907bv"&&(Ft.innerHTML=ss),bo.forEach(i),Dn=n(p),L=s(p,"DIV",{class:!0});var E=x(L);h(Re.$$.fragment,E),Sn=n(E),Zt=s(E,"P",{"data-svelte-h":!0}),c(Zt)!=="svelte-1n892mi"&&(Zt.textContent=as),Rn=n(E),Vt=s(E,"P",{"data-svelte-h":!0}),c(Vt)!=="svelte-1xir9yc"&&(Vt.innerHTML=is),An=n(E),Dt=s(E,"P",{"data-svelte-h":!0}),c(Dt)!=="svelte-28v9x1"&&(Dt.innerHTML=ds),En=n(E),h(ue.$$.fragment,E),E.forEach(i),Gn=n(p),he=s(p,"DIV",{class:!0});var Fo=x(he);h(Ae.$$.fragment,Fo),Hn=n(Fo),St=s(Fo,"P",{"data-svelte-h":!0}),c(St)!=="svelte-cwdwvn"&&(St.innerHTML=ls),Fo.forEach(i),Xn=n(p),fe=s(p,"DIV",{class:!0});var Zo=x(fe);h(Ee.$$.fragment,Zo),Yn=n(Zo),Rt=s(Zo,"P",{"data-svelte-h":!0}),c(Rt)!=="svelte-dtuae6"&&(Rt.textContent=cs),Zo.forEach(i),On=n(p),S=s(p,"DIV",{class:!0});var To=x(S);h(Ge.$$.fragment,To),Qn=n(To),At=s(To,"P",{"data-svelte-h":!0}),c(At)!=="svelte-tpmkl3"&&(At.textContent=ps),Kn=n(To),h(_e.$$.fragment,To),To.forEach(i),er=n(p),R=s(p,"DIV",{class:!0});var vo=x(R);h(He.$$.fragment,vo),tr=n(vo),Et=s(vo,"P",{"data-svelte-h":!0}),c(Et)!=="svelte-189h5u2"&&(Et.innerHTML=ms),or=n(vo),h(ge.$$.fragment,vo),vo.forEach(i),nr=n(p),j=s(p,"DIV",{class:!0});var we=x(j);h(Xe.$$.fragment,we),rr=n(we),Gt=s(we,"P",{"data-svelte-h":!0}),c(Gt)!=="svelte-u73u19"&&(Gt.textContent=us),sr=n(we),Ht=s(we,"P",{"data-svelte-h":!0}),c(Ht)!=="svelte-p0dt88"&&(Ht.innerHTML=hs),ar=n(we),Xt=s(we,"P",{"data-svelte-h":!0}),c(Xt)!=="svelte-1qee1z"&&(Xt.innerHTML=fs),we.forEach(i),ir=n(p),A=s(p,"DIV",{class:!0});var xo=x(A);h(Ye.$$.fragment,xo),dr=n(xo),Yt=s(xo,"P",{"data-svelte-h":!0}),c(Yt)!=="svelte-rx0wq1"&&(Yt.textContent=_s),lr=n(xo),Ot=s(xo,"P",{"data-svelte-h":!0}),c(Ot)!=="svelte-c9295b"&&(Ot.innerHTML=gs),xo.forEach(i),cr=n(p),ke=s(p,"DIV",{class:!0});var Vo=x(ke);h(Oe.$$.fragment,Vo),pr=n(Vo),Qt=s(Vo,"P",{"data-svelte-h":!0}),c(Qt)!=="svelte-ikoqgw"&&(Qt.innerHTML=ks),Vo.forEach(i),mr=n(p),be=s(p,"DIV",{class:!0});var Do=x(be);h(Qe.$$.fragment,Do),ur=n(Do),Kt=s(Do,"P",{"data-svelte-h":!0}),c(Kt)!=="svelte-fkofn"&&(Kt.textContent=bs),Do.forEach(i),p.forEach(i),Bo=n(t),h(Ke.$$.fragment,t),Mo=n(t),I=s(t,"DIV",{class:!0});var G=x(I);h(et.$$.fragment,G),hr=n(G),eo=s(G,"P",{"data-svelte-h":!0}),c(eo)!=="svelte-hhzagf"&&(eo.innerHTML=Ts),fr=n(G),M=s(G,"DIV",{class:!0});var q=x(M);h(tt.$$.fragment,q),_r=n(q),to=s(q,"P",{"data-svelte-h":!0}),c(to)!=="svelte-1j8s0i5"&&(to.textContent=vs),gr=n(q),oo=s(q,"P",{"data-svelte-h":!0}),c(oo)!=="svelte-1w3ayx9"&&(oo.textContent=xs),kr=n(q),no=s(q,"P",{"data-svelte-h":!0}),c(no)!=="svelte-1aaz4tx"&&(no.innerHTML=ys),br=n(q),ro=s(q,"P",{"data-svelte-h":!0}),c(ro)!=="svelte-5hxtpc"&&(ro.innerHTML=ws),Tr=n(q),so=s(q,"UL",{"data-svelte-h":!0}),c(so)!=="svelte-1pes0uj"&&(so.innerHTML=zs),vr=n(q),ao=s(q,"P",{"data-svelte-h":!0}),c(ao)!=="svelte-1butrf6"&&(ao.innerHTML=$s),xr=n(q),h(Te.$$.fragment,q),q.forEach(i),yr=n(G),C=s(G,"DIV",{class:!0});var H=x(C);h(ot.$$.fragment,H),wr=n(H),io=s(H,"P",{"data-svelte-h":!0}),c(io)!=="svelte-14rotd7"&&(io.textContent=Ps),zr=n(H),lo=s(H,"P",{"data-svelte-h":!0}),c(lo)!=="svelte-j0w5r1"&&(lo.textContent=Bs),$r=n(H),co=s(H,"P",{"data-svelte-h":!0}),c(co)!=="svelte-1aaz4tx"&&(co.innerHTML=Ms),Pr=n(H),h(ve.$$.fragment,H),H.forEach(i),Br=n(G),xe=s(G,"DIV",{class:!0});var So=x(xe);h(nt.$$.fragment,So),Mr=n(So),po=s(So,"P",{"data-svelte-h":!0}),c(po)!=="svelte-1em0285"&&(po.innerHTML=qs),So.forEach(i),G.forEach(i),qo=n(t),h(rt.$$.fragment,t),Io=n(t),X=s(t,"DIV",{class:!0});var Ro=x(X);h(st.$$.fragment,Ro),qr=n(Ro),mo=s(Ro,"P",{"data-svelte-h":!0}),c(mo)!=="svelte-agrmr3"&&(mo.innerHTML=Is),Ro.forEach(i),Lo=n(t),Y=s(t,"DIV",{class:!0});var Ao=x(Y);h(at.$$.fragment,Ao),Ir=n(Ao),uo=s(Ao,"P",{"data-svelte-h":!0}),c(uo)!=="svelte-136gduh"&&(uo.textContent=Ls),Ao.forEach(i),Co=n(t),O=s(t,"DIV",{class:!0});var Eo=x(O);h(it.$$.fragment,Eo),Lr=n(Eo),ho=s(Eo,"P",{"data-svelte-h":!0}),c(ho)!=="svelte-18ocfae"&&(ho.textContent=Cs),Eo.forEach(i),Wo=n(t),h(dt.$$.fragment,t),jo=n(t),yo=s(t,"P",{}),x(yo).forEach(i),this.h()},h(){y(a,"name","hf:doc:metadata"),y(a,"content",Ys),y(te,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(oe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(ne,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(re,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(U,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(J,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(ae,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(ie,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(N,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(F,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(Z,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(V,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(W,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(pe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(me,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(D,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(L,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(he,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(fe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(S,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(R,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(j,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(A,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(ke,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(be,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(d,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(M,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(C,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(xe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(I,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(X,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(Y,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),y(O,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(t,b){e(document.head,a),v(t,z,b),v(t,m,b),v(t,T,b),f($,t,b),v(t,l,b),v(t,P,b),v(t,zo,b),v(t,ze,b),v(t,$o,b),f($e,t,b),v(t,Po,b),v(t,d,b),f(Pe,d,null),e(d,Yo),e(d,ft),e(d,Oo),e(d,_t),e(d,Qo),e(d,gt),e(d,Ko),e(d,kt),e(d,en),e(d,te),f(Be,te,null),e(te,tn),e(te,bt),e(d,on),e(d,oe),f(Me,oe,null),e(oe,nn),e(oe,Tt),e(d,rn),e(d,ne),f(qe,ne,null),e(ne,sn),e(ne,vt),e(d,an),e(d,re),f(Ie,re,null),e(re,dn),e(re,xt),e(d,ln),e(d,U),f(Le,U,null),e(U,cn),e(U,yt),e(U,pn),f(se,U,null),e(d,mn),e(d,J),f(Ce,J,null),e(J,un),e(J,wt),e(J,hn),e(J,zt),e(d,fn),e(d,ae),f(We,ae,null),e(ae,_n),e(ae,$t),e(d,gn),e(d,ie),f(je,ie,null),e(ie,kn),e(ie,Pt),e(d,bn),e(d,N),f(Ue,N,null),e(N,Tn),e(N,Bt),e(N,vn),e(N,Mt),e(d,xn),e(d,F),f(Je,F,null),e(F,yn),e(F,qt),e(F,wn),e(F,It),e(d,zn),e(d,Z),f(Ne,Z,null),e(Z,$n),e(Z,Lt),e(Z,Pn),e(Z,Ct),e(d,Bn),e(d,V),f(Fe,V,null),e(V,Mn),e(V,Wt),e(V,qn),f(de,V,null),e(d,In),e(d,W),f(Ze,W,null),e(W,Ln),e(W,jt),e(W,Cn),f(le,W,null),e(W,Wn),f(ce,W,null),e(d,jn),e(d,pe),f(Ve,pe,null),e(pe,Un),e(pe,Ut),e(d,Jn),e(d,me),f(De,me,null),e(me,Nn),e(me,Jt),e(d,Fn),e(d,D),f(Se,D,null),e(D,Zn),e(D,Nt),e(D,Vn),e(D,Ft),e(d,Dn),e(d,L),f(Re,L,null),e(L,Sn),e(L,Zt),e(L,Rn),e(L,Vt),e(L,An),e(L,Dt),e(L,En),f(ue,L,null),e(d,Gn),e(d,he),f(Ae,he,null),e(he,Hn),e(he,St),e(d,Xn),e(d,fe),f(Ee,fe,null),e(fe,Yn),e(fe,Rt),e(d,On),e(d,S),f(Ge,S,null),e(S,Qn),e(S,At),e(S,Kn),f(_e,S,null),e(d,er),e(d,R),f(He,R,null),e(R,tr),e(R,Et),e(R,or),f(ge,R,null),e(d,nr),e(d,j),f(Xe,j,null),e(j,rr),e(j,Gt),e(j,sr),e(j,Ht),e(j,ar),e(j,Xt),e(d,ir),e(d,A),f(Ye,A,null),e(A,dr),e(A,Yt),e(A,lr),e(A,Ot),e(d,cr),e(d,ke),f(Oe,ke,null),e(ke,pr),e(ke,Qt),e(d,mr),e(d,be),f(Qe,be,null),e(be,ur),e(be,Kt),v(t,Bo,b),f(Ke,t,b),v(t,Mo,b),v(t,I,b),f(et,I,null),e(I,hr),e(I,eo),e(I,fr),e(I,M),f(tt,M,null),e(M,_r),e(M,to),e(M,gr),e(M,oo),e(M,kr),e(M,no),e(M,br),e(M,ro),e(M,Tr),e(M,so),e(M,vr),e(M,ao),e(M,xr),f(Te,M,null),e(I,yr),e(I,C),f(ot,C,null),e(C,wr),e(C,io),e(C,zr),e(C,lo),e(C,$r),e(C,co),e(C,Pr),f(ve,C,null),e(I,Br),e(I,xe),f(nt,xe,null),e(xe,Mr),e(xe,po),v(t,qo,b),f(rt,t,b),v(t,Io,b),v(t,X,b),f(st,X,null),e(X,qr),e(X,mo),v(t,Lo,b),v(t,Y,b),f(at,Y,null),e(Y,Ir),e(Y,uo),v(t,Co,b),v(t,O,b),f(it,O,null),e(O,Lr),e(O,ho),v(t,Wo,b),f(dt,t,b),v(t,jo,b),v(t,yo,b),Uo=!0},p(t,[b]){const p={};b&2&&(p.$$scope={dirty:b,ctx:t}),se.$set(p);const lt={};b&2&&(lt.$$scope={dirty:b,ctx:t}),de.$set(lt);const ct={};b&2&&(ct.$$scope={dirty:b,ctx:t}),le.$set(ct);const pt={};b&2&&(pt.$$scope={dirty:b,ctx:t}),ce.$set(pt);const mt={};b&2&&(mt.$$scope={dirty:b,ctx:t}),ue.$set(mt);const Q={};b&2&&(Q.$$scope={dirty:b,ctx:t}),_e.$set(Q);const K={};b&2&&(K.$$scope={dirty:b,ctx:t}),ge.$set(K);const ut={};b&2&&(ut.$$scope={dirty:b,ctx:t}),Te.$set(ut);const ht={};b&2&&(ht.$$scope={dirty:b,ctx:t}),ve.$set(ht)},i(t){Uo||(_($.$$.fragment,t),_($e.$$.fragment,t),_(Pe.$$.fragment,t),_(Be.$$.fragment,t),_(Me.$$.fragment,t),_(qe.$$.fragment,t),_(Ie.$$.fragment,t),_(Le.$$.fragment,t),_(se.$$.fragment,t),_(Ce.$$.fragment,t),_(We.$$.fragment,t),_(je.$$.fragment,t),_(Ue.$$.fragment,t),_(Je.$$.fragment,t),_(Ne.$$.fragment,t),_(Fe.$$.fragment,t),_(de.$$.fragment,t),_(Ze.$$.fragment,t),_(le.$$.fragment,t),_(ce.$$.fragment,t),_(Ve.$$.fragment,t),_(De.$$.fragment,t),_(Se.$$.fragment,t),_(Re.$$.fragment,t),_(ue.$$.fragment,t),_(Ae.$$.fragment,t),_(Ee.$$.fragment,t),_(Ge.$$.fragment,t),_(_e.$$.fragment,t),_(He.$$.fragment,t),_(ge.$$.fragment,t),_(Xe.$$.fragment,t),_(Ye.$$.fragment,t),_(Oe.$$.fragment,t),_(Qe.$$.fragment,t),_(Ke.$$.fragment,t),_(et.$$.fragment,t),_(tt.$$.fragment,t),_(Te.$$.fragment,t),_(ot.$$.fragment,t),_(ve.$$.fragment,t),_(nt.$$.fragment,t),_(rt.$$.fragment,t),_(st.$$.fragment,t),_(at.$$.fragment,t),_(it.$$.fragment,t),_(dt.$$.fragment,t),Uo=!0)},o(t){g($.$$.fragment,t),g($e.$$.fragment,t),g(Pe.$$.fragment,t),g(Be.$$.fragment,t),g(Me.$$.fragment,t),g(qe.$$.fragment,t),g(Ie.$$.fragment,t),g(Le.$$.fragment,t),g(se.$$.fragment,t),g(Ce.$$.fragment,t),g(We.$$.fragment,t),g(je.$$.fragment,t),g(Ue.$$.fragment,t),g(Je.$$.fragment,t),g(Ne.$$.fragment,t),g(Fe.$$.fragment,t),g(de.$$.fragment,t),g(Ze.$$.fragment,t),g(le.$$.fragment,t),g(ce.$$.fragment,t),g(Ve.$$.fragment,t),g(De.$$.fragment,t),g(Se.$$.fragment,t),g(Re.$$.fragment,t),g(ue.$$.fragment,t),g(Ae.$$.fragment,t),g(Ee.$$.fragment,t),g(Ge.$$.fragment,t),g(_e.$$.fragment,t),g(He.$$.fragment,t),g(ge.$$.fragment,t),g(Xe.$$.fragment,t),g(Ye.$$.fragment,t),g(Oe.$$.fragment,t),g(Qe.$$.fragment,t),g(Ke.$$.fragment,t),g(et.$$.fragment,t),g(tt.$$.fragment,t),g(Te.$$.fragment,t),g(ot.$$.fragment,t),g(ve.$$.fragment,t),g(nt.$$.fragment,t),g(rt.$$.fragment,t),g(st.$$.fragment,t),g(at.$$.fragment,t),g(it.$$.fragment,t),g(dt.$$.fragment,t),Uo=!1},d(t){t&&(i(z),i(m),i(T),i(l),i(P),i(zo),i(ze),i($o),i(Po),i(d),i(Bo),i(Mo),i(I),i(qo),i(Io),i(X),i(Lo),i(Y),i(Co),i(O),i(Wo),i(jo),i(yo)),i(a),k($,t),k($e,t),k(Pe),k(Be),k(Me),k(qe),k(Ie),k(Le),k(se),k(Ce),k(We),k(je),k(Ue),k(Je),k(Ne),k(Fe),k(de),k(Ze),k(le),k(ce),k(Ve),k(De),k(Se),k(Re),k(ue),k(Ae),k(Ee),k(Ge),k(_e),k(He),k(ge),k(Xe),k(Ye),k(Oe),k(Qe),k(Ke,t),k(et),k(tt),k(Te),k(ot),k(ve),k(nt),k(rt,t),k(st),k(at),k(it),k(dt,t)}}}const Ys='{"title":"토크나이저를 위한 유틸리티","local":"utilities-for-tokenizers","sections":[{"title":"PreTrainedTokenizerBase","local":"transformers.PreTrainedTokenizerBase ][ transformers.PreTrainedTokenizerBase","sections":[],"depth":2},{"title":"SpecialTokensMixin","local":"transformers.SpecialTokensMixin ][ transformers.SpecialTokensMixin","sections":[],"depth":2},{"title":"Enums 및 namedtuples","local":"transformers.tokenization_utils_base.TruncationStrategy ][ transformers.tokenization_utils_base.TruncationStrategy","sections":[],"depth":2}],"depth":1}';function Os(B){return js(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class sa extends Us{constructor(a){super(),Js(this,a,Os,Xs,Ws,{})}}export{sa as component}; | |
Xet Storage Details
- Size:
- 173 kB
- Xet hash:
- 1c4648564f388c7bf25a0ca8cd0cd4b031dcace86bbf700a9c2f440a74fac32c
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.