Buckets:
| import{s as co,o as po,n as Ve}from"../chunks/scheduler.01eeda35.js";import{S as mo,i as ho,g as i,s as a,r as u,A as uo,h as l,f as d,c as r,j as I,u as f,x as h,k as L,y as n,a as p,v as _,d as g,t as T,w as b}from"../chunks/index.6dd51b66.js";import{T as lo}from"../chunks/Tip.de9bae2b.js";import{D as V}from"../chunks/Docstring.cb556860.js";import{C as rt}from"../chunks/CodeBlock.19ec9b8c.js";import{E as at}from"../chunks/ExampleCodeBlock.69db56ad.js";import{H as Te,E as fo}from"../chunks/index.58fe8f9d.js";function _o(x){let o,y="Examples:",m,c,k;return c=new rt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEZTTVRDb25maWclMkMlMjBGU01UTW9kZWwlMEElMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwRlNNVCUyMGZhY2Vib29rJTJGd210MTktZW4tcnUlMjBzdHlsZSUyMGNvbmZpZ3VyYXRpb24lMEFjb25maWclMjAlM0QlMjBGU01UQ29uZmlnKCklMEElMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwbW9kZWwlMjAod2l0aCUyMHJhbmRvbSUyMHdlaWdodHMpJTIwZnJvbSUyMHRoZSUyMGNvbmZpZ3VyYXRpb24lMEFtb2RlbCUyMCUzRCUyMEZTTVRNb2RlbChjb25maWcpJTBBJTBBJTIzJTIwQWNjZXNzaW5nJTIwdGhlJTIwbW9kZWwlMjBjb25maWd1cmF0aW9uJTBBY29uZmlndXJhdGlvbiUyMCUzRCUyMG1vZGVsLmNvbmZpZw==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> FSMTConfig, FSMTModel | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a FSMT facebook/wmt19-en-ru style configuration</span> | |
| <span class="hljs-meta">>>> </span>config = FSMTConfig() | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a model (with random weights) from the configuration</span> | |
| <span class="hljs-meta">>>> </span>model = FSMTModel(config) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Accessing the model configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = model.config`,wrap:!1}}),{c(){o=i("p"),o.textContent=y,m=a(),u(c.$$.fragment)},l(t){o=l(t,"P",{"data-svelte-h":!0}),h(o)!=="svelte-kvfsh7"&&(o.textContent=y),m=r(t),f(c.$$.fragment,t)},m(t,M){p(t,o,M),p(t,m,M),_(c,t,M),k=!0},p:Ve,i(t){k||(g(c.$$.fragment,t),k=!0)},o(t){T(c.$$.fragment,t),k=!1},d(t){t&&(d(o),d(m)),b(c,t)}}}function go(x){let o,y="Transformer sequence pair mask has the following format:",m,c,k;return c=new rt({props:{code:"MCUyMDAlMjAwJTIwMCUyMDAlMjAwJTIwMCUyMDAlMjAwJTIwMCUyMDAlMjAxJTIwMSUyMDElMjAxJTIwMSUyMDElMjAxJTIwMSUyMDElMEElN0MlMjBmaXJzdCUyMHNlcXVlbmNlJTIwJTIwJTIwJTIwJTdDJTIwc2Vjb25kJTIwc2VxdWVuY2UlMjAlN0M=",highlighted:`0<span class="hljs-number"> 0 </span>0<span class="hljs-number"> 0 </span>0<span class="hljs-number"> 0 </span>0<span class="hljs-number"> 0 </span>0<span class="hljs-number"> 0 </span>0<span class="hljs-number"> 1 </span>1<span class="hljs-number"> 1 </span>1<span class="hljs-number"> 1 </span>1<span class="hljs-number"> 1 </span>1 1 | |
| | first sequence | second sequence |`,wrap:!1}}),{c(){o=i("p"),o.textContent=y,m=a(),u(c.$$.fragment)},l(t){o=l(t,"P",{"data-svelte-h":!0}),h(o)!=="svelte-12v5j2d"&&(o.textContent=y),m=r(t),f(c.$$.fragment,t)},m(t,M){p(t,o,M),p(t,m,M),_(c,t,M),k=!0},p:Ve,i(t){k||(g(c.$$.fragment,t),k=!0)},o(t){T(c.$$.fragment,t),k=!1},d(t){t&&(d(o),d(m)),b(c,t)}}}function To(x){let o,y=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=i("p"),o.innerHTML=y},l(m){o=l(m,"P",{"data-svelte-h":!0}),h(o)!=="svelte-fincs2"&&(o.innerHTML=y)},m(m,c){p(m,o,c)},p:Ve,d(m){m&&d(o)}}}function bo(x){let o,y="Example:",m,c,k;return c=new rt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBGU01UTW9kZWwlMEFpbXBvcnQlMjB0b3JjaCUwQSUwQXRva2VuaXplciUyMCUzRCUyMEF1dG9Ub2tlbml6ZXIuZnJvbV9wcmV0cmFpbmVkKCUyMmZhY2Vib29rJTJGd210MTktcnUtZW4lMjIpJTBBbW9kZWwlMjAlM0QlMjBGU01UTW9kZWwuZnJvbV9wcmV0cmFpbmVkKCUyMmZhY2Vib29rJTJGd210MTktcnUtZW4lMjIpJTBBJTBBaW5wdXRzJTIwJTNEJTIwdG9rZW5pemVyKCUyMkhlbGxvJTJDJTIwbXklMjBkb2clMjBpcyUyMGN1dGUlMjIlMkMlMjByZXR1cm5fdGVuc29ycyUzRCUyMnB0JTIyKSUwQW91dHB1dHMlMjAlM0QlMjBtb2RlbCgqKmlucHV0cyklMEElMEFsYXN0X2hpZGRlbl9zdGF0ZXMlMjAlM0QlMjBvdXRwdXRzLmxhc3RfaGlkZGVuX3N0YXRl",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, FSMTModel | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span>tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"facebook/wmt19-ru-en"</span>) | |
| <span class="hljs-meta">>>> </span>model = FSMTModel.from_pretrained(<span class="hljs-string">"facebook/wmt19-ru-en"</span>) | |
| <span class="hljs-meta">>>> </span>inputs = tokenizer(<span class="hljs-string">"Hello, my dog is cute"</span>, return_tensors=<span class="hljs-string">"pt"</span>) | |
| <span class="hljs-meta">>>> </span>outputs = model(**inputs) | |
| <span class="hljs-meta">>>> </span>last_hidden_states = outputs.last_hidden_state`,wrap:!1}}),{c(){o=i("p"),o.textContent=y,m=a(),u(c.$$.fragment)},l(t){o=l(t,"P",{"data-svelte-h":!0}),h(o)!=="svelte-11lpom8"&&(o.textContent=y),m=r(t),f(c.$$.fragment,t)},m(t,M){p(t,o,M),p(t,m,M),_(c,t,M),k=!0},p:Ve,i(t){k||(g(c.$$.fragment,t),k=!0)},o(t){T(c.$$.fragment,t),k=!1},d(t){t&&(d(o),d(m)),b(c,t)}}}function ko(x){let o,y=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=i("p"),o.innerHTML=y},l(m){o=l(m,"P",{"data-svelte-h":!0}),h(o)!=="svelte-fincs2"&&(o.innerHTML=y)},m(m,c){p(m,o,c)},p:Ve,d(m){m&&d(o)}}}function Mo(x){let o,y="Translation example::",m,c,k;return c=new rt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBGU01URm9yQ29uZGl0aW9uYWxHZW5lcmF0aW9uJTBBJTBBbW5hbWUlMjAlM0QlMjAlMjJmYWNlYm9vayUyRndtdDE5LXJ1LWVuJTIyJTBBbW9kZWwlMjAlM0QlMjBGU01URm9yQ29uZGl0aW9uYWxHZW5lcmF0aW9uLmZyb21fcHJldHJhaW5lZChtbmFtZSklMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZChtbmFtZSklMEElMEFzcmNfdGV4dCUyMCUzRCUyMCUyMiVEMCU5QyVEMCVCMCVEMSU4OCVEMCVCOCVEMCVCRCVEMCVCRCVEMCVCRSVEMCVCNSUyMCVEMCVCRSVEMCVCMSVEMSU4MyVEMSU4NyVEMCVCNSVEMCVCRCVEMCVCOCVEMCVCNSUyMC0lMjAlRDElOEQlRDElODIlRDAlQkUlMjAlRDAlQjclRDAlQjQlRDAlQkUlRDElODAlRDAlQkUlRDAlQjIlRDAlQkUlMkMlMjAlRDAlQkQlRDAlQjUlMjAlRDElODIlRDAlQjAlRDAlQkElMjAlRDAlQkIlRDAlQjglM0YlMjIlMEFpbnB1dF9pZHMlMjAlM0QlMjB0b2tlbml6ZXIoc3JjX3RleHQlMkMlMjByZXR1cm5fdGVuc29ycyUzRCUyMnB0JTIyKS5pbnB1dF9pZHMlMEFvdXRwdXRzJTIwJTNEJTIwbW9kZWwuZ2VuZXJhdGUoaW5wdXRfaWRzJTJDJTIwbnVtX2JlYW1zJTNENSUyQyUyMG51bV9yZXR1cm5fc2VxdWVuY2VzJTNEMyklMEF0b2tlbml6ZXIuZGVjb2RlKG91dHB1dHMlNUIwJTVEJTJDJTIwc2tpcF9zcGVjaWFsX3Rva2VucyUzRFRydWUp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, FSMTForConditionalGeneration | |
| <span class="hljs-meta">>>> </span>mname = <span class="hljs-string">"facebook/wmt19-ru-en"</span> | |
| <span class="hljs-meta">>>> </span>model = FSMTForConditionalGeneration.from_pretrained(mname) | |
| <span class="hljs-meta">>>> </span>tokenizer = AutoTokenizer.from_pretrained(mname) | |
| <span class="hljs-meta">>>> </span>src_text = <span class="hljs-string">"Машинное обучение - это здорово, не так ли?"</span> | |
| <span class="hljs-meta">>>> </span>input_ids = tokenizer(src_text, return_tensors=<span class="hljs-string">"pt"</span>).input_ids | |
| <span class="hljs-meta">>>> </span>outputs = model.generate(input_ids, num_beams=<span class="hljs-number">5</span>, num_return_sequences=<span class="hljs-number">3</span>) | |
| <span class="hljs-meta">>>> </span>tokenizer.decode(outputs[<span class="hljs-number">0</span>], skip_special_tokens=<span class="hljs-literal">True</span>) | |
| <span class="hljs-string">"Machine learning is great, isn't it?"</span>`,wrap:!1}}),{c(){o=i("p"),o.textContent=y,m=a(),u(c.$$.fragment)},l(t){o=l(t,"P",{"data-svelte-h":!0}),h(o)!=="svelte-1ezebad"&&(o.textContent=y),m=r(t),f(c.$$.fragment,t)},m(t,M){p(t,o,M),p(t,m,M),_(c,t,M),k=!0},p:Ve,i(t){k||(g(c.$$.fragment,t),k=!0)},o(t){T(c.$$.fragment,t),k=!1},d(t){t&&(d(o),d(m)),b(c,t)}}}function yo(x){let o,y,m,c,k,t,M,Ge,X,Wt='FSMT (FairSeq MachineTranslation) models were introduced in <a href="https://arxiv.org/abs/1907.06616" rel="nofollow">Facebook FAIR’s WMT19 News Translation Task Submission</a> by Nathan Ng, Kyra Yee, Alexei Baevski, Myle Ott, Michael Auli, Sergey Edunov.',Je,Q,Vt="The abstract of the paper is the following:",Re,Y,At=`<em>This paper describes Facebook FAIR’s submission to the WMT19 shared news translation task. We participate in two | |
| language pairs and four language directions, English <-> German and English <-> Russian. Following our submission from | |
| last year, our baseline systems are large BPE-based transformer models trained with the Fairseq sequence modeling | |
| toolkit which rely on sampled back-translations. This year we experiment with different bitext data filtering schemes, | |
| as well as with adding filtered back-translated data. We also ensemble and fine-tune our models on domain-specific | |
| data, then decode using noisy channel model reranking. Our submissions are ranked first in all four directions of the | |
| human evaluation campaign. On En->De, our system significantly outperforms other systems as well as human translations. | |
| This system improves upon our WMT’18 submission by 4.5 BLEU points.</em>`,Be,K,Nt=`This model was contributed by <a href="https://huggingface.co/stas" rel="nofollow">stas</a>. The original code can be found | |
| <a href="https://github.com/pytorch/fairseq/tree/master/examples/wmt19" rel="nofollow">here</a>.`,Pe,ee,He,te,Gt=`<li>FSMT uses source and target vocabulary pairs that aren’t combined into one. It doesn’t share embeddings tokens | |
| either. Its tokenizer is very similar to <a href="/docs/transformers/pr_37177/en/model_doc/xlm#transformers.XLMTokenizer">XLMTokenizer</a> and the main model is derived from | |
| <a href="/docs/transformers/pr_37177/en/model_doc/bart#transformers.BartModel">BartModel</a>.</li>`,Oe,oe,Ze,S,ne,dt,be,Jt=`This is the configuration class to store the configuration of a <a href="/docs/transformers/pr_37177/en/model_doc/fsmt#transformers.FSMTModel">FSMTModel</a>. It is used to instantiate a FSMT | |
| model according to the specified arguments, defining the model architecture. Instantiating a configuration with the | |
| defaults will yield a similar configuration to that of the FSMT | |
| <a href="https://huggingface.co/facebook/wmt19-en-ru" rel="nofollow">facebook/wmt19-en-ru</a> architecture.`,it,ke,Rt=`Configuration objects inherit from <a href="/docs/transformers/pr_37177/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> and can be used to control the model outputs. Read the | |
| documentation from <a href="/docs/transformers/pr_37177/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> for more information.`,lt,N,Xe,se,Qe,v,ae,ct,Me,Bt="Construct an FAIRSEQ Transformer tokenizer. Based on Byte-Pair Encoding. The tokenization process is the following:",pt,ye,Pt=`<li>Moses preprocessing and tokenization.</li> <li>Normalizing all inputs text.</li> <li>The arguments <code>special_tokens</code> and the function <code>set_special_tokens</code>, can be used to add additional symbols (like | |
| ”<strong>classify</strong>”) to a vocabulary.</li> <li>The argument <code>langs</code> defines a pair of languages.</li>`,mt,ve,Ht=`This tokenizer inherits from <a href="/docs/transformers/pr_37177/en/main_classes/tokenizer#transformers.PreTrainedTokenizer">PreTrainedTokenizer</a> which contains most of the main methods. Users should refer to | |
| this superclass for more information regarding those methods.`,ht,E,re,ut,we,Ot=`Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and | |
| adding special tokens. A FAIRSEQ Transformer sequence has the following format:`,ft,Fe,Zt="<li>single sequence: <code><s> X </s></code></li> <li>pair of sequences: <code><s> A </s> B </s></code></li>",_t,G,de,gt,$e,Xt=`Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding | |
| special tokens using the tokenizer <code>prepare_for_model</code> method.`,Tt,C,ie,bt,xe,Qt="Create a mask from the two sequences passed to be used in a sequence-pair classification task. A FAIRSEQ",kt,J,Mt,Ce,Yt="If <code>token_ids_1</code> is <code>None</code>, this method only returns the first portion of the mask (0s).",yt,Se,Kt=`Creates a mask from the two sequences passed to be used in a sequence-pair classification task. An | |
| FAIRSEQ_TRANSFORMER sequence pair mask has the following format:`,vt,ze,le,Ye,ce,Ke,F,pe,wt,qe,eo="The bare FSMT Model outputting raw hidden-states without any specific head on top.",Ft,je,to=`This model inherits from <a href="/docs/transformers/pr_37177/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,$t,Ie,oo=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,xt,q,me,Ct,Le,no='The <a href="/docs/transformers/pr_37177/en/model_doc/fsmt#transformers.FSMTModel">FSMTModel</a> forward method, overrides the <code>__call__</code> special method.',St,R,zt,B,et,he,tt,$,ue,qt,Ue,so="The FSMT Model with a language modeling head. Can be used for summarization.",jt,Ee,ao=`This model inherits from <a href="/docs/transformers/pr_37177/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,It,De,ro=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,Lt,j,fe,Ut,We,io='The <a href="/docs/transformers/pr_37177/en/model_doc/fsmt#transformers.FSMTForConditionalGeneration">FSMTForConditionalGeneration</a> forward method, overrides the <code>__call__</code> special method.',Et,P,Dt,H,ot,_e,nt,Ae,st;return k=new Te({props:{title:"FSMT",local:"fsmt",headingTag:"h1"}}),M=new Te({props:{title:"Overview",local:"overview",headingTag:"h2"}}),ee=new Te({props:{title:"Implementation Notes",local:"implementation-notes",headingTag:"h2"}}),oe=new Te({props:{title:"FSMTConfig",local:"transformers.FSMTConfig",headingTag:"h2"}}),ne=new V({props:{name:"class transformers.FSMTConfig",anchor:"transformers.FSMTConfig",parameters:[{name:"langs",val:" = ['en', 'de']"},{name:"src_vocab_size",val:" = 42024"},{name:"tgt_vocab_size",val:" = 42024"},{name:"activation_function",val:" = 'relu'"},{name:"d_model",val:" = 1024"},{name:"max_length",val:" = 200"},{name:"max_position_embeddings",val:" = 1024"},{name:"encoder_ffn_dim",val:" = 4096"},{name:"encoder_layers",val:" = 12"},{name:"encoder_attention_heads",val:" = 16"},{name:"encoder_layerdrop",val:" = 0.0"},{name:"decoder_ffn_dim",val:" = 4096"},{name:"decoder_layers",val:" = 12"},{name:"decoder_attention_heads",val:" = 16"},{name:"decoder_layerdrop",val:" = 0.0"},{name:"attention_dropout",val:" = 0.0"},{name:"dropout",val:" = 0.1"},{name:"activation_dropout",val:" = 0.0"},{name:"init_std",val:" = 0.02"},{name:"decoder_start_token_id",val:" = 2"},{name:"is_encoder_decoder",val:" = True"},{name:"scale_embedding",val:" = True"},{name:"tie_word_embeddings",val:" = False"},{name:"num_beams",val:" = 5"},{name:"length_penalty",val:" = 1.0"},{name:"early_stopping",val:" = False"},{name:"use_cache",val:" = True"},{name:"pad_token_id",val:" = 1"},{name:"bos_token_id",val:" = 0"},{name:"eos_token_id",val:" = 2"},{name:"forced_eos_token_id",val:" = 2"},{name:"**common_kwargs",val:""}],parametersDescription:[{anchor:"transformers.FSMTConfig.langs",description:`<strong>langs</strong> (<code>List[str]</code>) — | |
| A list with source language and target_language (e.g., [‘en’, ‘ru’]).`,name:"langs"},{anchor:"transformers.FSMTConfig.src_vocab_size",description:`<strong>src_vocab_size</strong> (<code>int</code>) — | |
| Vocabulary size of the encoder. Defines the number of different tokens that can be represented by the | |
| <code>inputs_ids</code> passed to the forward method in the encoder.`,name:"src_vocab_size"},{anchor:"transformers.FSMTConfig.tgt_vocab_size",description:`<strong>tgt_vocab_size</strong> (<code>int</code>) — | |
| Vocabulary size of the decoder. Defines the number of different tokens that can be represented by the | |
| <code>inputs_ids</code> passed to the forward method in the decoder.`,name:"tgt_vocab_size"},{anchor:"transformers.FSMTConfig.d_model",description:`<strong>d_model</strong> (<code>int</code>, <em>optional</em>, defaults to 1024) — | |
| Dimensionality of the layers and the pooler layer.`,name:"d_model"},{anchor:"transformers.FSMTConfig.encoder_layers",description:`<strong>encoder_layers</strong> (<code>int</code>, <em>optional</em>, defaults to 12) — | |
| Number of encoder layers.`,name:"encoder_layers"},{anchor:"transformers.FSMTConfig.decoder_layers",description:`<strong>decoder_layers</strong> (<code>int</code>, <em>optional</em>, defaults to 12) — | |
| Number of decoder layers.`,name:"decoder_layers"},{anchor:"transformers.FSMTConfig.encoder_attention_heads",description:`<strong>encoder_attention_heads</strong> (<code>int</code>, <em>optional</em>, defaults to 16) — | |
| Number of attention heads for each attention layer in the Transformer encoder.`,name:"encoder_attention_heads"},{anchor:"transformers.FSMTConfig.decoder_attention_heads",description:`<strong>decoder_attention_heads</strong> (<code>int</code>, <em>optional</em>, defaults to 16) — | |
| Number of attention heads for each attention layer in the Transformer decoder.`,name:"decoder_attention_heads"},{anchor:"transformers.FSMTConfig.decoder_ffn_dim",description:`<strong>decoder_ffn_dim</strong> (<code>int</code>, <em>optional</em>, defaults to 4096) — | |
| Dimensionality of the “intermediate” (often named feed-forward) layer in decoder.`,name:"decoder_ffn_dim"},{anchor:"transformers.FSMTConfig.encoder_ffn_dim",description:`<strong>encoder_ffn_dim</strong> (<code>int</code>, <em>optional</em>, defaults to 4096) — | |
| Dimensionality of the “intermediate” (often named feed-forward) layer in decoder.`,name:"encoder_ffn_dim"},{anchor:"transformers.FSMTConfig.activation_function",description:`<strong>activation_function</strong> (<code>str</code> or <code>Callable</code>, <em>optional</em>, defaults to <code>"relu"</code>) — | |
| The non-linear activation function (function or string) in the encoder and pooler. If string, <code>"gelu"</code>, | |
| <code>"relu"</code>, <code>"silu"</code> and <code>"gelu_new"</code> are supported.`,name:"activation_function"},{anchor:"transformers.FSMTConfig.dropout",description:`<strong>dropout</strong> (<code>float</code>, <em>optional</em>, defaults to 0.1) — | |
| The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.`,name:"dropout"},{anchor:"transformers.FSMTConfig.attention_dropout",description:`<strong>attention_dropout</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The dropout ratio for the attention probabilities.`,name:"attention_dropout"},{anchor:"transformers.FSMTConfig.activation_dropout",description:`<strong>activation_dropout</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The dropout ratio for activations inside the fully connected layer.`,name:"activation_dropout"},{anchor:"transformers.FSMTConfig.max_position_embeddings",description:`<strong>max_position_embeddings</strong> (<code>int</code>, <em>optional</em>, defaults to 1024) — | |
| The maximum sequence length that this model might ever be used with. Typically set this to something large | |
| just in case (e.g., 512 or 1024 or 2048).`,name:"max_position_embeddings"},{anchor:"transformers.FSMTConfig.init_std",description:`<strong>init_std</strong> (<code>float</code>, <em>optional</em>, defaults to 0.02) — | |
| The standard deviation of the truncated_normal_initializer for initializing all weight matrices.`,name:"init_std"},{anchor:"transformers.FSMTConfig.scale_embedding",description:`<strong>scale_embedding</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Scale embeddings by diving by sqrt(d_model).`,name:"scale_embedding"},{anchor:"transformers.FSMTConfig.bos_token_id",description:`<strong>bos_token_id</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Beginning of stream token id.`,name:"bos_token_id"},{anchor:"transformers.FSMTConfig.pad_token_id",description:`<strong>pad_token_id</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Padding token id.`,name:"pad_token_id"},{anchor:"transformers.FSMTConfig.eos_token_id",description:`<strong>eos_token_id</strong> (<code>int</code>, <em>optional</em>, defaults to 2) — | |
| End of stream token id.`,name:"eos_token_id"},{anchor:"transformers.FSMTConfig.decoder_start_token_id",description:`<strong>decoder_start_token_id</strong> (<code>int</code>, <em>optional</em>) — | |
| This model starts decoding with <code>eos_token_id</code>`,name:"decoder_start_token_id"},{anchor:"transformers.FSMTConfig.encoder_layerdrop",description:`<strong>encoder_layerdrop</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| Google “layerdrop arxiv”, as its not explainable in one line.`,name:"encoder_layerdrop"},{anchor:"transformers.FSMTConfig.decoder_layerdrop",description:`<strong>decoder_layerdrop</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| Google “layerdrop arxiv”, as its not explainable in one line.`,name:"decoder_layerdrop"},{anchor:"transformers.FSMTConfig.is_encoder_decoder",description:`<strong>is_encoder_decoder</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether this is an encoder/decoder model.`,name:"is_encoder_decoder"},{anchor:"transformers.FSMTConfig.tie_word_embeddings",description:`<strong>tie_word_embeddings</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to tie input and output embeddings.`,name:"tie_word_embeddings"},{anchor:"transformers.FSMTConfig.num_beams",description:`<strong>num_beams</strong> (<code>int</code>, <em>optional</em>, defaults to 5) — | |
| Number of beams for beam search that will be used by default in the <code>generate</code> method of the model. 1 means | |
| no beam search.`,name:"num_beams"},{anchor:"transformers.FSMTConfig.length_penalty",description:`<strong>length_penalty</strong> (<code>float</code>, <em>optional</em>, defaults to 1) — | |
| Exponential penalty to the length that is used with beam-based generation. It is applied as an exponent to | |
| the sequence length, which in turn is used to divide the score of the sequence. Since the score is the log | |
| likelihood of the sequence (i.e. negative), <code>length_penalty</code> > 0.0 promotes longer sequences, while | |
| <code>length_penalty</code> < 0.0 encourages shorter sequences.`,name:"length_penalty"},{anchor:"transformers.FSMTConfig.early_stopping",description:`<strong>early_stopping</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Flag that will be used by default in the <code>generate</code> method of the model. Whether to stop the beam search | |
| when at least <code>num_beams</code> sentences are finished per batch or not.`,name:"early_stopping"},{anchor:"transformers.FSMTConfig.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not the model should return the last key/values attentions (not used by all models).`,name:"use_cache"},{anchor:"transformers.FSMTConfig.forced_eos_token_id",description:`<strong>forced_eos_token_id</strong> (<code>int</code>, <em>optional</em>, defaults to 2) — | |
| The id of the token to force as the last generated token when <code>max_length</code> is reached. Usually set to | |
| <code>eos_token_id</code>.`,name:"forced_eos_token_id"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/configuration_fsmt.py#L37"}}),N=new at({props:{anchor:"transformers.FSMTConfig.example",$$slots:{default:[_o]},$$scope:{ctx:x}}}),se=new Te({props:{title:"FSMTTokenizer",local:"transformers.FSMTTokenizer",headingTag:"h2"}}),ae=new V({props:{name:"class transformers.FSMTTokenizer",anchor:"transformers.FSMTTokenizer",parameters:[{name:"langs",val:" = None"},{name:"src_vocab_file",val:" = None"},{name:"tgt_vocab_file",val:" = None"},{name:"merges_file",val:" = None"},{name:"do_lower_case",val:" = False"},{name:"unk_token",val:" = '<unk>'"},{name:"bos_token",val:" = '<s>'"},{name:"sep_token",val:" = '</s>'"},{name:"pad_token",val:" = '<pad>'"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.FSMTTokenizer.langs",description:`<strong>langs</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| A list of two languages to translate from and to, for instance <code>["en", "ru"]</code>.`,name:"langs"},{anchor:"transformers.FSMTTokenizer.src_vocab_file",description:`<strong>src_vocab_file</strong> (<code>str</code>, <em>optional</em>) — | |
| File containing the vocabulary for the source language.`,name:"src_vocab_file"},{anchor:"transformers.FSMTTokenizer.tgt_vocab_file",description:`<strong>tgt_vocab_file</strong> (<code>st</code>, <em>optional</em>) — | |
| File containing the vocabulary for the target language.`,name:"tgt_vocab_file"},{anchor:"transformers.FSMTTokenizer.merges_file",description:`<strong>merges_file</strong> (<code>str</code>, <em>optional</em>) — | |
| File containing the merges.`,name:"merges_file"},{anchor:"transformers.FSMTTokenizer.do_lower_case",description:`<strong>do_lower_case</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to lowercase the input when tokenizing.`,name:"do_lower_case"},{anchor:"transformers.FSMTTokenizer.unk_token",description:`<strong>unk_token</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"<unk>"</code>) — | |
| The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this | |
| token instead.`,name:"unk_token"},{anchor:"transformers.FSMTTokenizer.bos_token",description:`<strong>bos_token</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"<s>"</code>) — | |
| The beginning of sequence token that was used during pretraining. Can be used a sequence classifier token.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p>When building a sequence using special tokens, this is not the token that is used for the beginning of | |
| sequence. The token used is the <code>cls_token</code>.</p> | |
| </div>`,name:"bos_token"},{anchor:"transformers.FSMTTokenizer.sep_token",description:`<strong>sep_token</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"</s>"</code>) — | |
| The separator token, which is used when building a sequence from multiple sequences, e.g. two sequences for | |
| sequence classification or for a text and a question for question answering. It is also used as the last | |
| token of a sequence built with special tokens.`,name:"sep_token"},{anchor:"transformers.FSMTTokenizer.pad_token",description:`<strong>pad_token</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"<pad>"</code>) — | |
| The token used for padding, for example when batching sequences of different lengths.`,name:"pad_token"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/tokenization_fsmt.py#L114"}}),re=new V({props:{name:"build_inputs_with_special_tokens",anchor:"transformers.FSMTTokenizer.build_inputs_with_special_tokens",parameters:[{name:"token_ids_0",val:": typing.List[int]"},{name:"token_ids_1",val:": typing.Optional[typing.List[int]] = None"}],parametersDescription:[{anchor:"transformers.FSMTTokenizer.build_inputs_with_special_tokens.token_ids_0",description:`<strong>token_ids_0</strong> (<code>List[int]</code>) — | |
| List of IDs to which the special tokens will be added.`,name:"token_ids_0"},{anchor:"transformers.FSMTTokenizer.build_inputs_with_special_tokens.token_ids_1",description:`<strong>token_ids_1</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| Optional second list of IDs for sequence pairs.`,name:"token_ids_1"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/tokenization_fsmt.py#L379",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>List of <a href="../glossary#input-ids">input IDs</a> with the appropriate special tokens.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[int]</code></p> | |
| `}}),de=new V({props:{name:"get_special_tokens_mask",anchor:"transformers.FSMTTokenizer.get_special_tokens_mask",parameters:[{name:"token_ids_0",val:": typing.List[int]"},{name:"token_ids_1",val:": typing.Optional[typing.List[int]] = None"},{name:"already_has_special_tokens",val:": bool = False"}],parametersDescription:[{anchor:"transformers.FSMTTokenizer.get_special_tokens_mask.token_ids_0",description:`<strong>token_ids_0</strong> (<code>List[int]</code>) — | |
| List of IDs.`,name:"token_ids_0"},{anchor:"transformers.FSMTTokenizer.get_special_tokens_mask.token_ids_1",description:`<strong>token_ids_1</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| Optional second list of IDs for sequence pairs.`,name:"token_ids_1"},{anchor:"transformers.FSMTTokenizer.get_special_tokens_mask.already_has_special_tokens",description:`<strong>already_has_special_tokens</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the token list is already formatted with special tokens for the model.`,name:"already_has_special_tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/tokenization_fsmt.py#L405",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[int]</code></p> | |
| `}}),ie=new V({props:{name:"create_token_type_ids_from_sequences",anchor:"transformers.FSMTTokenizer.create_token_type_ids_from_sequences",parameters:[{name:"token_ids_0",val:": typing.List[int]"},{name:"token_ids_1",val:": typing.Optional[typing.List[int]] = None"}],parametersDescription:[{anchor:"transformers.FSMTTokenizer.create_token_type_ids_from_sequences.token_ids_0",description:`<strong>token_ids_0</strong> (<code>List[int]</code>) — | |
| List of IDs.`,name:"token_ids_0"},{anchor:"transformers.FSMTTokenizer.create_token_type_ids_from_sequences.token_ids_1",description:`<strong>token_ids_1</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| Optional second list of IDs for sequence pairs.`,name:"token_ids_1"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/tokenization_fsmt.py#L433",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>List of <a href="../glossary#token-type-ids">token type IDs</a> according to the given sequence(s).</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[int]</code></p> | |
| `}}),J=new at({props:{anchor:"transformers.FSMTTokenizer.create_token_type_ids_from_sequences.example",$$slots:{default:[go]},$$scope:{ctx:x}}}),le=new V({props:{name:"save_vocabulary",anchor:"transformers.FSMTTokenizer.save_vocabulary",parameters:[{name:"save_directory",val:": str"},{name:"filename_prefix",val:": typing.Optional[str] = None"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/tokenization_fsmt.py#L466"}}),ce=new Te({props:{title:"FSMTModel",local:"transformers.FSMTModel",headingTag:"h2"}}),pe=new V({props:{name:"class transformers.FSMTModel",anchor:"transformers.FSMTModel",parameters:[{name:"config",val:": FSMTConfig"}],parametersDescription:[{anchor:"transformers.FSMTModel.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_37177/en/model_doc/fsmt#transformers.FSMTConfig">FSMTConfig</a>) — Model configuration class with all the parameters of the model. | |
| Initializing with a config file does not load the weights associated with the model, only the | |
| configuration. Check out the <a href="/docs/transformers/pr_37177/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/modeling_fsmt.py#L1035"}}),me=new V({props:{name:"forward",anchor:"transformers.FSMTModel.forward",parameters:[{name:"input_ids",val:": LongTensor"},{name:"attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"decoder_input_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"decoder_attention_mask",val:": typing.Optional[torch.BoolTensor] = None"},{name:"head_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"decoder_head_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"cross_attn_head_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"encoder_outputs",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"past_key_values",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"use_cache",val:": typing.Optional[bool] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"decoder_inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"}],parametersDescription:[{anchor:"transformers.FSMTModel.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>) — | |
| Indices of input sequence tokens in the vocabulary.</p> | |
| <p>Indices can be obtained using <code>FSTMTokenizer</code>. See <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.FSMTModel.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"attention_mask"},{anchor:"transformers.FSMTModel.forward.decoder_input_ids",description:`<strong>decoder_input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, target_sequence_length)</code>, <em>optional</em>) — | |
| Indices of decoder input sequence tokens in the vocabulary.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_37177/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#decoder-input-ids">What are decoder input IDs?</a></p> | |
| <p>FSMT uses the <code>eos_token_id</code> as the starting token for <code>decoder_input_ids</code> generation. If <code>past_key_values</code> | |
| is used, optionally only the last <code>decoder_input_ids</code> have to be input (see <code>past_key_values</code>).`,name:"decoder_input_ids"},{anchor:"transformers.FSMTModel.forward.decoder_attention_mask",description:`<strong>decoder_attention_mask</strong> (<code>torch.BoolTensor</code> of shape <code>(batch_size, target_sequence_length)</code>, <em>optional</em>) — | |
| Default behavior: generate a tensor that ignores pad tokens in <code>decoder_input_ids</code>. Causal mask will also | |
| be used by default.`,name:"decoder_attention_mask"},{anchor:"transformers.FSMTModel.forward.head_mask",description:`<strong>head_mask</strong> (<code>torch.Tensor</code> of shape <code>(encoder_layers, encoder_attention_heads)</code>, <em>optional</em>) — | |
| Mask to nullify selected heads of the attention modules in the encoder. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"head_mask"},{anchor:"transformers.FSMTModel.forward.decoder_head_mask",description:`<strong>decoder_head_mask</strong> (<code>torch.Tensor</code> of shape <code>(decoder_layers, decoder_attention_heads)</code>, <em>optional</em>) — | |
| Mask to nullify selected heads of the attention modules in the decoder. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"decoder_head_mask"},{anchor:"transformers.FSMTModel.forward.cross_attn_head_mask",description:`<strong>cross_attn_head_mask</strong> (<code>torch.Tensor</code> of shape <code>(decoder_layers, decoder_attention_heads)</code>, <em>optional</em>) — | |
| Mask to nullify selected heads of the cross-attention modules in the decoder. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"cross_attn_head_mask"},{anchor:"transformers.FSMTModel.forward.encoder_outputs",description:`<strong>encoder_outputs</strong> (<code>Tuple(torch.FloatTensor)</code>, <em>optional</em>) — | |
| Tuple consists of (<code>last_hidden_state</code>, <em>optional</em>: <code>hidden_states</code>, <em>optional</em>: <code>attentions</code>) | |
| <code>last_hidden_state</code> of shape <code>(batch_size, sequence_length, hidden_size)</code> is a sequence of hidden-states at | |
| the output of the last layer of the encoder. Used in the cross-attention of the decoder.`,name:"encoder_outputs"},{anchor:"transformers.FSMTModel.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>Tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code> with each tuple having 4 tensors of shape <code>(batch_size, num_heads, sequence_length - 1, embed_size_per_head)</code>) — | |
| Contains precomputed key and value hidden-states of the attention blocks. Can be used to speed up decoding. | |
| If <code>past_key_values</code> are used, the user can optionally input only the last <code>decoder_input_ids</code> (those that | |
| don’t have their past key value states given to this model) of shape <code>(batch_size, 1)</code> instead of all | |
| <code>decoder_input_ids</code> of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.FSMTModel.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.FSMTModel.forward.decoder_inputs_embeds",description:`<strong>decoder_inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, target_sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>decoder_input_ids</code> you can choose to directly pass an embedded | |
| representation. If <code>past_key_values</code> is used, optionally only the last <code>decoder_inputs_embeds</code> have to be | |
| input (see <code>past_key_values</code>). This is useful if you want more control over how to convert | |
| <code>decoder_input_ids</code> indices into associated vectors than the model’s internal embedding lookup matrix.</p> | |
| <p>If <code>decoder_input_ids</code> and <code>decoder_inputs_embeds</code> are both unset, <code>decoder_inputs_embeds</code> takes the value | |
| of <code>inputs_embeds</code>.`,name:"decoder_inputs_embeds"},{anchor:"transformers.FSMTModel.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.FSMTModel.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.FSMTModel.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.FSMTModel.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_37177/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/modeling_fsmt.py#L1066",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_37177/en/main_classes/output#transformers.modeling_outputs.Seq2SeqModelOutput" | |
| >transformers.modeling_outputs.Seq2SeqModelOutput</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_37177/en/model_doc/fsmt#transformers.FSMTConfig" | |
| >FSMTConfig</a>) and inputs.</p> | |
| <ul> | |
| <li> | |
| <p><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — Sequence of hidden-states at the output of the last layer of the decoder of the model.</p> | |
| <p>If <code>past_key_values</code> is used only the last hidden-state of the sequences of shape <code>(batch_size, 1, hidden_size)</code> is output.</p> | |
| </li> | |
| <li> | |
| <p><strong>past_key_values</strong> (<code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of shape | |
| <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>) and 2 additional tensors of shape | |
| <code>(batch_size, num_heads, encoder_sequence_length, embed_size_per_head)</code>.</p> | |
| <p>Contains pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used (see <code>past_key_values</code> input) to speed up sequential decoding.</p> | |
| </li> | |
| <li> | |
| <p><strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> | |
| <p>Hidden-states of the decoder at the output of each layer plus the optional initial embedding outputs.</p> | |
| </li> | |
| <li> | |
| <p><strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights of the decoder, after the attention softmax, used to compute the weighted average in the | |
| self-attention heads.</p> | |
| </li> | |
| <li> | |
| <p><strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights of the decoder’s cross-attention layer, after the attention softmax, used to compute the | |
| weighted average in the cross-attention heads.</p> | |
| </li> | |
| <li> | |
| <p><strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the encoder of the model.</p> | |
| </li> | |
| <li> | |
| <p><strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> | |
| <p>Hidden-states of the encoder at the output of each layer plus the optional initial embedding outputs.</p> | |
| </li> | |
| <li> | |
| <p><strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights of the encoder, after the attention softmax, used to compute the weighted average in the | |
| self-attention heads.</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_37177/en/main_classes/output#transformers.modeling_outputs.Seq2SeqModelOutput" | |
| >transformers.modeling_outputs.Seq2SeqModelOutput</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),R=new lo({props:{$$slots:{default:[To]},$$scope:{ctx:x}}}),B=new at({props:{anchor:"transformers.FSMTModel.forward.example",$$slots:{default:[bo]},$$scope:{ctx:x}}}),he=new Te({props:{title:"FSMTForConditionalGeneration",local:"transformers.FSMTForConditionalGeneration",headingTag:"h2"}}),ue=new V({props:{name:"class transformers.FSMTForConditionalGeneration",anchor:"transformers.FSMTForConditionalGeneration",parameters:[{name:"config",val:": FSMTConfig"}],parametersDescription:[{anchor:"transformers.FSMTForConditionalGeneration.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_37177/en/model_doc/fsmt#transformers.FSMTConfig">FSMTConfig</a>) — Model configuration class with all the parameters of the model. | |
| Initializing with a config file does not load the weights associated with the model, only the | |
| configuration. Check out the <a href="/docs/transformers/pr_37177/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/modeling_fsmt.py#L1177"}}),fe=new V({props:{name:"forward",anchor:"transformers.FSMTForConditionalGeneration.forward",parameters:[{name:"input_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"decoder_input_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"decoder_attention_mask",val:": typing.Optional[torch.BoolTensor] = None"},{name:"head_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"decoder_head_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"cross_attn_head_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"encoder_outputs",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"past_key_values",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"decoder_inputs_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"labels",val:": typing.Optional[torch.LongTensor] = None"},{name:"use_cache",val:": typing.Optional[bool] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"}],parametersDescription:[{anchor:"transformers.FSMTForConditionalGeneration.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>) — | |
| Indices of input sequence tokens in the vocabulary.</p> | |
| <p>Indices can be obtained using <code>FSTMTokenizer</code>. See <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.FSMTForConditionalGeneration.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"attention_mask"},{anchor:"transformers.FSMTForConditionalGeneration.forward.decoder_input_ids",description:`<strong>decoder_input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, target_sequence_length)</code>, <em>optional</em>) — | |
| Indices of decoder input sequence tokens in the vocabulary.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_37177/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_37177/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#decoder-input-ids">What are decoder input IDs?</a></p> | |
| <p>FSMT uses the <code>eos_token_id</code> as the starting token for <code>decoder_input_ids</code> generation. If <code>past_key_values</code> | |
| is used, optionally only the last <code>decoder_input_ids</code> have to be input (see <code>past_key_values</code>).`,name:"decoder_input_ids"},{anchor:"transformers.FSMTForConditionalGeneration.forward.decoder_attention_mask",description:`<strong>decoder_attention_mask</strong> (<code>torch.BoolTensor</code> of shape <code>(batch_size, target_sequence_length)</code>, <em>optional</em>) — | |
| Default behavior: generate a tensor that ignores pad tokens in <code>decoder_input_ids</code>. Causal mask will also | |
| be used by default.`,name:"decoder_attention_mask"},{anchor:"transformers.FSMTForConditionalGeneration.forward.head_mask",description:`<strong>head_mask</strong> (<code>torch.Tensor</code> of shape <code>(encoder_layers, encoder_attention_heads)</code>, <em>optional</em>) — | |
| Mask to nullify selected heads of the attention modules in the encoder. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"head_mask"},{anchor:"transformers.FSMTForConditionalGeneration.forward.decoder_head_mask",description:`<strong>decoder_head_mask</strong> (<code>torch.Tensor</code> of shape <code>(decoder_layers, decoder_attention_heads)</code>, <em>optional</em>) — | |
| Mask to nullify selected heads of the attention modules in the decoder. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"decoder_head_mask"},{anchor:"transformers.FSMTForConditionalGeneration.forward.cross_attn_head_mask",description:`<strong>cross_attn_head_mask</strong> (<code>torch.Tensor</code> of shape <code>(decoder_layers, decoder_attention_heads)</code>, <em>optional</em>) — | |
| Mask to nullify selected heads of the cross-attention modules in the decoder. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"cross_attn_head_mask"},{anchor:"transformers.FSMTForConditionalGeneration.forward.encoder_outputs",description:`<strong>encoder_outputs</strong> (<code>Tuple(torch.FloatTensor)</code>, <em>optional</em>) — | |
| Tuple consists of (<code>last_hidden_state</code>, <em>optional</em>: <code>hidden_states</code>, <em>optional</em>: <code>attentions</code>) | |
| <code>last_hidden_state</code> of shape <code>(batch_size, sequence_length, hidden_size)</code> is a sequence of hidden-states at | |
| the output of the last layer of the encoder. Used in the cross-attention of the decoder.`,name:"encoder_outputs"},{anchor:"transformers.FSMTForConditionalGeneration.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>Tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code> with each tuple having 4 tensors of shape <code>(batch_size, num_heads, sequence_length - 1, embed_size_per_head)</code>) — | |
| Contains precomputed key and value hidden-states of the attention blocks. Can be used to speed up decoding. | |
| If <code>past_key_values</code> are used, the user can optionally input only the last <code>decoder_input_ids</code> (those that | |
| don’t have their past key value states given to this model) of shape <code>(batch_size, 1)</code> instead of all | |
| <code>decoder_input_ids</code> of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.FSMTForConditionalGeneration.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.FSMTForConditionalGeneration.forward.decoder_inputs_embeds",description:`<strong>decoder_inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, target_sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>decoder_input_ids</code> you can choose to directly pass an embedded | |
| representation. If <code>past_key_values</code> is used, optionally only the last <code>decoder_inputs_embeds</code> have to be | |
| input (see <code>past_key_values</code>). This is useful if you want more control over how to convert | |
| <code>decoder_input_ids</code> indices into associated vectors than the model’s internal embedding lookup matrix.</p> | |
| <p>If <code>decoder_input_ids</code> and <code>decoder_inputs_embeds</code> are both unset, <code>decoder_inputs_embeds</code> takes the value | |
| of <code>inputs_embeds</code>.`,name:"decoder_inputs_embeds"},{anchor:"transformers.FSMTForConditionalGeneration.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.FSMTForConditionalGeneration.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.FSMTForConditionalGeneration.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.FSMTForConditionalGeneration.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_37177/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.FSMTForConditionalGeneration.forward.labels",description:`<strong>labels</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Labels for computing the masked language modeling loss. Indices should either be in <code>[0, ..., config.vocab_size]</code> or -100 (see <code>input_ids</code> docstring). Tokens with indices set to <code>-100</code> are ignored | |
| (masked), the loss is only computed for the tokens with labels in <code>[0, ..., config.vocab_size]</code>.`,name:"labels"}],source:"https://github.com/huggingface/transformers/blob/vr_37177/src/transformers/models/fsmt/modeling_fsmt.py#L1192",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_37177/en/main_classes/output#transformers.modeling_outputs.Seq2SeqLMOutput" | |
| >transformers.modeling_outputs.Seq2SeqLMOutput</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_37177/en/model_doc/fsmt#transformers.FSMTConfig" | |
| >FSMTConfig</a>) and inputs.</p> | |
| <ul> | |
| <li> | |
| <p><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> is provided) — Language modeling loss.</p> | |
| </li> | |
| <li> | |
| <p><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, config.vocab_size)</code>) — Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).</p> | |
| </li> | |
| <li> | |
| <p><strong>past_key_values</strong> (<code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of shape | |
| <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>) and 2 additional tensors of shape | |
| <code>(batch_size, num_heads, encoder_sequence_length, embed_size_per_head)</code>.</p> | |
| <p>Contains pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used (see <code>past_key_values</code> input) to speed up sequential decoding.</p> | |
| </li> | |
| <li> | |
| <p><strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> | |
| <p>Hidden-states of the decoder at the output of each layer plus the initial embedding outputs.</p> | |
| </li> | |
| <li> | |
| <p><strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights of the decoder, after the attention softmax, used to compute the weighted average in the | |
| self-attention heads.</p> | |
| </li> | |
| <li> | |
| <p><strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights of the decoder’s cross-attention layer, after the attention softmax, used to compute the | |
| weighted average in the cross-attention heads.</p> | |
| </li> | |
| <li> | |
| <p><strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the encoder of the model.</p> | |
| </li> | |
| <li> | |
| <p><strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> | |
| <p>Hidden-states of the encoder at the output of each layer plus the initial embedding outputs.</p> | |
| </li> | |
| <li> | |
| <p><strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights of the encoder, after the attention softmax, used to compute the weighted average in the | |
| self-attention heads.</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_37177/en/main_classes/output#transformers.modeling_outputs.Seq2SeqLMOutput" | |
| >transformers.modeling_outputs.Seq2SeqLMOutput</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),P=new lo({props:{$$slots:{default:[ko]},$$scope:{ctx:x}}}),H=new at({props:{anchor:"transformers.FSMTForConditionalGeneration.forward.example",$$slots:{default:[Mo]},$$scope:{ctx:x}}}),_e=new fo({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/fsmt.md"}}),{c(){o=i("meta"),y=a(),m=i("p"),c=a(),u(k.$$.fragment),t=a(),u(M.$$.fragment),Ge=a(),X=i("p"),X.innerHTML=Wt,Je=a(),Q=i("p"),Q.textContent=Vt,Re=a(),Y=i("p"),Y.innerHTML=At,Be=a(),K=i("p"),K.innerHTML=Nt,Pe=a(),u(ee.$$.fragment),He=a(),te=i("ul"),te.innerHTML=Gt,Oe=a(),u(oe.$$.fragment),Ze=a(),S=i("div"),u(ne.$$.fragment),dt=a(),be=i("p"),be.innerHTML=Jt,it=a(),ke=i("p"),ke.innerHTML=Rt,lt=a(),u(N.$$.fragment),Xe=a(),u(se.$$.fragment),Qe=a(),v=i("div"),u(ae.$$.fragment),ct=a(),Me=i("p"),Me.textContent=Bt,pt=a(),ye=i("ul"),ye.innerHTML=Pt,mt=a(),ve=i("p"),ve.innerHTML=Ht,ht=a(),E=i("div"),u(re.$$.fragment),ut=a(),we=i("p"),we.textContent=Ot,ft=a(),Fe=i("ul"),Fe.innerHTML=Zt,_t=a(),G=i("div"),u(de.$$.fragment),gt=a(),$e=i("p"),$e.innerHTML=Xt,Tt=a(),C=i("div"),u(ie.$$.fragment),bt=a(),xe=i("p"),xe.textContent=Qt,kt=a(),u(J.$$.fragment),Mt=a(),Ce=i("p"),Ce.innerHTML=Yt,yt=a(),Se=i("p"),Se.textContent=Kt,vt=a(),ze=i("div"),u(le.$$.fragment),Ye=a(),u(ce.$$.fragment),Ke=a(),F=i("div"),u(pe.$$.fragment),wt=a(),qe=i("p"),qe.textContent=eo,Ft=a(),je=i("p"),je.innerHTML=to,$t=a(),Ie=i("p"),Ie.innerHTML=oo,xt=a(),q=i("div"),u(me.$$.fragment),Ct=a(),Le=i("p"),Le.innerHTML=no,St=a(),u(R.$$.fragment),zt=a(),u(B.$$.fragment),et=a(),u(he.$$.fragment),tt=a(),$=i("div"),u(ue.$$.fragment),qt=a(),Ue=i("p"),Ue.textContent=so,jt=a(),Ee=i("p"),Ee.innerHTML=ao,It=a(),De=i("p"),De.innerHTML=ro,Lt=a(),j=i("div"),u(fe.$$.fragment),Ut=a(),We=i("p"),We.innerHTML=io,Et=a(),u(P.$$.fragment),Dt=a(),u(H.$$.fragment),ot=a(),u(_e.$$.fragment),nt=a(),Ae=i("p"),this.h()},l(e){const s=uo("svelte-u9bgzb",document.head);o=l(s,"META",{name:!0,content:!0}),s.forEach(d),y=r(e),m=l(e,"P",{}),I(m).forEach(d),c=r(e),f(k.$$.fragment,e),t=r(e),f(M.$$.fragment,e),Ge=r(e),X=l(e,"P",{"data-svelte-h":!0}),h(X)!=="svelte-ppl3f1"&&(X.innerHTML=Wt),Je=r(e),Q=l(e,"P",{"data-svelte-h":!0}),h(Q)!=="svelte-wu27l3"&&(Q.textContent=Vt),Re=r(e),Y=l(e,"P",{"data-svelte-h":!0}),h(Y)!=="svelte-75g4jk"&&(Y.innerHTML=At),Be=r(e),K=l(e,"P",{"data-svelte-h":!0}),h(K)!=="svelte-1uemjgo"&&(K.innerHTML=Nt),Pe=r(e),f(ee.$$.fragment,e),He=r(e),te=l(e,"UL",{"data-svelte-h":!0}),h(te)!=="svelte-7hx4dr"&&(te.innerHTML=Gt),Oe=r(e),f(oe.$$.fragment,e),Ze=r(e),S=l(e,"DIV",{class:!0});var U=I(S);f(ne.$$.fragment,U),dt=r(U),be=l(U,"P",{"data-svelte-h":!0}),h(be)!=="svelte-1lfjrr6"&&(be.innerHTML=Jt),it=r(U),ke=l(U,"P",{"data-svelte-h":!0}),h(ke)!=="svelte-1kj23fz"&&(ke.innerHTML=Rt),lt=r(U),f(N.$$.fragment,U),U.forEach(d),Xe=r(e),f(se.$$.fragment,e),Qe=r(e),v=l(e,"DIV",{class:!0});var w=I(v);f(ae.$$.fragment,w),ct=r(w),Me=l(w,"P",{"data-svelte-h":!0}),h(Me)!=="svelte-5bbjru"&&(Me.textContent=Bt),pt=r(w),ye=l(w,"UL",{"data-svelte-h":!0}),h(ye)!=="svelte-stmybw"&&(ye.innerHTML=Pt),mt=r(w),ve=l(w,"P",{"data-svelte-h":!0}),h(ve)!=="svelte-tpb7rp"&&(ve.innerHTML=Ht),ht=r(w),E=l(w,"DIV",{class:!0});var A=I(E);f(re.$$.fragment,A),ut=r(A),we=l(A,"P",{"data-svelte-h":!0}),h(we)!=="svelte-ym5sov"&&(we.textContent=Ot),ft=r(A),Fe=l(A,"UL",{"data-svelte-h":!0}),h(Fe)!=="svelte-1w73b42"&&(Fe.innerHTML=Zt),A.forEach(d),_t=r(w),G=l(w,"DIV",{class:!0});var ge=I(G);f(de.$$.fragment,ge),gt=r(ge),$e=l(ge,"P",{"data-svelte-h":!0}),h($e)!=="svelte-1f4f5kp"&&($e.innerHTML=Xt),ge.forEach(d),Tt=r(w),C=l(w,"DIV",{class:!0});var z=I(C);f(ie.$$.fragment,z),bt=r(z),xe=l(z,"P",{"data-svelte-h":!0}),h(xe)!=="svelte-13qcmkg"&&(xe.textContent=Qt),kt=r(z),f(J.$$.fragment,z),Mt=r(z),Ce=l(z,"P",{"data-svelte-h":!0}),h(Ce)!=="svelte-owoxgn"&&(Ce.innerHTML=Yt),yt=r(z),Se=l(z,"P",{"data-svelte-h":!0}),h(Se)!=="svelte-1fvbzf1"&&(Se.textContent=Kt),z.forEach(d),vt=r(w),ze=l(w,"DIV",{class:!0});var Ne=I(ze);f(le.$$.fragment,Ne),Ne.forEach(d),w.forEach(d),Ye=r(e),f(ce.$$.fragment,e),Ke=r(e),F=l(e,"DIV",{class:!0});var D=I(F);f(pe.$$.fragment,D),wt=r(D),qe=l(D,"P",{"data-svelte-h":!0}),h(qe)!=="svelte-42ql7q"&&(qe.textContent=eo),Ft=r(D),je=l(D,"P",{"data-svelte-h":!0}),h(je)!=="svelte-1as71mv"&&(je.innerHTML=to),$t=r(D),Ie=l(D,"P",{"data-svelte-h":!0}),h(Ie)!=="svelte-hswkmf"&&(Ie.innerHTML=oo),xt=r(D),q=l(D,"DIV",{class:!0});var O=I(q);f(me.$$.fragment,O),Ct=r(O),Le=l(O,"P",{"data-svelte-h":!0}),h(Le)!=="svelte-1cc8nq4"&&(Le.innerHTML=no),St=r(O),f(R.$$.fragment,O),zt=r(O),f(B.$$.fragment,O),O.forEach(d),D.forEach(d),et=r(e),f(he.$$.fragment,e),tt=r(e),$=l(e,"DIV",{class:!0});var W=I($);f(ue.$$.fragment,W),qt=r(W),Ue=l(W,"P",{"data-svelte-h":!0}),h(Ue)!=="svelte-1449fju"&&(Ue.textContent=so),jt=r(W),Ee=l(W,"P",{"data-svelte-h":!0}),h(Ee)!=="svelte-1as71mv"&&(Ee.innerHTML=ao),It=r(W),De=l(W,"P",{"data-svelte-h":!0}),h(De)!=="svelte-hswkmf"&&(De.innerHTML=ro),Lt=r(W),j=l(W,"DIV",{class:!0});var Z=I(j);f(fe.$$.fragment,Z),Ut=r(Z),We=l(Z,"P",{"data-svelte-h":!0}),h(We)!=="svelte-c50nsm"&&(We.innerHTML=io),Et=r(Z),f(P.$$.fragment,Z),Dt=r(Z),f(H.$$.fragment,Z),Z.forEach(d),W.forEach(d),ot=r(e),f(_e.$$.fragment,e),nt=r(e),Ae=l(e,"P",{}),I(Ae).forEach(d),this.h()},h(){L(o,"name","hf:doc:metadata"),L(o,"content",vo),L(S,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(E,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(G,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(C,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(ze,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(v,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(q,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(F,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(j,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L($,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(e,s){n(document.head,o),p(e,y,s),p(e,m,s),p(e,c,s),_(k,e,s),p(e,t,s),_(M,e,s),p(e,Ge,s),p(e,X,s),p(e,Je,s),p(e,Q,s),p(e,Re,s),p(e,Y,s),p(e,Be,s),p(e,K,s),p(e,Pe,s),_(ee,e,s),p(e,He,s),p(e,te,s),p(e,Oe,s),_(oe,e,s),p(e,Ze,s),p(e,S,s),_(ne,S,null),n(S,dt),n(S,be),n(S,it),n(S,ke),n(S,lt),_(N,S,null),p(e,Xe,s),_(se,e,s),p(e,Qe,s),p(e,v,s),_(ae,v,null),n(v,ct),n(v,Me),n(v,pt),n(v,ye),n(v,mt),n(v,ve),n(v,ht),n(v,E),_(re,E,null),n(E,ut),n(E,we),n(E,ft),n(E,Fe),n(v,_t),n(v,G),_(de,G,null),n(G,gt),n(G,$e),n(v,Tt),n(v,C),_(ie,C,null),n(C,bt),n(C,xe),n(C,kt),_(J,C,null),n(C,Mt),n(C,Ce),n(C,yt),n(C,Se),n(v,vt),n(v,ze),_(le,ze,null),p(e,Ye,s),_(ce,e,s),p(e,Ke,s),p(e,F,s),_(pe,F,null),n(F,wt),n(F,qe),n(F,Ft),n(F,je),n(F,$t),n(F,Ie),n(F,xt),n(F,q),_(me,q,null),n(q,Ct),n(q,Le),n(q,St),_(R,q,null),n(q,zt),_(B,q,null),p(e,et,s),_(he,e,s),p(e,tt,s),p(e,$,s),_(ue,$,null),n($,qt),n($,Ue),n($,jt),n($,Ee),n($,It),n($,De),n($,Lt),n($,j),_(fe,j,null),n(j,Ut),n(j,We),n(j,Et),_(P,j,null),n(j,Dt),_(H,j,null),p(e,ot,s),_(_e,e,s),p(e,nt,s),p(e,Ae,s),st=!0},p(e,[s]){const U={};s&2&&(U.$$scope={dirty:s,ctx:e}),N.$set(U);const w={};s&2&&(w.$$scope={dirty:s,ctx:e}),J.$set(w);const A={};s&2&&(A.$$scope={dirty:s,ctx:e}),R.$set(A);const ge={};s&2&&(ge.$$scope={dirty:s,ctx:e}),B.$set(ge);const z={};s&2&&(z.$$scope={dirty:s,ctx:e}),P.$set(z);const Ne={};s&2&&(Ne.$$scope={dirty:s,ctx:e}),H.$set(Ne)},i(e){st||(g(k.$$.fragment,e),g(M.$$.fragment,e),g(ee.$$.fragment,e),g(oe.$$.fragment,e),g(ne.$$.fragment,e),g(N.$$.fragment,e),g(se.$$.fragment,e),g(ae.$$.fragment,e),g(re.$$.fragment,e),g(de.$$.fragment,e),g(ie.$$.fragment,e),g(J.$$.fragment,e),g(le.$$.fragment,e),g(ce.$$.fragment,e),g(pe.$$.fragment,e),g(me.$$.fragment,e),g(R.$$.fragment,e),g(B.$$.fragment,e),g(he.$$.fragment,e),g(ue.$$.fragment,e),g(fe.$$.fragment,e),g(P.$$.fragment,e),g(H.$$.fragment,e),g(_e.$$.fragment,e),st=!0)},o(e){T(k.$$.fragment,e),T(M.$$.fragment,e),T(ee.$$.fragment,e),T(oe.$$.fragment,e),T(ne.$$.fragment,e),T(N.$$.fragment,e),T(se.$$.fragment,e),T(ae.$$.fragment,e),T(re.$$.fragment,e),T(de.$$.fragment,e),T(ie.$$.fragment,e),T(J.$$.fragment,e),T(le.$$.fragment,e),T(ce.$$.fragment,e),T(pe.$$.fragment,e),T(me.$$.fragment,e),T(R.$$.fragment,e),T(B.$$.fragment,e),T(he.$$.fragment,e),T(ue.$$.fragment,e),T(fe.$$.fragment,e),T(P.$$.fragment,e),T(H.$$.fragment,e),T(_e.$$.fragment,e),st=!1},d(e){e&&(d(y),d(m),d(c),d(t),d(Ge),d(X),d(Je),d(Q),d(Re),d(Y),d(Be),d(K),d(Pe),d(He),d(te),d(Oe),d(Ze),d(S),d(Xe),d(Qe),d(v),d(Ye),d(Ke),d(F),d(et),d(tt),d($),d(ot),d(nt),d(Ae)),d(o),b(k,e),b(M,e),b(ee,e),b(oe,e),b(ne),b(N),b(se,e),b(ae),b(re),b(de),b(ie),b(J),b(le),b(ce,e),b(pe),b(me),b(R),b(B),b(he,e),b(ue),b(fe),b(P),b(H),b(_e,e)}}}const vo='{"title":"FSMT","local":"fsmt","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"Implementation Notes","local":"implementation-notes","sections":[],"depth":2},{"title":"FSMTConfig","local":"transformers.FSMTConfig","sections":[],"depth":2},{"title":"FSMTTokenizer","local":"transformers.FSMTTokenizer","sections":[],"depth":2},{"title":"FSMTModel","local":"transformers.FSMTModel","sections":[],"depth":2},{"title":"FSMTForConditionalGeneration","local":"transformers.FSMTForConditionalGeneration","sections":[],"depth":2}],"depth":1}';function wo(x){return po(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class jo extends mo{constructor(o){super(),ho(this,o,wo,yo,co,{})}}export{jo as component}; | |
Xet Storage Details
- Size:
- 75.6 kB
- Xet hash:
- c5e2698f3154fb48fc03210e8bd1bf71a5aa95a1daa6c631c6a77cf3b5aa417e
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.