Buckets:
| import{s as Ut,o as Ct,n as Le}from"../chunks/scheduler.31fdf58d.js";import{S as $t,i as jt,e as i,s,c as u,h as It,a as d,d as o,b as a,f as de,j as y,g as h,k as D,l as p,m as n,n as f,t as g,o as b,p as _}from"../chunks/index.2f76fdf0.js";import{T as Jt}from"../chunks/Tip.8d349121.js";import{C as Ft}from"../chunks/CopyLLMTxtMenu.6ffa41dc.js";import{D as Te}from"../chunks/Docstring.ac2bce02.js";import{C as we}from"../chunks/CodeBlock.e52df5d6.js";import{E as kt}from"../chunks/ExampleCodeBlock.f8b50594.js";import{H as ye,E as Nt}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.66a55e6f.js";function Rt(j){let r,T="Example:",c,m,M;return m=new we({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMERicnhDb25maWclMkMlMjBEYnJ4TW9kZWwlMEElMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwRGJyeCUyMGNvbmZpZ3VyYXRpb24lMEFjb25maWd1cmF0aW9uJTIwJTNEJTIwRGJyeENvbmZpZyhuX2xheWVycyUzRDIlMkMlMjBkX21vZGVsJTNEMjU2JTJDJTIwbl9oZWFkcyUzRDglMkMlMjB2b2NhYl9zaXplJTNEMTI4KSUwQSUwQSUyMyUyMEluaXRpYWxpemluZyUyMGElMjBtb2RlbCUyMCh3aXRoJTIwcmFuZG9tJTIwd2VpZ2h0cyklMjBmcm9tJTIwdGhlJTIwY29uZmlndXJhdGlvbiUwQW1vZGVsJTIwJTNEJTIwRGJyeE1vZGVsKGNvbmZpZ3VyYXRpb24pJTBBJTBBJTIzJTIwQWNjZXNzaW5nJTIwdGhlJTIwbW9kZWwlMjBjb25maWd1cmF0aW9uJTBBY29uZmlndXJhdGlvbiUyMCUzRCUyMG1vZGVsLmNvbmZpZw==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> DbrxConfig, DbrxModel | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a Dbrx configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = DbrxConfig(n_layers=<span class="hljs-number">2</span>, d_model=<span class="hljs-number">256</span>, n_heads=<span class="hljs-number">8</span>, vocab_size=<span class="hljs-number">128</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a model (with random weights) from the configuration</span> | |
| <span class="hljs-meta">>>> </span>model = DbrxModel(configuration) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Accessing the model configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = model.config`,lang:"python",wrap:!1}}),{c(){r=i("p"),r.textContent=T,c=s(),u(m.$$.fragment)},l(l){r=d(l,"P",{"data-svelte-h":!0}),y(r)!=="svelte-11lpom8"&&(r.textContent=T),c=a(l),h(m.$$.fragment,l)},m(l,k){n(l,r,k),n(l,c,k),f(m,l,k),M=!0},p:Le,i(l){M||(g(m.$$.fragment,l),M=!0)},o(l){b(m.$$.fragment,l),M=!1},d(l){l&&(o(r),o(c)),_(m,l)}}}function zt(j){let r,T=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){r=i("p"),r.innerHTML=T},l(c){r=d(c,"P",{"data-svelte-h":!0}),y(r)!=="svelte-fincs2"&&(r.innerHTML=T)},m(c,m){n(c,r,m)},p:Le,d(c){c&&o(r)}}}function Zt(j){let r,T=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){r=i("p"),r.innerHTML=T},l(c){r=d(c,"P",{"data-svelte-h":!0}),y(r)!=="svelte-fincs2"&&(r.innerHTML=T)},m(c,m){n(c,r,m)},p:Le,d(c){c&&o(r)}}}function Wt(j){let r,T="Example:",c,m,M;return m=new we({props:{code:"JTNFJTNFJTIwZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBEYnJ4Rm9yQ2F1c2FsTE0lMEElMEElM0UlM0UlMjBtb2RlbCUyMCUzRCUyMERicnhGb3JDYXVzYWxMTS5mcm9tX3ByZXRyYWluZWQoJTIydHJhbnNmb3JtZXJzLWNvbW11bml0eSUyRmRicngtaW5zdHJ1Y3QlMjIpJTBBJTNFJTNFJTIwdG9rZW5pemVyJTIwJTNEJTIwQXV0b1Rva2VuaXplci5mcm9tX3ByZXRyYWluZWQoJTIydHJhbnNmb3JtZXJzLWNvbW11bml0eSUyRmRicngtaW5zdHJ1Y3QlMjIpJTBBJTBBJTNFJTNFJTIwcHJvbXB0JTIwJTNEJTIwJTIySGV5JTJDJTIwYXJlJTIweW91JTIwY29uc2Npb3VzJTNGJTIwQ2FuJTIweW91JTIwdGFsayUyMHRvJTIwbWUlM0YlMjIlMEElM0UlM0UlMjBpbnB1dHMlMjAlM0QlMjB0b2tlbml6ZXIocHJvbXB0JTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiklMEElMEElM0UlM0UlMjAlMjMlMjBHZW5lcmF0ZSUwQSUzRSUzRSUyMGdlbmVyYXRlX2lkcyUyMCUzRCUyMG1vZGVsLmdlbmVyYXRlKGlucHV0cy5pbnB1dF9pZHMlMkMlMjBtYXhfbGVuZ3RoJTNEMzApJTBBJTNFJTNFJTIwdG9rZW5pemVyLmJhdGNoX2RlY29kZShnZW5lcmF0ZV9pZHMlMkMlMjBza2lwX3NwZWNpYWxfdG9rZW5zJTNEVHJ1ZSUyQyUyMGNsZWFuX3VwX3Rva2VuaXphdGlvbl9zcGFjZXMlM0RGYWxzZSklNUIwJTVEJTBBJTIySGV5JTJDJTIwYXJlJTIweW91JTIwY29uc2Npb3VzJTNGJTIwQ2FuJTIweW91JTIwdGFsayUyMHRvJTIwbWUlM0YlNUNuSSdtJTIwbm90JTIwY29uc2Npb3VzJTJDJTIwYnV0JTIwSSUyMGNhbiUyMHRhbGslMjB0byUyMHlvdS4lMjI=",highlighted:`>> <span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, DbrxForCausalLM | |
| >> model = DbrxForCausalLM.from_pretrained(<span class="hljs-string">"transformers-community/dbrx-instruct"</span>) | |
| >> tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"transformers-community/dbrx-instruct"</span>) | |
| >> prompt = <span class="hljs-string">"Hey, are you conscious? Can you talk to me?"</span> | |
| >> inputs = tokenizer(prompt, return_tensors=<span class="hljs-string">"pt"</span>) | |
| >> <span class="hljs-comment"># Generate</span> | |
| >> generate_ids = model.generate(inputs.input_ids, max_length=<span class="hljs-number">30</span>) | |
| >> tokenizer.batch_decode(generate_ids, skip_special_tokens=<span class="hljs-literal">True</span>, clean_up_tokenization_spaces=<span class="hljs-literal">False</span>)[<span class="hljs-number">0</span>] | |
| <span class="hljs-string">"Hey, are you conscious? Can you talk to me?\\nI'm not conscious, but I can talk to you."</span>`,lang:"python",wrap:!1}}),{c(){r=i("p"),r.textContent=T,c=s(),u(m.$$.fragment)},l(l){r=d(l,"P",{"data-svelte-h":!0}),y(r)!=="svelte-11lpom8"&&(r.textContent=T),c=a(l),h(m.$$.fragment,l)},m(l,k){n(l,r,k),n(l,c,k),f(m,l,k),M=!0},p:Le,i(l){M||(g(m.$$.fragment,l),M=!0)},o(l){b(m.$$.fragment,l),M=!1},d(l){l&&(o(r),o(c)),_(m,l)}}}function Dt(j){let r,T,c,m,M,l="<em>This model was contributed to Hugging Face Transformers on 2024-04-18.</em>",k,E,xe,B,ve,N,it='<img alt="FlashAttention" src="https://img.shields.io/badge/%E2%9A%A1%EF%B8%8E%20FlashAttention-eae0c8?style=flat"/> <img alt="SDPA" src="https://img.shields.io/badge/SDPA-DE3412?style=flat&logo=pytorch&logoColor=white"/>',Je,X,ke,G,dt=`DBRX is a <a href="https://www.isattentionallyouneed.com/" rel="nofollow">transformer-based</a> decoder-only large language model (LLM) that was trained using next-token prediction. | |
| It uses a <em>fine-grained</em> mixture-of-experts (MoE) architecture with 132B total parameters of which 36B parameters are active on any input. | |
| It was pre-trained on 12T tokens of text and code data. | |
| Compared to other open MoE models like Mixtral-8x7B and Grok-1, DBRX is fine-grained, meaning it uses a larger number of smaller experts. DBRX has 16 experts and chooses 4, while Mixtral-8x7B and Grok-1 have 8 experts and choose 2. | |
| This provides 65x more possible combinations of experts and we found that this improves model quality. | |
| DBRX uses rotary position encodings (RoPE), gated linear units (GLU), and grouped query attention (GQA). | |
| It is a BPE based model and uses the GPT-4 tokenizer as described in the <a href="https://github.com/openai/tiktoken" rel="nofollow">tiktoken</a> repository. | |
| We made these choices based on exhaustive evaluation and scaling experiments.`,Ue,q,ct=`DBRX was pretrained on 12T tokens of carefully curated data and a maximum context length of 32K tokens. | |
| We estimate that this data is at least 2x better token-for-token than the data we used to pretrain the MPT family of models. | |
| This new dataset was developed using the full suite of Databricks tools, including Apache Spark™ and Databricks notebooks for data processing, and Unity Catalog for data management and governance. | |
| We used curriculum learning for pretraining, changing the data mix during training in ways we found to substantially improve model quality.`,Ce,V,pt='More detailed information about DBRX Instruct and DBRX Base can be found in our <a href="https://www.databricks.com/blog/introducing-dbrx-new-state-art-open-llm" rel="nofollow">technical blog post</a>.',$e,H,mt=`This model was contributed by <a href="https://huggingface.co/eitanturok" rel="nofollow">eitan-turok</a> and <a href="https://huggingface.co/abhi-db" rel="nofollow">abhi-db</a>. | |
| Note: The original <code>databricks/dbrx-instruct</code> checkpoint was closed; <a href="https://huggingface.co/transformers-community/dbrx-instruct" rel="nofollow"><code>transformers-community/dbrx-instruct</code></a> is a re-upload for compatibility, and the snippets below use that re-upload.`,je,L,Ie,Q,ut="The <code>generate()</code> method can be used to generate text using DBRX. You can generate using the standard attention implementation, flash-attention, and the PyTorch scaled dot product attention. The last two attention implementations give speed ups.",Fe,Y,Ne,S,ht='If you have flash-attention installed (<code>pip install flash-attn</code>), it is possible to generate faster. (The HuggingFace documentation for flash-attention can be found <a href="https://huggingface.co/docs/transformers/perf_infer_gpu_one#flashattention-2" rel="nofollow">here</a>.)',Re,P,ze,A,ft='You can also generate faster using the PyTorch scaled dot product attention. (The HuggingFace documentation for scaled dot product attention can be found <a href="https://huggingface.co/docs/transformers/perf_infer_gpu_one#pytorch-scaled-dot-product-attention" rel="nofollow">here</a>.)',Ze,O,We,K,De,v,ee,Qe,ce,gt=`This is the configuration class to store the configuration of a DbrxModel. It is used to instantiate a Dbrx | |
| model according to the specified arguments, defining the model architecture. Instantiating a configuration with the | |
| defaults will yield a similar configuration to that of the <a href="https://huggingface.co/transformers-community/dbrx-instruct" rel="nofollow">transformers-community/dbrx-instruct</a>`,Ye,pe,bt=`Configuration objects inherit from <a href="/docs/transformers/pr_41116/en/main_classes/configuration#transformers.PreTrainedConfig">PreTrainedConfig</a> and can be used to control the model outputs. Read the | |
| documentation from <a href="/docs/transformers/pr_41116/en/main_classes/configuration#transformers.PreTrainedConfig">PreTrainedConfig</a> for more information.`,Se,R,Ee,te,Be,w,oe,Pe,me,_t="The bare Dbrx Model outputting raw hidden-states without any specific head on top.",Ae,ue,yt=`This model inherits from <a href="/docs/transformers/pr_41116/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,Oe,he,Mt=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,Ke,U,ne,et,fe,Tt='The <a href="/docs/transformers/pr_41116/en/model_doc/dbrx#transformers.DbrxModel">DbrxModel</a> forward method, overrides the <code>__call__</code> special method.',tt,z,ot,ge,wt=`<li><p><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — Sequence of hidden-states at the output of the last layer of the model.</p></li> <li><p><strong>past_key_values</strong> (<code>Cache</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — It is a <a href="/docs/transformers/pr_41116/en/internal/generation_utils#transformers.Cache">Cache</a> instance. For more details, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>.</p> <p>Contains pre-computed hidden-states (key and values in the self-attention blocks and optionally if | |
| <code>config.is_encoder_decoder=True</code> in the cross-attention blocks) that can be used (see <code>past_key_values</code> | |
| input) to speed up sequential decoding.</p></li> <li><p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p></li> <li><p><strong>router_logits</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_router_probs=True</code> and <code>config.add_router_probs=True</code> is passed or when <code>config.output_router_probs=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, sequence_length, num_experts)</code>.</p> <p>Raw router logtis (post-softmax) that are computed by MoE routers, these terms are used to compute the auxiliary | |
| loss for Mixture of Experts models.</p></li>`,Xe,se,Ge,F,ae,nt,x,re,st,be,xt='The <a href="/docs/transformers/pr_41116/en/model_doc/dbrx#transformers.DbrxForCausalLM">DbrxForCausalLM</a> forward method, overrides the <code>__call__</code> special method.',at,Z,rt,_e,vt=`<li><p><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> is provided) — Language modeling loss (for next-token prediction).</p></li> <li><p><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, config.vocab_size)</code>) — Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).</p></li> <li><p><strong>aux_loss</strong> (<code>torch.FloatTensor</code>, <em>optional</em>, returned when <code>labels</code> is provided) — aux_loss for the sparse modules.</p></li> <li><p><strong>router_logits</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_router_probs=True</code> and <code>config.add_router_probs=True</code> is passed or when <code>config.output_router_probs=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, sequence_length, num_experts)</code>.</p> <p>Raw router logtis (post-softmax) that are computed by MoE routers, these terms are used to compute the auxiliary | |
| loss for Mixture of Experts models.</p></li> <li><p><strong>past_key_values</strong> (<code>Cache</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — It is a <a href="/docs/transformers/pr_41116/en/internal/generation_utils#transformers.Cache">Cache</a> instance. For more details, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>.</p> <p>Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see | |
| <code>past_key_values</code> input) to speed up sequential decoding.</p></li> <li><p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p></li>`,lt,W,qe,le,Ve,Me,He;return E=new Ft({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),B=new ye({props:{title:"DBRX",local:"dbrx",headingTag:"h1"}}),X=new ye({props:{title:"Overview",local:"overview",headingTag:"h2"}}),L=new ye({props:{title:"Usage Examples",local:"usage-examples",headingTag:"h2"}}),Y=new we({props:{code:"JTBBZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBEYnJ4Rm9yQ2F1c2FsTE0lMEElMEElMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZCglMjJ0cmFuc2Zvcm1lcnMtY29tbXVuaXR5JTJGZGJyeC1pbnN0cnVjdCUyMiUyQyUyMHRva2VuJTNEJTIyWU9VUl9IRl9UT0tFTiUyMiklMEFtb2RlbCUyMCUzRCUyMERicnhGb3JDYXVzYWxMTS5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwJTIydHJhbnNmb3JtZXJzLWNvbW11bml0eSUyRmRicngtaW5zdHJ1Y3QlMjIlMkMlMEElMjAlMjAlMjAlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMiUyQyUwQSUyMCUyMCUyMCUyMHRva2VuJTNEJTIyWU9VUl9IRl9UT0tFTiUyMiUyQyUwQSUyMCUyMCUyMCUyMCklMEElMEFpbnB1dF90ZXh0JTIwJTNEJTIwJTIyV2hhdCUyMGRvZXMlMjBpdCUyMHRha2UlMjB0byUyMGJ1aWxkJTIwYSUyMGdyZWF0JTIwTExNJTNGJTIyJTBBbWVzc2FnZXMlMjAlM0QlMjAlNUIlN0IlMjJyb2xlJTIyJTNBJTIwJTIydXNlciUyMiUyQyUyMCUyMmNvbnRlbnQlMjIlM0ElMjBpbnB1dF90ZXh0JTdEJTVEJTBBaW5wdXRfaWRzJTIwJTNEJTIwdG9rZW5pemVyLmFwcGx5X2NoYXRfdGVtcGxhdGUobWVzc2FnZXMlMkMlMjByZXR1cm5fZGljdCUzRFRydWUlMkMlMjB0b2tlbml6ZSUzRFRydWUlMkMlMjBhZGRfZ2VuZXJhdGlvbl9wcm9tcHQlM0RUcnVlJTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMikudG8obW9kZWwuZGV2aWNlKSUwQSUwQW91dHB1dHMlMjAlM0QlMjBtb2RlbC5nZW5lcmF0ZSgqKmlucHV0X2lkcyUyQyUyMG1heF9uZXdfdG9rZW5zJTNEMjAwKSUwQXByaW50KHRva2VuaXplci5kZWNvZGUob3V0cHV0cyU1QjAlNUQpKQ==",highlighted:` | |
| <span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, DbrxForCausalLM | |
| tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"transformers-community/dbrx-instruct"</span>, token=<span class="hljs-string">"YOUR_HF_TOKEN"</span>) | |
| model = DbrxForCausalLM.from_pretrained( | |
| <span class="hljs-string">"transformers-community/dbrx-instruct"</span>, | |
| device_map=<span class="hljs-string">"auto"</span>, | |
| token=<span class="hljs-string">"YOUR_HF_TOKEN"</span>, | |
| ) | |
| input_text = <span class="hljs-string">"What does it take to build a great LLM?"</span> | |
| messages = [{<span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, <span class="hljs-string">"content"</span>: input_text}] | |
| input_ids = tokenizer.apply_chat_template(messages, return_dict=<span class="hljs-literal">True</span>, tokenize=<span class="hljs-literal">True</span>, add_generation_prompt=<span class="hljs-literal">True</span>, return_tensors=<span class="hljs-string">"pt"</span>).to(model.device) | |
| outputs = model.generate(**input_ids, max_new_tokens=<span class="hljs-number">200</span>) | |
| <span class="hljs-built_in">print</span>(tokenizer.decode(outputs[<span class="hljs-number">0</span>]))`,lang:"python",wrap:!1}}),P=new we({props:{code:"JTBBZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBEYnJ4Rm9yQ2F1c2FsTE0lMEElMEElMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZCglMjJ0cmFuc2Zvcm1lcnMtY29tbXVuaXR5JTJGZGJyeC1pbnN0cnVjdCUyMiUyQyUyMHRva2VuJTNEJTIyWU9VUl9IRl9UT0tFTiUyMiklMEFtb2RlbCUyMCUzRCUyMERicnhGb3JDYXVzYWxMTS5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwJTIydHJhbnNmb3JtZXJzLWNvbW11bml0eSUyRmRicngtaW5zdHJ1Y3QlMjIlMkMlMEElMjAlMjAlMjAlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMiUyQyUwQSUyMCUyMCUyMCUyMHRva2VuJTNEJTIyWU9VUl9IRl9UT0tFTiUyMiUyQyUwQSUyMCUyMCUyMCUyMGF0dG5faW1wbGVtZW50YXRpb24lM0QlMjJmbGFzaF9hdHRlbnRpb25fMiUyMiUyQyUwQSUyMCUyMCUyMCUyMCklMEElMEFpbnB1dF90ZXh0JTIwJTNEJTIwJTIyV2hhdCUyMGRvZXMlMjBpdCUyMHRha2UlMjB0byUyMGJ1aWxkJTIwYSUyMGdyZWF0JTIwTExNJTNGJTIyJTBBbWVzc2FnZXMlMjAlM0QlMjAlNUIlN0IlMjJyb2xlJTIyJTNBJTIwJTIydXNlciUyMiUyQyUyMCUyMmNvbnRlbnQlMjIlM0ElMjBpbnB1dF90ZXh0JTdEJTVEJTBBaW5wdXRfaWRzJTIwJTNEJTIwdG9rZW5pemVyLmFwcGx5X2NoYXRfdGVtcGxhdGUobWVzc2FnZXMlMkMlMjByZXR1cm5fZGljdCUzRFRydWUlMkMlMjB0b2tlbml6ZSUzRFRydWUlMkMlMjBhZGRfZ2VuZXJhdGlvbl9wcm9tcHQlM0RUcnVlJTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMikudG8obW9kZWwuZGV2aWNlKSUwQSUwQW91dHB1dHMlMjAlM0QlMjBtb2RlbC5nZW5lcmF0ZSgqKmlucHV0X2lkcyUyQyUyMG1heF9uZXdfdG9rZW5zJTNEMjAwKSUwQXByaW50KHRva2VuaXplci5kZWNvZGUob3V0cHV0cyU1QjAlNUQpKQ==",highlighted:` | |
| <span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, DbrxForCausalLM | |
| tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"transformers-community/dbrx-instruct"</span>, token=<span class="hljs-string">"YOUR_HF_TOKEN"</span>) | |
| model = DbrxForCausalLM.from_pretrained( | |
| <span class="hljs-string">"transformers-community/dbrx-instruct"</span>, | |
| device_map=<span class="hljs-string">"auto"</span>, | |
| token=<span class="hljs-string">"YOUR_HF_TOKEN"</span>, | |
| attn_implementation=<span class="hljs-string">"flash_attention_2"</span>, | |
| ) | |
| input_text = <span class="hljs-string">"What does it take to build a great LLM?"</span> | |
| messages = [{<span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, <span class="hljs-string">"content"</span>: input_text}] | |
| input_ids = tokenizer.apply_chat_template(messages, return_dict=<span class="hljs-literal">True</span>, tokenize=<span class="hljs-literal">True</span>, add_generation_prompt=<span class="hljs-literal">True</span>, return_tensors=<span class="hljs-string">"pt"</span>).to(model.device) | |
| outputs = model.generate(**input_ids, max_new_tokens=<span class="hljs-number">200</span>) | |
| <span class="hljs-built_in">print</span>(tokenizer.decode(outputs[<span class="hljs-number">0</span>]))`,lang:"python",wrap:!1}}),O=new we({props:{code:"JTBBZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBEYnJ4Rm9yQ2F1c2FsTE0lMEElMEElMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZCglMjJ0cmFuc2Zvcm1lcnMtY29tbXVuaXR5JTJGZGJyeC1pbnN0cnVjdCUyMiUyQyUyMHRva2VuJTNEJTIyWU9VUl9IRl9UT0tFTiUyMiklMEFtb2RlbCUyMCUzRCUyMERicnhGb3JDYXVzYWxMTS5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwJTIydHJhbnNmb3JtZXJzLWNvbW11bml0eSUyRmRicngtaW5zdHJ1Y3QlMjIlMkMlMEElMjAlMjAlMjAlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMiUyQyUwQSUyMCUyMCUyMCUyMHRva2VuJTNEJTIyWU9VUl9IRl9UT0tFTiUyMiUyQyUwQSUyMCUyMCUyMCUyMGF0dG5faW1wbGVtZW50YXRpb24lM0QlMjJzZHBhJTIyJTJDJTBBJTIwJTIwJTIwJTIwKSUwQSUwQWlucHV0X3RleHQlMjAlM0QlMjAlMjJXaGF0JTIwZG9lcyUyMGl0JTIwdGFrZSUyMHRvJTIwYnVpbGQlMjBhJTIwZ3JlYXQlMjBMTE0lM0YlMjIlMEFtZXNzYWdlcyUyMCUzRCUyMCU1QiU3QiUyMnJvbGUlMjIlM0ElMjAlMjJ1c2VyJTIyJTJDJTIwJTIyY29udGVudCUyMiUzQSUyMGlucHV0X3RleHQlN0QlNUQlMEFpbnB1dF9pZHMlMjAlM0QlMjB0b2tlbml6ZXIuYXBwbHlfY2hhdF90ZW1wbGF0ZShtZXNzYWdlcyUyQyUyMHJldHVybl9kaWN0JTNEVHJ1ZSUyQyUyMHRva2VuaXplJTNEVHJ1ZSUyQyUyMGFkZF9nZW5lcmF0aW9uX3Byb21wdCUzRFRydWUlMkMlMjByZXR1cm5fdGVuc29ycyUzRCUyMnB0JTIyKS50byhtb2RlbC5kZXZpY2UpJTBBJTBBb3V0cHV0cyUyMCUzRCUyMG1vZGVsLmdlbmVyYXRlKCoqaW5wdXRfaWRzJTJDJTIwbWF4X25ld190b2tlbnMlM0QyMDApJTBBcHJpbnQodG9rZW5pemVyLmRlY29kZShvdXRwdXRzJTVCMCU1RCkp",highlighted:` | |
| <span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, DbrxForCausalLM | |
| tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"transformers-community/dbrx-instruct"</span>, token=<span class="hljs-string">"YOUR_HF_TOKEN"</span>) | |
| model = DbrxForCausalLM.from_pretrained( | |
| <span class="hljs-string">"transformers-community/dbrx-instruct"</span>, | |
| device_map=<span class="hljs-string">"auto"</span>, | |
| token=<span class="hljs-string">"YOUR_HF_TOKEN"</span>, | |
| attn_implementation=<span class="hljs-string">"sdpa"</span>, | |
| ) | |
| input_text = <span class="hljs-string">"What does it take to build a great LLM?"</span> | |
| messages = [{<span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, <span class="hljs-string">"content"</span>: input_text}] | |
| input_ids = tokenizer.apply_chat_template(messages, return_dict=<span class="hljs-literal">True</span>, tokenize=<span class="hljs-literal">True</span>, add_generation_prompt=<span class="hljs-literal">True</span>, return_tensors=<span class="hljs-string">"pt"</span>).to(model.device) | |
| outputs = model.generate(**input_ids, max_new_tokens=<span class="hljs-number">200</span>) | |
| <span class="hljs-built_in">print</span>(tokenizer.decode(outputs[<span class="hljs-number">0</span>]))`,lang:"python",wrap:!1}}),K=new ye({props:{title:"DbrxConfig",local:"transformers.DbrxConfig",headingTag:"h2"}}),ee=new Te({props:{name:"class transformers.DbrxConfig",anchor:"transformers.DbrxConfig",parameters:[{name:"transformers_version",val:": str | None = None"},{name:"architectures",val:": list[str] | None = None"},{name:"output_hidden_states",val:": bool | None = False"},{name:"return_dict",val:": bool | None = True"},{name:"dtype",val:": typing.Union[str, ForwardRef('torch.dtype'), NoneType] = None"},{name:"chunk_size_feed_forward",val:": int = 0"},{name:"is_encoder_decoder",val:": bool = False"},{name:"id2label",val:": dict[int, str] | dict[str, str] | None = None"},{name:"label2id",val:": dict[str, int] | dict[str, str] | None = None"},{name:"problem_type",val:": typing.Optional[typing.Literal['regression', 'single_label_classification', 'multi_label_classification']] = None"},{name:"d_model",val:": int | None = 2048"},{name:"n_heads",val:": int | None = 16"},{name:"n_layers",val:": int | None = 24"},{name:"max_seq_len",val:": int | None = 2048"},{name:"vocab_size",val:": int = 32000"},{name:"resid_pdrop",val:": float | None = 0.0"},{name:"emb_pdrop",val:": float | None = 0.0"},{name:"attn_config",val:": transformers.models.dbrx.configuration_dbrx.DbrxAttentionConfig | dict | None = None"},{name:"ffn_config",val:": transformers.models.dbrx.configuration_dbrx.DbrxFFNConfig | dict | None = None"},{name:"use_cache",val:": bool = True"},{name:"initializer_range",val:": float = 0.02"},{name:"output_router_logits",val:": bool | None = False"},{name:"rope_parameters",val:": transformers.modeling_rope_utils.RopeParameters | dict | None = None"},{name:"pad_token_id",val:": int | None = None"},{name:"bos_token_id",val:": int | None = None"},{name:"eos_token_id",val:": int | list[int] | None = None"},{name:"tie_word_embeddings",val:": bool = False"}],parametersDescription:[{anchor:"transformers.DbrxConfig.d_model",description:`<strong>d_model</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2048</code>) — | |
| Size of the encoder layers and the pooler layer.`,name:"d_model"},{anchor:"transformers.DbrxConfig.n_heads",description:`<strong>n_heads</strong> (<code>int</code>, <em>optional</em>, defaults to <code>16</code>) — | |
| Number of attention heads for each attention layer in the Transformer decoder.`,name:"n_heads"},{anchor:"transformers.DbrxConfig.n_layers",description:`<strong>n_layers</strong> (<code>int</code>, <em>optional</em>, defaults to <code>24</code>) — | |
| Number of hidden layers in the Transformer decoder.`,name:"n_layers"},{anchor:"transformers.DbrxConfig.max_seq_len",description:`<strong>max_seq_len</strong> (<code>int</code>, <em>optional</em>, defaults to 2048) — | |
| The maximum sequence length of the model.`,name:"max_seq_len"},{anchor:"transformers.DbrxConfig.vocab_size",description:`<strong>vocab_size</strong> (<code>int</code>, <em>optional</em>, defaults to <code>32000</code>) — | |
| Vocabulary size of the model. Defines the number of different tokens that can be represented by the <code>input_ids</code>.`,name:"vocab_size"},{anchor:"transformers.DbrxConfig.resid_pdrop",description:`<strong>resid_pdrop</strong> (<code>float</code>, <em>optional</em>, defaults to <code>0.0</code>) — | |
| The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.`,name:"resid_pdrop"},{anchor:"transformers.DbrxConfig.emb_pdrop",description:`<strong>emb_pdrop</strong> (<code>float</code>, <em>optional</em>, defaults to <code>0.0</code>) — | |
| The dropout ratio for the embeddings.`,name:"emb_pdrop"},{anchor:"transformers.DbrxConfig.attn_config",description:`<strong>attn_config</strong> (<code>dict</code>, <em>optional</em>) — | |
| A dictionary used to configure the model’s attention module.`,name:"attn_config"},{anchor:"transformers.DbrxConfig.ffn_config",description:`<strong>ffn_config</strong> (<code>dict</code>, <em>optional</em>) — | |
| A dictionary used to configure the model’s FFN module.`,name:"ffn_config"},{anchor:"transformers.DbrxConfig.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not the model should return the last key/values attentions (not used by all models). Only | |
| relevant if <code>config.is_decoder=True</code> or when the model is a decoder-only generative model.`,name:"use_cache"},{anchor:"transformers.DbrxConfig.initializer_range",description:`<strong>initializer_range</strong> (<code>float</code>, <em>optional</em>, defaults to <code>0.02</code>) — | |
| The standard deviation of the truncated_normal_initializer for initializing all weight matrices.`,name:"initializer_range"},{anchor:"transformers.DbrxConfig.output_router_logits",description:`<strong>output_router_logits</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not the router logits should be returned by the model. Enabling this will also allow the model | |
| to output the auxiliary loss, including load balancing loss and router z-loss.`,name:"output_router_logits"},{anchor:"transformers.DbrxConfig.rope_parameters",description:`<strong>rope_parameters</strong> (<code>Union[~modeling_rope_utils.RopeParameters, dict]</code>, <em>optional</em>) — | |
| Dictionary containing the configuration parameters for the RoPE embeddings. The dictionary should contain | |
| a value for <code>rope_theta</code> and optionally parameters used for scaling in case you want to use RoPE | |
| with longer <code>max_position_embeddings</code>.`,name:"rope_parameters"},{anchor:"transformers.DbrxConfig.pad_token_id",description:`<strong>pad_token_id</strong> (<code>int</code>, <em>optional</em>) — | |
| Token id used for padding in the vocabulary.`,name:"pad_token_id"},{anchor:"transformers.DbrxConfig.bos_token_id",description:`<strong>bos_token_id</strong> (<code>int</code>, <em>optional</em>) — | |
| Token id used for beginning-of-stream in the vocabulary.`,name:"bos_token_id"},{anchor:"transformers.DbrxConfig.eos_token_id",description:`<strong>eos_token_id</strong> (<code>Union[int, list[int]]</code>, <em>optional</em>) — | |
| Token id used for end-of-stream in the vocabulary.`,name:"eos_token_id"},{anchor:"transformers.DbrxConfig.tie_word_embeddings",description:`<strong>tie_word_embeddings</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to tie weight embeddings according to model’s <code>tied_weights_keys</code> mapping.`,name:"tie_word_embeddings"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dbrx/configuration_dbrx.py#L104"}}),R=new kt({props:{anchor:"transformers.DbrxConfig.example",$$slots:{default:[Rt]},$$scope:{ctx:j}}}),te=new ye({props:{title:"DbrxModel",local:"transformers.DbrxModel",headingTag:"h2"}}),oe=new Te({props:{name:"class transformers.DbrxModel",anchor:"transformers.DbrxModel",parameters:[{name:"config",val:": DbrxConfig"}],parametersDescription:[{anchor:"transformers.DbrxModel.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_41116/en/model_doc/dbrx#transformers.DbrxConfig">DbrxConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_41116/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dbrx/modeling_dbrx.py#L473"}}),ne=new Te({props:{name:"forward",anchor:"transformers.DbrxModel.forward",parameters:[{name:"input_ids",val:": torch.LongTensor | None = None"},{name:"attention_mask",val:": torch.Tensor | None = None"},{name:"position_ids",val:": torch.LongTensor | None = None"},{name:"past_key_values",val:": transformers.cache_utils.Cache | None = None"},{name:"inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"use_cache",val:": bool | None = None"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.utils.generic.TransformersKwargs]"}],parametersDescription:[{anchor:"transformers.DbrxModel.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_41116/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_41116/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_41116/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.DbrxModel.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"attention_mask"},{anchor:"transformers.DbrxModel.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.DbrxModel.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>~cache_utils.Cache</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Only <a href="/docs/transformers/pr_41116/en/internal/generation_utils#transformers.Cache">Cache</a> instance is allowed as input, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>. | |
| If no <code>past_key_values</code> are passed, <a href="/docs/transformers/pr_41116/en/internal/generation_utils#transformers.DynamicCache">DynamicCache</a> will be initialized by default.</p> | |
| <p>The model will output the same cache format that is fed as input.</p> | |
| <p>If <code>past_key_values</code> are used, the user is expected to input only unprocessed <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, unprocessed_length)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.DbrxModel.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.DbrxModel.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dbrx/modeling_dbrx.py#L502",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>MoeModelOutputWithPast</code> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_41116/en/model_doc/dbrx#transformers.DbrxConfig" | |
| >DbrxConfig</a>) and inputs.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>MoeModelOutputWithPast</code> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),z=new Jt({props:{$$slots:{default:[zt]},$$scope:{ctx:j}}}),se=new ye({props:{title:"DbrxForCausalLM",local:"transformers.DbrxForCausalLM",headingTag:"h2"}}),ae=new Te({props:{name:"class transformers.DbrxForCausalLM",anchor:"transformers.DbrxForCausalLM",parameters:[{name:"config",val:": DbrxConfig"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dbrx/modeling_dbrx.py#L643"}}),re=new Te({props:{name:"forward",anchor:"transformers.DbrxForCausalLM.forward",parameters:[{name:"input_ids",val:": torch.LongTensor | None = None"},{name:"attention_mask",val:": torch.Tensor | None = None"},{name:"position_ids",val:": torch.LongTensor | None = None"},{name:"past_key_values",val:": transformers.cache_utils.Cache | None = None"},{name:"inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"labels",val:": torch.LongTensor | None = None"},{name:"use_cache",val:": bool | None = None"},{name:"output_router_logits",val:": bool | None = None"},{name:"logits_to_keep",val:": int | torch.Tensor = 0"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.utils.generic.TransformersKwargs]"}],parametersDescription:[{anchor:"transformers.DbrxForCausalLM.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_41116/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_41116/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_41116/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.DbrxForCausalLM.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"attention_mask"},{anchor:"transformers.DbrxForCausalLM.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.DbrxForCausalLM.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>~cache_utils.Cache</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Only <a href="/docs/transformers/pr_41116/en/internal/generation_utils#transformers.Cache">Cache</a> instance is allowed as input, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>. | |
| If no <code>past_key_values</code> are passed, <a href="/docs/transformers/pr_41116/en/internal/generation_utils#transformers.DynamicCache">DynamicCache</a> will be initialized by default.</p> | |
| <p>The model will output the same cache format that is fed as input.</p> | |
| <p>If <code>past_key_values</code> are used, the user is expected to input only unprocessed <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, unprocessed_length)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.DbrxForCausalLM.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.DbrxForCausalLM.forward.labels",description:`<strong>labels</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Labels for computing the masked language modeling loss. Indices should either be in <code>[0, ..., config.vocab_size]</code> or -100 (see <code>input_ids</code> docstring). Tokens with indices set to <code>-100</code> are ignored | |
| (masked), the loss is only computed for the tokens with labels in <code>[0, ..., config.vocab_size]</code>.`,name:"labels"},{anchor:"transformers.DbrxForCausalLM.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.DbrxForCausalLM.forward.output_router_logits",description:`<strong>output_router_logits</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the logits of all the routers. They are useful for computing the router loss, and | |
| should not be returned during inference.`,name:"output_router_logits"},{anchor:"transformers.DbrxForCausalLM.forward.logits_to_keep",description:`<strong>logits_to_keep</strong> (<code>Union[int, torch.Tensor]</code>, <em>optional</em>, defaults to <code>0</code>) — | |
| If an <code>int</code>, compute logits for the last <code>logits_to_keep</code> tokens. If <code>0</code>, calculate logits for all | |
| <code>input_ids</code> (special case). Only last token logits are needed for generation, and calculating them only for that | |
| token can save memory, which becomes pretty significant for long sequences or large vocabulary size. | |
| If a <code>torch.Tensor</code>, must be 1D corresponding to the indices to keep in the sequence length dimension. | |
| This is useful when using packed tensor format (single dimension for batch and sequence length).`,name:"logits_to_keep"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dbrx/modeling_dbrx.py#L676",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>MoeCausalLMOutputWithPast</code> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_41116/en/model_doc/dbrx#transformers.DbrxConfig" | |
| >DbrxConfig</a>) and inputs.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>MoeCausalLMOutputWithPast</code> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),Z=new Jt({props:{$$slots:{default:[Zt]},$$scope:{ctx:j}}}),W=new kt({props:{anchor:"transformers.DbrxForCausalLM.forward.example",$$slots:{default:[Wt]},$$scope:{ctx:j}}}),le=new Nt({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/dbrx.md"}}),{c(){r=i("meta"),T=s(),c=i("p"),m=s(),M=i("p"),M.innerHTML=l,k=s(),u(E.$$.fragment),xe=s(),u(B.$$.fragment),ve=s(),N=i("div"),N.innerHTML=it,Je=s(),u(X.$$.fragment),ke=s(),G=i("p"),G.innerHTML=dt,Ue=s(),q=i("p"),q.textContent=ct,Ce=s(),V=i("p"),V.innerHTML=pt,$e=s(),H=i("p"),H.innerHTML=mt,je=s(),u(L.$$.fragment),Ie=s(),Q=i("p"),Q.innerHTML=ut,Fe=s(),u(Y.$$.fragment),Ne=s(),S=i("p"),S.innerHTML=ht,Re=s(),u(P.$$.fragment),ze=s(),A=i("p"),A.innerHTML=ft,Ze=s(),u(O.$$.fragment),We=s(),u(K.$$.fragment),De=s(),v=i("div"),u(ee.$$.fragment),Qe=s(),ce=i("p"),ce.innerHTML=gt,Ye=s(),pe=i("p"),pe.innerHTML=bt,Se=s(),u(R.$$.fragment),Ee=s(),u(te.$$.fragment),Be=s(),w=i("div"),u(oe.$$.fragment),Pe=s(),me=i("p"),me.textContent=_t,Ae=s(),ue=i("p"),ue.innerHTML=yt,Oe=s(),he=i("p"),he.innerHTML=Mt,Ke=s(),U=i("div"),u(ne.$$.fragment),et=s(),fe=i("p"),fe.innerHTML=Tt,tt=s(),u(z.$$.fragment),ot=s(),ge=i("ul"),ge.innerHTML=wt,Xe=s(),u(se.$$.fragment),Ge=s(),F=i("div"),u(ae.$$.fragment),nt=s(),x=i("div"),u(re.$$.fragment),st=s(),be=i("p"),be.innerHTML=xt,at=s(),u(Z.$$.fragment),rt=s(),_e=i("ul"),_e.innerHTML=vt,lt=s(),u(W.$$.fragment),qe=s(),u(le.$$.fragment),Ve=s(),Me=i("p"),this.h()},l(e){const t=It("svelte-u9bgzb",document.head);r=d(t,"META",{name:!0,content:!0}),t.forEach(o),T=a(e),c=d(e,"P",{}),de(c).forEach(o),m=a(e),M=d(e,"P",{"data-svelte-h":!0}),y(M)!=="svelte-1m4sckv"&&(M.innerHTML=l),k=a(e),h(E.$$.fragment,e),xe=a(e),h(B.$$.fragment,e),ve=a(e),N=d(e,"DIV",{class:!0,"data-svelte-h":!0}),y(N)!=="svelte-mn8zh"&&(N.innerHTML=it),Je=a(e),h(X.$$.fragment,e),ke=a(e),G=d(e,"P",{"data-svelte-h":!0}),y(G)!=="svelte-14fqb4b"&&(G.innerHTML=dt),Ue=a(e),q=d(e,"P",{"data-svelte-h":!0}),y(q)!=="svelte-6tnpsh"&&(q.textContent=ct),Ce=a(e),V=d(e,"P",{"data-svelte-h":!0}),y(V)!=="svelte-p78060"&&(V.innerHTML=pt),$e=a(e),H=d(e,"P",{"data-svelte-h":!0}),y(H)!=="svelte-1ow2ny4"&&(H.innerHTML=mt),je=a(e),h(L.$$.fragment,e),Ie=a(e),Q=d(e,"P",{"data-svelte-h":!0}),y(Q)!=="svelte-1n2ymew"&&(Q.innerHTML=ut),Fe=a(e),h(Y.$$.fragment,e),Ne=a(e),S=d(e,"P",{"data-svelte-h":!0}),y(S)!=="svelte-45m1hv"&&(S.innerHTML=ht),Re=a(e),h(P.$$.fragment,e),ze=a(e),A=d(e,"P",{"data-svelte-h":!0}),y(A)!=="svelte-1cyfh2c"&&(A.innerHTML=ft),Ze=a(e),h(O.$$.fragment,e),We=a(e),h(K.$$.fragment,e),De=a(e),v=d(e,"DIV",{class:!0});var C=de(v);h(ee.$$.fragment,C),Qe=a(C),ce=d(C,"P",{"data-svelte-h":!0}),y(ce)!=="svelte-d4dk3m"&&(ce.innerHTML=gt),Ye=a(C),pe=d(C,"P",{"data-svelte-h":!0}),y(pe)!=="svelte-1plkghn"&&(pe.innerHTML=bt),Se=a(C),h(R.$$.fragment,C),C.forEach(o),Ee=a(e),h(te.$$.fragment,e),Be=a(e),w=d(e,"DIV",{class:!0});var J=de(w);h(oe.$$.fragment,J),Pe=a(J),me=d(J,"P",{"data-svelte-h":!0}),y(me)!=="svelte-1msk8jg"&&(me.textContent=_t),Ae=a(J),ue=d(J,"P",{"data-svelte-h":!0}),y(ue)!=="svelte-g65tf5"&&(ue.innerHTML=yt),Oe=a(J),he=d(J,"P",{"data-svelte-h":!0}),y(he)!=="svelte-hswkmf"&&(he.innerHTML=Mt),Ke=a(J),U=d(J,"DIV",{class:!0});var $=de(U);h(ne.$$.fragment,$),et=a($),fe=d($,"P",{"data-svelte-h":!0}),y(fe)!=="svelte-1po8x3s"&&(fe.innerHTML=Tt),tt=a($),h(z.$$.fragment,$),ot=a($),ge=d($,"UL",{"data-svelte-h":!0}),y(ge)!=="svelte-pd3ssm"&&(ge.innerHTML=wt),$.forEach(o),J.forEach(o),Xe=a(e),h(se.$$.fragment,e),Ge=a(e),F=d(e,"DIV",{class:!0});var ie=de(F);h(ae.$$.fragment,ie),nt=a(ie),x=d(ie,"DIV",{class:!0});var I=de(x);h(re.$$.fragment,I),st=a(I),be=d(I,"P",{"data-svelte-h":!0}),y(be)!=="svelte-9m7ses"&&(be.innerHTML=xt),at=a(I),h(Z.$$.fragment,I),rt=a(I),_e=d(I,"UL",{"data-svelte-h":!0}),y(_e)!=="svelte-nqsg2"&&(_e.innerHTML=vt),lt=a(I),h(W.$$.fragment,I),I.forEach(o),ie.forEach(o),qe=a(e),h(le.$$.fragment,e),Ve=a(e),Me=d(e,"P",{}),de(Me).forEach(o),this.h()},h(){D(r,"name","hf:doc:metadata"),D(r,"content",Et),D(N,"class","flex flex-wrap space-x-1"),D(v,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),D(U,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),D(w,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),D(x,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),D(F,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(e,t){p(document.head,r),n(e,T,t),n(e,c,t),n(e,m,t),n(e,M,t),n(e,k,t),f(E,e,t),n(e,xe,t),f(B,e,t),n(e,ve,t),n(e,N,t),n(e,Je,t),f(X,e,t),n(e,ke,t),n(e,G,t),n(e,Ue,t),n(e,q,t),n(e,Ce,t),n(e,V,t),n(e,$e,t),n(e,H,t),n(e,je,t),f(L,e,t),n(e,Ie,t),n(e,Q,t),n(e,Fe,t),f(Y,e,t),n(e,Ne,t),n(e,S,t),n(e,Re,t),f(P,e,t),n(e,ze,t),n(e,A,t),n(e,Ze,t),f(O,e,t),n(e,We,t),f(K,e,t),n(e,De,t),n(e,v,t),f(ee,v,null),p(v,Qe),p(v,ce),p(v,Ye),p(v,pe),p(v,Se),f(R,v,null),n(e,Ee,t),f(te,e,t),n(e,Be,t),n(e,w,t),f(oe,w,null),p(w,Pe),p(w,me),p(w,Ae),p(w,ue),p(w,Oe),p(w,he),p(w,Ke),p(w,U),f(ne,U,null),p(U,et),p(U,fe),p(U,tt),f(z,U,null),p(U,ot),p(U,ge),n(e,Xe,t),f(se,e,t),n(e,Ge,t),n(e,F,t),f(ae,F,null),p(F,nt),p(F,x),f(re,x,null),p(x,st),p(x,be),p(x,at),f(Z,x,null),p(x,rt),p(x,_e),p(x,lt),f(W,x,null),n(e,qe,t),f(le,e,t),n(e,Ve,t),n(e,Me,t),He=!0},p(e,[t]){const C={};t&2&&(C.$$scope={dirty:t,ctx:e}),R.$set(C);const J={};t&2&&(J.$$scope={dirty:t,ctx:e}),z.$set(J);const $={};t&2&&($.$$scope={dirty:t,ctx:e}),Z.$set($);const ie={};t&2&&(ie.$$scope={dirty:t,ctx:e}),W.$set(ie)},i(e){He||(g(E.$$.fragment,e),g(B.$$.fragment,e),g(X.$$.fragment,e),g(L.$$.fragment,e),g(Y.$$.fragment,e),g(P.$$.fragment,e),g(O.$$.fragment,e),g(K.$$.fragment,e),g(ee.$$.fragment,e),g(R.$$.fragment,e),g(te.$$.fragment,e),g(oe.$$.fragment,e),g(ne.$$.fragment,e),g(z.$$.fragment,e),g(se.$$.fragment,e),g(ae.$$.fragment,e),g(re.$$.fragment,e),g(Z.$$.fragment,e),g(W.$$.fragment,e),g(le.$$.fragment,e),He=!0)},o(e){b(E.$$.fragment,e),b(B.$$.fragment,e),b(X.$$.fragment,e),b(L.$$.fragment,e),b(Y.$$.fragment,e),b(P.$$.fragment,e),b(O.$$.fragment,e),b(K.$$.fragment,e),b(ee.$$.fragment,e),b(R.$$.fragment,e),b(te.$$.fragment,e),b(oe.$$.fragment,e),b(ne.$$.fragment,e),b(z.$$.fragment,e),b(se.$$.fragment,e),b(ae.$$.fragment,e),b(re.$$.fragment,e),b(Z.$$.fragment,e),b(W.$$.fragment,e),b(le.$$.fragment,e),He=!1},d(e){e&&(o(T),o(c),o(m),o(M),o(k),o(xe),o(ve),o(N),o(Je),o(ke),o(G),o(Ue),o(q),o(Ce),o(V),o($e),o(H),o(je),o(Ie),o(Q),o(Fe),o(Ne),o(S),o(Re),o(ze),o(A),o(Ze),o(We),o(De),o(v),o(Ee),o(Be),o(w),o(Xe),o(Ge),o(F),o(qe),o(Ve),o(Me)),o(r),_(E,e),_(B,e),_(X,e),_(L,e),_(Y,e),_(P,e),_(O,e),_(K,e),_(ee),_(R),_(te,e),_(oe),_(ne),_(z),_(se,e),_(ae),_(re),_(Z),_(W),_(le,e)}}}const Et='{"title":"DBRX","local":"dbrx","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"Usage Examples","local":"usage-examples","sections":[],"depth":2},{"title":"DbrxConfig","local":"transformers.DbrxConfig","sections":[],"depth":2},{"title":"DbrxModel","local":"transformers.DbrxModel","sections":[],"depth":2},{"title":"DbrxForCausalLM","local":"transformers.DbrxForCausalLM","sections":[],"depth":2}],"depth":1}';function Bt(j){return Ct(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class St extends $t{constructor(r){super(),jt(this,r,Bt,Dt,Ut,{})}}export{St as component}; | |
Xet Storage Details
- Size:
- 53.8 kB
- Xet hash:
- c98423352a0a9de3849d37691f4039f2f791b70be2867e47564eeac98642f1e9
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.