Buckets:
| import{s as lo,o as co,n as we}from"../chunks/scheduler.25b97de1.js";import{S as po,i as mo,g as d,s,r as h,A as ho,h as c,f as n,c as a,j as F,u,x as T,k as L,y as i,a as l,v as f,d as g,t as _,w as b}from"../chunks/index.d9030fc9.js";import{T as dt}from"../chunks/Tip.baa67368.js";import{D as E}from"../chunks/Docstring.ffac8efa.js";import{C as ct}from"../chunks/CodeBlock.e6cd0d95.js";import{E as qt}from"../chunks/ExampleCodeBlock.22dfe688.js";import{H as O,E as uo}from"../chunks/EditOnGithub.91d95064.js";function fo(M){let o,y;return o=new ct({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEdsbU1vZGVsJTJDJTIwR2xtQ29uZmlnJTBBJTIzJTIwSW5pdGlhbGl6aW5nJTIwYSUyMEdsbSUyMGdsbS00LTliLWNoYXQlMjBzdHlsZSUyMGNvbmZpZ3VyYXRpb24lMEFjb25maWd1cmF0aW9uJTIwJTNEJTIwR2xtQ29uZmlnKCklMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwbW9kZWwlMjBmcm9tJTIwdGhlJTIwZ2xtLTQtOWItY2hhdCUyMHN0eWxlJTIwY29uZmlndXJhdGlvbiUwQW1vZGVsJTIwJTNEJTIwR2xtTW9kZWwoY29uZmlndXJhdGlvbiklMEElMjMlMjBBY2Nlc3NpbmclMjB0aGUlMjBtb2RlbCUyMGNvbmZpZ3VyYXRpb24lMEFjb25maWd1cmF0aW9uJTIwJTNEJTIwbW9kZWwuY29uZmln",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> GlmModel, GlmConfig | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a Glm glm-4-9b-chat style configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = GlmConfig() | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a model from the glm-4-9b-chat style configuration</span> | |
| <span class="hljs-meta">>>> </span>model = GlmModel(configuration) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Accessing the model configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = model.config`,wrap:!1}}),{c(){h(o.$$.fragment)},l(r){u(o.$$.fragment,r)},m(r,m){f(o,r,m),y=!0},p:we,i(r){y||(g(o.$$.fragment,r),y=!0)},o(r){_(o.$$.fragment,r),y=!1},d(r){b(o,r)}}}function go(M){let o,y=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=d("p"),o.innerHTML=y},l(r){o=c(r,"P",{"data-svelte-h":!0}),T(o)!=="svelte-fincs2"&&(o.innerHTML=y)},m(r,m){l(r,o,m)},p:we,d(r){r&&n(o)}}}function _o(M){let o,y=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=d("p"),o.innerHTML=y},l(r){o=c(r,"P",{"data-svelte-h":!0}),T(o)!=="svelte-fincs2"&&(o.innerHTML=y)},m(r,m){l(r,o,m)},p:we,d(r){r&&n(o)}}}function bo(M){let o,y="Example:",r,m,v;return m=new ct({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBHbG1Gb3JDYXVzYWxMTSUwQSUwQW1vZGVsJTIwJTNEJTIwR2xtRm9yQ2F1c2FsTE0uZnJvbV9wcmV0cmFpbmVkKCUyMm1ldGEtZ2xtJTJGR2xtLTItN2ItaGYlMjIpJTBBdG9rZW5pemVyJTIwJTNEJTIwQXV0b1Rva2VuaXplci5mcm9tX3ByZXRyYWluZWQoJTIybWV0YS1nbG0lMkZHbG0tMi03Yi1oZiUyMiklMEElMEFwcm9tcHQlMjAlM0QlMjAlMjJIZXklMkMlMjBhcmUlMjB5b3UlMjBjb25zY2lvdXMlM0YlMjBDYW4lMjB5b3UlMjB0YWxrJTIwdG8lMjBtZSUzRiUyMiUwQWlucHV0cyUyMCUzRCUyMHRva2VuaXplcihwcm9tcHQlMkMlMjByZXR1cm5fdGVuc29ycyUzRCUyMnB0JTIyKSUwQSUwQSUyMyUyMEdlbmVyYXRlJTBBZ2VuZXJhdGVfaWRzJTIwJTNEJTIwbW9kZWwuZ2VuZXJhdGUoaW5wdXRzLmlucHV0X2lkcyUyQyUyMG1heF9sZW5ndGglM0QzMCklMEF0b2tlbml6ZXIuYmF0Y2hfZGVjb2RlKGdlbmVyYXRlX2lkcyUyQyUyMHNraXBfc3BlY2lhbF90b2tlbnMlM0RUcnVlJTJDJTIwY2xlYW5fdXBfdG9rZW5pemF0aW9uX3NwYWNlcyUzREZhbHNlKSU1QjAlNUQ=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, GlmForCausalLM | |
| <span class="hljs-meta">>>> </span>model = GlmForCausalLM.from_pretrained(<span class="hljs-string">"meta-glm/Glm-2-7b-hf"</span>) | |
| <span class="hljs-meta">>>> </span>tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"meta-glm/Glm-2-7b-hf"</span>) | |
| <span class="hljs-meta">>>> </span>prompt = <span class="hljs-string">"Hey, are you conscious? Can you talk to me?"</span> | |
| <span class="hljs-meta">>>> </span>inputs = tokenizer(prompt, return_tensors=<span class="hljs-string">"pt"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Generate</span> | |
| <span class="hljs-meta">>>> </span>generate_ids = model.generate(inputs.input_ids, max_length=<span class="hljs-number">30</span>) | |
| <span class="hljs-meta">>>> </span>tokenizer.batch_decode(generate_ids, skip_special_tokens=<span class="hljs-literal">True</span>, clean_up_tokenization_spaces=<span class="hljs-literal">False</span>)[<span class="hljs-number">0</span>] | |
| <span class="hljs-string">"Hey, are you conscious? Can you talk to me?\\nI'm not conscious, but I can talk to you."</span>`,wrap:!1}}),{c(){o=d("p"),o.textContent=y,r=s(),h(m.$$.fragment)},l(p){o=c(p,"P",{"data-svelte-h":!0}),T(o)!=="svelte-11lpom8"&&(o.textContent=y),r=a(p),u(m.$$.fragment,p)},m(p,G){l(p,o,G),l(p,r,G),f(m,p,G),v=!0},p:we,i(p){v||(g(m.$$.fragment,p),v=!0)},o(p){_(m.$$.fragment,p),v=!1},d(p){p&&(n(o),n(r)),b(m,p)}}}function To(M){let o,y=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=d("p"),o.innerHTML=y},l(r){o=c(r,"P",{"data-svelte-h":!0}),T(o)!=="svelte-fincs2"&&(o.innerHTML=y)},m(r,m){l(r,o,m)},p:we,d(r){r&&n(o)}}}function yo(M){let o,y=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=d("p"),o.innerHTML=y},l(r){o=c(r,"P",{"data-svelte-h":!0}),T(o)!=="svelte-fincs2"&&(o.innerHTML=y)},m(r,m){l(r,o,m)},p:we,d(r){r&&n(o)}}}function ko(M){let o,y="Example:",r,m,v;return m=new ct({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMkMlMjBHbG1Gb3JUb2tlbkNsYXNzaWZpY2F0aW9uJTBBaW1wb3J0JTIwdG9yY2glMEElMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZCglMjJUSFVETSUyRmdsbS00LTliJTIyKSUwQW1vZGVsJTIwJTNEJTIwR2xtRm9yVG9rZW5DbGFzc2lmaWNhdGlvbi5mcm9tX3ByZXRyYWluZWQoJTIyVEhVRE0lMkZnbG0tNC05YiUyMiklMEElMEFpbnB1dHMlMjAlM0QlMjB0b2tlbml6ZXIoJTBBJTIwJTIwJTIwJTIwJTIySHVnZ2luZ0ZhY2UlMjBpcyUyMGElMjBjb21wYW55JTIwYmFzZWQlMjBpbiUyMFBhcmlzJTIwYW5kJTIwTmV3JTIwWW9yayUyMiUyQyUyMGFkZF9zcGVjaWFsX3Rva2VucyUzREZhbHNlJTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiUwQSklMEElMEF3aXRoJTIwdG9yY2gubm9fZ3JhZCgpJTNBJTBBJTIwJTIwJTIwJTIwbG9naXRzJTIwJTNEJTIwbW9kZWwoKippbnB1dHMpLmxvZ2l0cyUwQSUwQXByZWRpY3RlZF90b2tlbl9jbGFzc19pZHMlMjAlM0QlMjBsb2dpdHMuYXJnbWF4KC0xKSUwQSUwQSUyMyUyME5vdGUlMjB0aGF0JTIwdG9rZW5zJTIwYXJlJTIwY2xhc3NpZmllZCUyMHJhdGhlciUyMHRoZW4lMjBpbnB1dCUyMHdvcmRzJTIwd2hpY2glMjBtZWFucyUyMHRoYXQlMEElMjMlMjB0aGVyZSUyMG1pZ2h0JTIwYmUlMjBtb3JlJTIwcHJlZGljdGVkJTIwdG9rZW4lMjBjbGFzc2VzJTIwdGhhbiUyMHdvcmRzLiUwQSUyMyUyME11bHRpcGxlJTIwdG9rZW4lMjBjbGFzc2VzJTIwbWlnaHQlMjBhY2NvdW50JTIwZm9yJTIwdGhlJTIwc2FtZSUyMHdvcmQlMEFwcmVkaWN0ZWRfdG9rZW5zX2NsYXNzZXMlMjAlM0QlMjAlNUJtb2RlbC5jb25maWcuaWQybGFiZWwlNUJ0Lml0ZW0oKSU1RCUyMGZvciUyMHQlMjBpbiUyMHByZWRpY3RlZF90b2tlbl9jbGFzc19pZHMlNUIwJTVEJTVEJTBBJTBBbGFiZWxzJTIwJTNEJTIwcHJlZGljdGVkX3Rva2VuX2NsYXNzX2lkcyUwQWxvc3MlMjAlM0QlMjBtb2RlbCgqKmlucHV0cyUyQyUyMGxhYmVscyUzRGxhYmVscykubG9zcw==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer, GlmForTokenClassification | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span>tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"THUDM/glm-4-9b"</span>) | |
| <span class="hljs-meta">>>> </span>model = GlmForTokenClassification.from_pretrained(<span class="hljs-string">"THUDM/glm-4-9b"</span>) | |
| <span class="hljs-meta">>>> </span>inputs = tokenizer( | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"HuggingFace is a company based in Paris and New York"</span>, add_special_tokens=<span class="hljs-literal">False</span>, return_tensors=<span class="hljs-string">"pt"</span> | |
| <span class="hljs-meta">... </span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">with</span> torch.no_grad(): | |
| <span class="hljs-meta">... </span> logits = model(**inputs).logits | |
| <span class="hljs-meta">>>> </span>predicted_token_class_ids = logits.argmax(-<span class="hljs-number">1</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Note that tokens are classified rather then input words which means that</span> | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># there might be more predicted token classes than words.</span> | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Multiple token classes might account for the same word</span> | |
| <span class="hljs-meta">>>> </span>predicted_tokens_classes = [model.config.id2label[t.item()] <span class="hljs-keyword">for</span> t <span class="hljs-keyword">in</span> predicted_token_class_ids[<span class="hljs-number">0</span>]] | |
| <span class="hljs-meta">>>> </span>labels = predicted_token_class_ids | |
| <span class="hljs-meta">>>> </span>loss = model(**inputs, labels=labels).loss`,wrap:!1}}),{c(){o=d("p"),o.textContent=y,r=s(),h(m.$$.fragment)},l(p){o=c(p,"P",{"data-svelte-h":!0}),T(o)!=="svelte-11lpom8"&&(o.textContent=y),r=a(p),u(m.$$.fragment,p)},m(p,G){l(p,o,G),l(p,r,G),f(m,p,G),v=!0},p:we,i(p){v||(g(m.$$.fragment,p),v=!0)},o(p){_(m.$$.fragment,p),v=!1},d(p){p&&(n(o),n(r)),b(m,p)}}}function vo(M){let o,y,r,m,v,p,G,Be,D,Ht=`The GLM Model was proposed | |
| in <a href="https://arxiv.org/html/2406.12793v1" rel="nofollow">ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools</a> | |
| by GLM Team, THUDM & ZhipuAI.`,Ee,K,Nt="The abstract from the paper is the following:",Pe,ee,Bt=`<em>We introduce ChatGLM, an evolving family of large language models that we have been developing over time. This report | |
| primarily focuses on the GLM-4 language series, which includes GLM-4, GLM-4-Air, and GLM-4-9B. They represent our most | |
| capable models that are trained with all the insights and lessons gained from the preceding three generations of | |
| ChatGLM. To date, the GLM-4 models are pre-trained on ten trillions of tokens mostly in Chinese and English, along with | |
| a small set of corpus from 24 languages, and aligned primarily for Chinese and English usage. The high-quality alignment | |
| is achieved via a multi-stage post-training process, which involves supervised fine-tuning and learning from human | |
| feedback. Evaluations show that GLM-4 1) closely rivals or outperforms GPT-4 in terms of general metrics such as MMLU, | |
| GSM8K, MATH, BBH, GPQA, and HumanEval, 2) gets close to GPT-4-Turbo in instruction following as measured by IFEval, 3) | |
| matches GPT-4 Turbo (128K) and Claude 3 for long context tasks, and 4) outperforms GPT-4 in Chinese alignments as | |
| measured by AlignBench. The GLM-4 All Tools model is further aligned to understand user intent and autonomously decide | |
| when and which tool(s) to use—including web browser, Python interpreter, text-to-image model, and user-defined | |
| functions—to effectively complete complex tasks. In practical applications, it matches and even surpasses GPT-4 All | |
| Tools in tasks like accessing online information via web browsing and solving math problems using Python interpreter. | |
| Over the course, we have open-sourced a series of models, including ChatGLM-6B (three generations), GLM-4-9B (128K, 1M), | |
| GLM-4V-9B, WebGLM, and CodeGeeX, attracting over 10 million downloads on Hugging face in the year 2023 alone.</em>`,Se,te,Et="Tips:",Re,oe,Pt=`<li>This model was contributed by <a href="https://huggingface.co/THUDM" rel="nofollow">THUDM</a>. The most recent code can be | |
| found <a href="https://github.com/thudm/GLM-4" rel="nofollow">here</a>.</li>`,Ve,ne,Xe,se,St='<code>GLM-4</code> can be found on the <a href="https://huggingface.co/collections/THUDM/glm-4-665fcf188c414b03c2f7e3b7" rel="nofollow">Huggingface Hub</a>',Ae,ae,Rt="In the following, we demonstrate how to use <code>glm-4-9b-chat</code> for the inference. Note that we have used the ChatML format for dialog, in this demo we show how to leverage <code>apply_chat_template</code> for this purpose.",Qe,re,Ye,ie,Oe,J,le,pt,Me,Vt=`This is the configuration class to store the configuration of a <a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmModel">GlmModel</a>. It is used to instantiate an Glm | |
| model according to the specified arguments, defining the model architecture. Instantiating a configuration with the | |
| defaults will yield a similar configuration to that of the Glm-4-9b-chat. | |
| e.g. <a href="https://huggingface.co/THUDM/glm-4-9b-chat" rel="nofollow">THUDM/glm-4-9b-chat</a> | |
| Configuration objects inherit from <a href="/docs/transformers/pr_35674/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> and can be used to control the model outputs. Read the | |
| documentation from <a href="/docs/transformers/pr_35674/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> for more information.`,mt,P,De,de,Ke,$,ce,ht,Ge,Xt=`The bare Glm Model outputting raw hidden-states without any specific head on top. | |
| This model inherits from <a href="/docs/transformers/pr_35674/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,ut,$e,At=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,ft,Ce,Qt="Transformer decoder consisting of <em>config.num_hidden_layers</em> layers. Each layer is a <code>GlmDecoderLayer</code>",gt,U,pe,_t,ze,Yt='The <a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmModel">GlmModel</a> forward method, overrides the <code>__call__</code> special method.',bt,S,et,me,tt,q,he,Tt,x,ue,yt,xe,Ot='The <a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmForCausalLM">GlmForCausalLM</a> forward method, overrides the <code>__call__</code> special method.',kt,R,vt,V,ot,fe,nt,k,ge,wt,Ie,Dt="The Glm Model transformer with a sequence classification head on top (linear layer).",Mt,Je,Kt=`<a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmForSequenceClassification">GlmForSequenceClassification</a> uses the last token in order to do the classification, as other causal models | |
| (e.g. GPT-2) do.`,Gt,je,eo=`Since it does classification on the last token, it requires to know the position of the last token. If a | |
| <code>pad_token_id</code> is defined in the configuration, it finds the last token that is not a padding token in each row. If | |
| no <code>pad_token_id</code> is defined, it simply takes the last value in each row of the batch. Since it cannot guess the | |
| padding tokens when <code>inputs_embeds</code> are passed instead of <code>input_ids</code>, it does the same (take the last value in | |
| each row of the batch).`,$t,Fe,to=`This model inherits from <a href="/docs/transformers/pr_35674/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,Ct,Le,oo=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,zt,W,_e,xt,Ue,no='The <a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmForSequenceClassification">GlmForSequenceClassification</a> forward method, overrides the <code>__call__</code> special method.',It,X,st,be,at,C,Te,Jt,We,so=`The Glm Model transformer with a token classification head on top (a linear layer on top of the hidden-states | |
| output) e.g. for Named-Entity-Recognition (NER) tasks.`,jt,Ze,ao=`This model inherits from <a href="/docs/transformers/pr_35674/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,Ft,qe,ro=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,Lt,I,ye,Ut,He,io='The <a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmForTokenClassification">GlmForTokenClassification</a> forward method, overrides the <code>__call__</code> special method.',Wt,A,Zt,Q,rt,ke,it,Ne,lt;return v=new O({props:{title:"GLM",local:"glm",headingTag:"h1"}}),G=new O({props:{title:"Overview",local:"overview",headingTag:"h2"}}),ne=new O({props:{title:"Usage tips",local:"usage-tips",headingTag:"h2"}}),re=new ct({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Nb2RlbEZvckNhdXNhbExNJTJDJTIwQXV0b1Rva2VuaXplciUwQWRldmljZSUyMCUzRCUyMCUyMmN1ZGElMjIlMjAlMjMlMjB0aGUlMjBkZXZpY2UlMjB0byUyMGxvYWQlMjB0aGUlMjBtb2RlbCUyMG9udG8lMEElMEFtb2RlbCUyMCUzRCUyMEF1dG9Nb2RlbEZvckNhdXNhbExNLmZyb21fcHJldHJhaW5lZCglMjJUSFVETSUyRmdsbS00LTliLWNoYXQlMjIlMkMlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMiklMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZCglMjJUSFVETSUyRmdsbS00LTliLWNoYXQlMjIpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyR2l2ZSUyMG1lJTIwYSUyMHNob3J0JTIwaW50cm9kdWN0aW9uJTIwdG8lMjBsYXJnZSUyMGxhbmd1YWdlJTIwbW9kZWwuJTIyJTBBJTBBbWVzc2FnZXMlMjAlM0QlMjAlNUIlN0IlMjJyb2xlJTIyJTNBJTIwJTIydXNlciUyMiUyQyUyMCUyMmNvbnRlbnQlMjIlM0ElMjBwcm9tcHQlN0QlNUQlMEElMEF0ZXh0JTIwJTNEJTIwdG9rZW5pemVyLmFwcGx5X2NoYXRfdGVtcGxhdGUobWVzc2FnZXMlMkMlMjB0b2tlbml6ZSUzREZhbHNlJTJDJTIwYWRkX2dlbmVyYXRpb25fcHJvbXB0JTNEVHJ1ZSklMEElMEFtb2RlbF9pbnB1dHMlMjAlM0QlMjB0b2tlbml6ZXIoJTVCdGV4dCU1RCUyQyUyMHJldHVybl90ZW5zb3JzJTNEJTIycHQlMjIpLnRvKGRldmljZSklMEElMEFnZW5lcmF0ZWRfaWRzJTIwJTNEJTIwbW9kZWwuZ2VuZXJhdGUobW9kZWxfaW5wdXRzLmlucHV0X2lkcyUyQyUyMG1heF9uZXdfdG9rZW5zJTNENTEyJTJDJTIwZG9fc2FtcGxlJTNEVHJ1ZSklMEElMEFnZW5lcmF0ZWRfaWRzJTIwJTNEJTIwJTVCb3V0cHV0X2lkcyU1QmxlbihpbnB1dF9pZHMpJTNBJTVEJTIwZm9yJTIwaW5wdXRfaWRzJTJDJTIwb3V0cHV0X2lkcyUyMGluJTIwemlwKG1vZGVsX2lucHV0cy5pbnB1dF9pZHMlMkMlMjBnZW5lcmF0ZWRfaWRzKSU1RCUwQSUwQXJlc3BvbnNlJTIwJTNEJTIwdG9rZW5pemVyLmJhdGNoX2RlY29kZShnZW5lcmF0ZWRfaWRzJTJDJTIwc2tpcF9zcGVjaWFsX3Rva2VucyUzRFRydWUpJTVCMCU1RA==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoModelForCausalLM, AutoTokenizer | |
| <span class="hljs-meta">>>> </span>device = <span class="hljs-string">"cuda"</span> <span class="hljs-comment"># the device to load the model onto</span> | |
| <span class="hljs-meta">>>> </span>model = AutoModelForCausalLM.from_pretrained(<span class="hljs-string">"THUDM/glm-4-9b-chat"</span>, device_map=<span class="hljs-string">"auto"</span>) | |
| <span class="hljs-meta">>>> </span>tokenizer = AutoTokenizer.from_pretrained(<span class="hljs-string">"THUDM/glm-4-9b-chat"</span>) | |
| <span class="hljs-meta">>>> </span>prompt = <span class="hljs-string">"Give me a short introduction to large language model."</span> | |
| <span class="hljs-meta">>>> </span>messages = [{<span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, <span class="hljs-string">"content"</span>: prompt}] | |
| <span class="hljs-meta">>>> </span>text = tokenizer.apply_chat_template(messages, tokenize=<span class="hljs-literal">False</span>, add_generation_prompt=<span class="hljs-literal">True</span>) | |
| <span class="hljs-meta">>>> </span>model_inputs = tokenizer([text], return_tensors=<span class="hljs-string">"pt"</span>).to(device) | |
| <span class="hljs-meta">>>> </span>generated_ids = model.generate(model_inputs.input_ids, max_new_tokens=<span class="hljs-number">512</span>, do_sample=<span class="hljs-literal">True</span>) | |
| <span class="hljs-meta">>>> </span>generated_ids = [output_ids[<span class="hljs-built_in">len</span>(input_ids):] <span class="hljs-keyword">for</span> input_ids, output_ids <span class="hljs-keyword">in</span> <span class="hljs-built_in">zip</span>(model_inputs.input_ids, generated_ids)] | |
| <span class="hljs-meta">>>> </span>response = tokenizer.batch_decode(generated_ids, skip_special_tokens=<span class="hljs-literal">True</span>)[<span class="hljs-number">0</span>]`,wrap:!1}}),ie=new O({props:{title:"GlmConfig",local:"transformers.GlmConfig",headingTag:"h2"}}),le=new E({props:{name:"class transformers.GlmConfig",anchor:"transformers.GlmConfig",parameters:[{name:"vocab_size",val:" = 151552"},{name:"hidden_size",val:" = 4096"},{name:"intermediate_size",val:" = 13696"},{name:"num_hidden_layers",val:" = 40"},{name:"num_attention_heads",val:" = 32"},{name:"num_key_value_heads",val:" = 2"},{name:"partial_rotary_factor",val:" = 0.5"},{name:"head_dim",val:" = 128"},{name:"hidden_act",val:" = 'silu'"},{name:"attention_dropout",val:" = 0.0"},{name:"max_position_embeddings",val:" = 131072"},{name:"initializer_range",val:" = 0.02"},{name:"rms_norm_eps",val:" = 1.5625e-07"},{name:"use_cache",val:" = True"},{name:"tie_word_embeddings",val:" = False"},{name:"rope_theta",val:" = 10000.0"},{name:"pad_token_id",val:" = 151329"},{name:"eos_token_id",val:" = [151329, 151336, 151338]"},{name:"bos_token_id",val:" = None"},{name:"attention_bias",val:" = True"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.GlmConfig.vocab_size",description:`<strong>vocab_size</strong> (<code>int</code>, <em>optional</em>, defaults to 151552) — | |
| Vocabulary size of the Glm model. Defines the number of different tokens that can be represented by the | |
| <code>inputs_ids</code> passed when calling <a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmModel">GlmModel</a>`,name:"vocab_size"},{anchor:"transformers.GlmConfig.hidden_size",description:`<strong>hidden_size</strong> (<code>int</code>, <em>optional</em>, defaults to 4096) — | |
| Dimension of the hidden representations.`,name:"hidden_size"},{anchor:"transformers.GlmConfig.intermediate_size",description:`<strong>intermediate_size</strong> (<code>int</code>, <em>optional</em>, defaults to 13696) — | |
| Dimension of the MLP representations.`,name:"intermediate_size"},{anchor:"transformers.GlmConfig.num_hidden_layers",description:`<strong>num_hidden_layers</strong> (<code>int</code>, <em>optional</em>, defaults to 40) — | |
| Number of hidden layers in the Transformer decoder.`,name:"num_hidden_layers"},{anchor:"transformers.GlmConfig.num_attention_heads",description:`<strong>num_attention_heads</strong> (<code>int</code>, <em>optional</em>, defaults to 32) — | |
| Number of attention heads for each attention layer in the Transformer decoder.`,name:"num_attention_heads"},{anchor:"transformers.GlmConfig.num_key_value_heads",description:`<strong>num_key_value_heads</strong> (<code>int</code>, <em>optional</em>, defaults to 2) — | |
| This is the number of key_value heads that should be used to implement Grouped Query Attention. If | |
| <code>num_key_value_heads=num_attention_heads</code>, the model will use Multi Head Attention (MHA), if | |
| <code>num_key_value_heads=1</code> the model will use Multi Query Attention (MQA) otherwise GQA is used. When | |
| converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed | |
| by meanpooling all the original heads within that group. For more details checkout <a href="https://arxiv.org/pdf/2305.13245.pdf" rel="nofollow">this | |
| paper</a>. If it is not specified, will default to | |
| <code>num_attention_heads</code>.`,name:"num_key_value_heads"},{anchor:"transformers.GlmConfig.partial_rotary_factor",description:"<strong>partial_rotary_factor</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — The factor of the partial rotary position.",name:"partial_rotary_factor"},{anchor:"transformers.GlmConfig.head_dim",description:`<strong>head_dim</strong> (<code>int</code>, <em>optional</em>, defaults to 128) — | |
| The attention head dimension.`,name:"head_dim"},{anchor:"transformers.GlmConfig.hidden_act",description:`<strong>hidden_act</strong> (<code>str</code> or <code>function</code>, <em>optional</em>, defaults to <code>"silu"</code>) — | |
| The legacy activation function. It is overwritten by the <code>hidden_activation</code>.`,name:"hidden_act"},{anchor:"transformers.GlmConfig.attention_dropout",description:`<strong>attention_dropout</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The dropout ratio for the attention probabilities.`,name:"attention_dropout"},{anchor:"transformers.GlmConfig.max_position_embeddings",description:`<strong>max_position_embeddings</strong> (<code>int</code>, <em>optional</em>, defaults to 131072) — | |
| The maximum sequence length that this model might ever be used with.`,name:"max_position_embeddings"},{anchor:"transformers.GlmConfig.initializer_range",description:`<strong>initializer_range</strong> (<code>float</code>, <em>optional</em>, defaults to 0.02) — | |
| The standard deviation of the truncated_normal_initializer for initializing all weight matrices.`,name:"initializer_range"},{anchor:"transformers.GlmConfig.rms_norm_eps",description:`<strong>rms_norm_eps</strong> (<code>float</code>, <em>optional</em>, defaults to 1.5625e-07) — | |
| The epsilon used by the rms normalization layers.`,name:"rms_norm_eps"},{anchor:"transformers.GlmConfig.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not the model should return the last key/values attentions (not used by all models). Only | |
| relevant if <code>config.is_decoder=True</code>.`,name:"use_cache"},{anchor:"transformers.GlmConfig.tie_word_embeddings",description:`<strong>tie_word_embeddings</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to tie weight embeddings`,name:"tie_word_embeddings"},{anchor:"transformers.GlmConfig.rope_theta",description:`<strong>rope_theta</strong> (<code>float</code>, <em>optional</em>, defaults to 10000.0) — | |
| The base period of the RoPE embeddings.`,name:"rope_theta"},{anchor:"transformers.GlmConfig.pad_token_id",description:`<strong>pad_token_id</strong> (<code>int</code>, <em>optional</em>, defaults to 151329) — | |
| Padding token id.`,name:"pad_token_id"},{anchor:"transformers.GlmConfig.eos_token_id",description:`<strong>eos_token_id</strong> (<code>int</code> | <code>list</code>, <em>optional</em>, defaults to <code>[151329, 151336, 151338]</code>) — | |
| End of stream token id.`,name:"eos_token_id"},{anchor:"transformers.GlmConfig.bos_token_id",description:`<strong>bos_token_id</strong> (<code>int</code>, <em>optional</em>) — | |
| Beginning of stream token id.`,name:"bos_token_id"},{anchor:"transformers.GlmConfig.attention_bias",description:`<strong>attention_bias</strong> (<code>bool</code>, defaults to <code>False</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to use a bias in the query, key, value and output projection layers during self-attention.`,name:"attention_bias"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/configuration_glm.py#L20"}}),P=new qt({props:{anchor:"transformers.GlmConfig.example",$$slots:{default:[fo]},$$scope:{ctx:M}}}),de=new O({props:{title:"GlmModel",local:"transformers.GlmModel",headingTag:"h2"}}),ce=new E({props:{name:"class transformers.GlmModel",anchor:"transformers.GlmModel",parameters:[{name:"config",val:": GlmConfig"}],parametersDescription:[{anchor:"transformers.GlmModel.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmConfig">GlmConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_35674/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"},{anchor:"transformers.GlmModel.config",description:"<strong>config</strong> — GlmConfig",name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L493"}}),pe=new E({props:{name:"forward",anchor:"transformers.GlmModel.forward",parameters:[{name:"input_ids",val:": LongTensor = None"},{name:"attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"position_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"past_key_values",val:": typing.Optional[transformers.cache_utils.Cache] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"use_cache",val:": typing.Optional[bool] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"},{name:"cache_position",val:": typing.Optional[torch.LongTensor] = None"},{name:"**flash_attn_kwargs",val:": typing_extensions.Unpack[transformers.modeling_flash_attention_utils.FlashAttentionKwargs]"}],parametersDescription:[{anchor:"transformers.GlmModel.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide | |
| it.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.GlmModel.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p>If <code>past_key_values</code> is used, optionally only the last <code>input_ids</code> have to be input (see | |
| <code>past_key_values</code>).</p> | |
| <p>If you want to change padding behavior, you should read <code>modeling_opt._prepare_decoder_attention_mask</code> | |
| and modify to your needs. See diagram 1 in <a href="https://arxiv.org/abs/1910.13461" rel="nofollow">the paper</a> for more | |
| information on the default strategy.</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"attention_mask"},{anchor:"transformers.GlmModel.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.GlmModel.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>Cache</code> or <code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Two formats are allowed:</p> | |
| <ul> | |
| <li>a <a href="/docs/transformers/pr_35674/en/internal/generation_utils#transformers.Cache">Cache</a> instance, see our | |
| <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>;</li> | |
| <li>Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of | |
| shape <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>). This is also known as the legacy | |
| cache format.</li> | |
| </ul> | |
| <p>The model will output the same cache format that is fed as input. If no <code>past_key_values</code> are passed, the | |
| legacy cache format will be returned.</p> | |
| <p>If <code>past_key_values</code> are used, the user can optionally input only the last <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, 1)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.GlmModel.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.GlmModel.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.GlmModel.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.GlmModel.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.GlmModel.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_35674/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.GlmModel.forward.cache_position",description:`<strong>cache_position</strong> (<code>torch.LongTensor</code> of shape <code>(sequence_length)</code>, <em>optional</em>) — | |
| Indices depicting the position of the input sequence tokens in the sequence. Contrarily to <code>position_ids</code>, | |
| this tensor is not affected by padding. It is used to update the cache in the correct position and to infer | |
| the complete sequence length.`,name:"cache_position"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L527"}}),S=new dt({props:{$$slots:{default:[go]},$$scope:{ctx:M}}}),me=new O({props:{title:"GlmForCausalLM",local:"transformers.GlmForCausalLM",headingTag:"h2"}}),he=new E({props:{name:"class transformers.GlmForCausalLM",anchor:"transformers.GlmForCausalLM",parameters:[{name:"config",val:""}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L759"}}),ue=new E({props:{name:"forward",anchor:"transformers.GlmForCausalLM.forward",parameters:[{name:"input_ids",val:": LongTensor = None"},{name:"attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"position_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"past_key_values",val:": typing.Union[transformers.cache_utils.Cache, typing.List[torch.FloatTensor], NoneType] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"labels",val:": typing.Optional[torch.LongTensor] = None"},{name:"use_cache",val:": typing.Optional[bool] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"},{name:"cache_position",val:": typing.Optional[torch.LongTensor] = None"},{name:"num_logits_to_keep",val:": int = 0"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.models.glm.modeling_glm.KwargsForCausalLM]"}],parametersDescription:[{anchor:"transformers.GlmForCausalLM.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide | |
| it.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.GlmForCausalLM.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p>If <code>past_key_values</code> is used, optionally only the last <code>input_ids</code> have to be input (see | |
| <code>past_key_values</code>).</p> | |
| <p>If you want to change padding behavior, you should read <code>modeling_opt._prepare_decoder_attention_mask</code> | |
| and modify to your needs. See diagram 1 in <a href="https://arxiv.org/abs/1910.13461" rel="nofollow">the paper</a> for more | |
| information on the default strategy.</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"attention_mask"},{anchor:"transformers.GlmForCausalLM.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.GlmForCausalLM.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>Cache</code> or <code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Two formats are allowed:</p> | |
| <ul> | |
| <li>a <a href="/docs/transformers/pr_35674/en/internal/generation_utils#transformers.Cache">Cache</a> instance, see our | |
| <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>;</li> | |
| <li>Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of | |
| shape <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>). This is also known as the legacy | |
| cache format.</li> | |
| </ul> | |
| <p>The model will output the same cache format that is fed as input. If no <code>past_key_values</code> are passed, the | |
| legacy cache format will be returned.</p> | |
| <p>If <code>past_key_values</code> are used, the user can optionally input only the last <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, 1)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.GlmForCausalLM.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.GlmForCausalLM.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.GlmForCausalLM.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.GlmForCausalLM.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.GlmForCausalLM.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_35674/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.GlmForCausalLM.forward.cache_position",description:`<strong>cache_position</strong> (<code>torch.LongTensor</code> of shape <code>(sequence_length)</code>, <em>optional</em>) — | |
| Indices depicting the position of the input sequence tokens in the sequence. Contrarily to <code>position_ids</code>, | |
| this tensor is not affected by padding. It is used to update the cache in the correct position and to infer | |
| the complete sequence length.`,name:"cache_position"},{anchor:"transformers.GlmForCausalLM.forward.Args",description:`<strong>Args</strong> — | |
| labels (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>): | |
| Labels for computing the masked language modeling loss. Indices should either be in <code>[0, ..., config.vocab_size]</code> or -100 (see <code>input_ids</code> docstring). Tokens with indices set to <code>-100</code> are ignored | |
| (masked), the loss is only computed for the tokens with labels in <code>[0, ..., config.vocab_size]</code>.</p> | |
| <p>num_logits_to_keep (<code>int</code>, <em>optional</em>): | |
| Calculate logits for the last <code>num_logits_to_keep</code> tokens. If <code>0</code>, calculate logits for all | |
| <code>input_ids</code> (special case). Only last token logits are needed for generation, and calculating them only for that | |
| token can save memory, which becomes pretty significant for long sequences or large vocabulary size.`,name:"Args"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L790",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_35674/en/main_classes/output#transformers.modeling_outputs.CausalLMOutputWithPast" | |
| >transformers.modeling_outputs.CausalLMOutputWithPast</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmConfig" | |
| >GlmConfig</a>) and inputs.</p> | |
| <ul> | |
| <li> | |
| <p><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> is provided) — Language modeling loss (for next-token prediction).</p> | |
| </li> | |
| <li> | |
| <p><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, config.vocab_size)</code>) — Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).</p> | |
| </li> | |
| <li> | |
| <p><strong>past_key_values</strong> (<code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of shape | |
| <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>)</p> | |
| <p>Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see | |
| <code>past_key_values</code> input) to speed up sequential decoding.</p> | |
| </li> | |
| <li> | |
| <p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> | |
| <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p> | |
| </li> | |
| <li> | |
| <p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_35674/en/main_classes/output#transformers.modeling_outputs.CausalLMOutputWithPast" | |
| >transformers.modeling_outputs.CausalLMOutputWithPast</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),R=new dt({props:{$$slots:{default:[_o]},$$scope:{ctx:M}}}),V=new qt({props:{anchor:"transformers.GlmForCausalLM.forward.example",$$slots:{default:[bo]},$$scope:{ctx:M}}}),fe=new O({props:{title:"GlmForSequenceClassification",local:"transformers.GlmForSequenceClassification",headingTag:"h2"}}),ge=new E({props:{name:"class transformers.GlmForSequenceClassification",anchor:"transformers.GlmForSequenceClassification",parameters:[{name:"config",val:""}],parametersDescription:[{anchor:"transformers.GlmForSequenceClassification.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmConfig">GlmConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_35674/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L880"}}),_e=new E({props:{name:"forward",anchor:"transformers.GlmForSequenceClassification.forward",parameters:[{name:"input_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"position_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"past_key_values",val:": typing.Union[transformers.cache_utils.Cache, typing.List[torch.FloatTensor], NoneType] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"labels",val:": typing.Optional[torch.LongTensor] = None"},{name:"use_cache",val:": typing.Optional[bool] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"}],parametersDescription:[{anchor:"transformers.GlmForSequenceClassification.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide | |
| it.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.GlmForSequenceClassification.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p>If <code>past_key_values</code> is used, optionally only the last <code>input_ids</code> have to be input (see | |
| <code>past_key_values</code>).</p> | |
| <p>If you want to change padding behavior, you should read <code>modeling_opt._prepare_decoder_attention_mask</code> | |
| and modify to your needs. See diagram 1 in <a href="https://arxiv.org/abs/1910.13461" rel="nofollow">the paper</a> for more | |
| information on the default strategy.</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"attention_mask"},{anchor:"transformers.GlmForSequenceClassification.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.GlmForSequenceClassification.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>Cache</code> or <code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Two formats are allowed:</p> | |
| <ul> | |
| <li>a <a href="/docs/transformers/pr_35674/en/internal/generation_utils#transformers.Cache">Cache</a> instance, see our | |
| <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>;</li> | |
| <li>Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of | |
| shape <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>). This is also known as the legacy | |
| cache format.</li> | |
| </ul> | |
| <p>The model will output the same cache format that is fed as input. If no <code>past_key_values</code> are passed, the | |
| legacy cache format will be returned.</p> | |
| <p>If <code>past_key_values</code> are used, the user can optionally input only the last <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, 1)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.GlmForSequenceClassification.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.GlmForSequenceClassification.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.GlmForSequenceClassification.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.GlmForSequenceClassification.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.GlmForSequenceClassification.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_35674/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.GlmForSequenceClassification.forward.cache_position",description:`<strong>cache_position</strong> (<code>torch.LongTensor</code> of shape <code>(sequence_length)</code>, <em>optional</em>) — | |
| Indices depicting the position of the input sequence tokens in the sequence. Contrarily to <code>position_ids</code>, | |
| this tensor is not affected by padding. It is used to update the cache in the correct position and to infer | |
| the complete sequence length.`,name:"cache_position"},{anchor:"transformers.GlmForSequenceClassification.forward.labels",description:`<strong>labels</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size,)</code>, <em>optional</em>) — | |
| Labels for computing the sequence classification/regression loss. Indices should be in <code>[0, ..., config.num_labels - 1]</code>. If <code>config.num_labels == 1</code> a regression loss is computed (Mean-Square loss), If | |
| <code>config.num_labels > 1</code> a classification loss is computed (Cross-Entropy).`,name:"labels"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L911"}}),X=new dt({props:{$$slots:{default:[To]},$$scope:{ctx:M}}}),be=new O({props:{title:"GlmForTokenClassification",local:"transformers.GlmForTokenClassification",headingTag:"h2"}}),Te=new E({props:{name:"class transformers.GlmForTokenClassification",anchor:"transformers.GlmForTokenClassification",parameters:[{name:"config",val:""}],parametersDescription:[{anchor:"transformers.GlmForTokenClassification.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmConfig">GlmConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_35674/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L984"}}),ye=new E({props:{name:"forward",anchor:"transformers.GlmForTokenClassification.forward",parameters:[{name:"input_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"position_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"past_key_values",val:": typing.Optional[typing.List[torch.FloatTensor]] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"labels",val:": typing.Optional[torch.LongTensor] = None"},{name:"use_cache",val:": typing.Optional[bool] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"}],parametersDescription:[{anchor:"transformers.GlmForTokenClassification.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide | |
| it.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.GlmForTokenClassification.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_35674/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_35674/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p>If <code>past_key_values</code> is used, optionally only the last <code>input_ids</code> have to be input (see | |
| <code>past_key_values</code>).</p> | |
| <p>If you want to change padding behavior, you should read <code>modeling_opt._prepare_decoder_attention_mask</code> | |
| and modify to your needs. See diagram 1 in <a href="https://arxiv.org/abs/1910.13461" rel="nofollow">the paper</a> for more | |
| information on the default strategy.</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"attention_mask"},{anchor:"transformers.GlmForTokenClassification.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.GlmForTokenClassification.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>Cache</code> or <code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Two formats are allowed:</p> | |
| <ul> | |
| <li>a <a href="/docs/transformers/pr_35674/en/internal/generation_utils#transformers.Cache">Cache</a> instance, see our | |
| <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>;</li> | |
| <li>Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of | |
| shape <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>). This is also known as the legacy | |
| cache format.</li> | |
| </ul> | |
| <p>The model will output the same cache format that is fed as input. If no <code>past_key_values</code> are passed, the | |
| legacy cache format will be returned.</p> | |
| <p>If <code>past_key_values</code> are used, the user can optionally input only the last <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, 1)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.GlmForTokenClassification.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.GlmForTokenClassification.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.GlmForTokenClassification.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.GlmForTokenClassification.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.GlmForTokenClassification.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_35674/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.GlmForTokenClassification.forward.cache_position",description:`<strong>cache_position</strong> (<code>torch.LongTensor</code> of shape <code>(sequence_length)</code>, <em>optional</em>) — | |
| Indices depicting the position of the input sequence tokens in the sequence. Contrarily to <code>position_ids</code>, | |
| this tensor is not affected by padding. It is used to update the cache in the correct position and to infer | |
| the complete sequence length.`,name:"cache_position"},{anchor:"transformers.GlmForTokenClassification.forward.labels",description:`<strong>labels</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size,)</code>, <em>optional</em>) — | |
| Labels for computing the sequence classification/regression loss. Indices should be in <code>[0, ..., config.num_labels - 1]</code>. If <code>config.num_labels == 1</code> a regression loss is computed (Mean-Square loss), If | |
| <code>config.num_labels > 1</code> a classification loss is computed (Cross-Entropy).`,name:"labels"}],source:"https://github.com/huggingface/transformers/blob/vr_35674/src/transformers/models/glm/modeling_glm.py#L1014",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_35674/en/main_classes/output#transformers.modeling_outputs.TokenClassifierOutput" | |
| >transformers.modeling_outputs.TokenClassifierOutput</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_35674/en/model_doc/glm#transformers.GlmConfig" | |
| >GlmConfig</a>) and inputs.</p> | |
| <ul> | |
| <li> | |
| <p><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> is provided) — Classification loss.</p> | |
| </li> | |
| <li> | |
| <p><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, config.num_labels)</code>) — Classification scores (before SoftMax).</p> | |
| </li> | |
| <li> | |
| <p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> | |
| <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p> | |
| </li> | |
| <li> | |
| <p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_35674/en/main_classes/output#transformers.modeling_outputs.TokenClassifierOutput" | |
| >transformers.modeling_outputs.TokenClassifierOutput</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),A=new dt({props:{$$slots:{default:[yo]},$$scope:{ctx:M}}}),Q=new qt({props:{anchor:"transformers.GlmForTokenClassification.forward.example",$$slots:{default:[ko]},$$scope:{ctx:M}}}),ke=new uo({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/glm.md"}}),{c(){o=d("meta"),y=s(),r=d("p"),m=s(),h(v.$$.fragment),p=s(),h(G.$$.fragment),Be=s(),D=d("p"),D.innerHTML=Ht,Ee=s(),K=d("p"),K.textContent=Nt,Pe=s(),ee=d("p"),ee.innerHTML=Bt,Se=s(),te=d("p"),te.textContent=Et,Re=s(),oe=d("ul"),oe.innerHTML=Pt,Ve=s(),h(ne.$$.fragment),Xe=s(),se=d("p"),se.innerHTML=St,Ae=s(),ae=d("p"),ae.innerHTML=Rt,Qe=s(),h(re.$$.fragment),Ye=s(),h(ie.$$.fragment),Oe=s(),J=d("div"),h(le.$$.fragment),pt=s(),Me=d("p"),Me.innerHTML=Vt,mt=s(),h(P.$$.fragment),De=s(),h(de.$$.fragment),Ke=s(),$=d("div"),h(ce.$$.fragment),ht=s(),Ge=d("p"),Ge.innerHTML=Xt,ut=s(),$e=d("p"),$e.innerHTML=At,ft=s(),Ce=d("p"),Ce.innerHTML=Qt,gt=s(),U=d("div"),h(pe.$$.fragment),_t=s(),ze=d("p"),ze.innerHTML=Yt,bt=s(),h(S.$$.fragment),et=s(),h(me.$$.fragment),tt=s(),q=d("div"),h(he.$$.fragment),Tt=s(),x=d("div"),h(ue.$$.fragment),yt=s(),xe=d("p"),xe.innerHTML=Ot,kt=s(),h(R.$$.fragment),vt=s(),h(V.$$.fragment),ot=s(),h(fe.$$.fragment),nt=s(),k=d("div"),h(ge.$$.fragment),wt=s(),Ie=d("p"),Ie.textContent=Dt,Mt=s(),Je=d("p"),Je.innerHTML=Kt,Gt=s(),je=d("p"),je.innerHTML=eo,$t=s(),Fe=d("p"),Fe.innerHTML=to,Ct=s(),Le=d("p"),Le.innerHTML=oo,zt=s(),W=d("div"),h(_e.$$.fragment),xt=s(),Ue=d("p"),Ue.innerHTML=no,It=s(),h(X.$$.fragment),st=s(),h(be.$$.fragment),at=s(),C=d("div"),h(Te.$$.fragment),Jt=s(),We=d("p"),We.textContent=so,jt=s(),Ze=d("p"),Ze.innerHTML=ao,Ft=s(),qe=d("p"),qe.innerHTML=ro,Lt=s(),I=d("div"),h(ye.$$.fragment),Ut=s(),He=d("p"),He.innerHTML=io,Wt=s(),h(A.$$.fragment),Zt=s(),h(Q.$$.fragment),rt=s(),h(ke.$$.fragment),it=s(),Ne=d("p"),this.h()},l(e){const t=ho("svelte-u9bgzb",document.head);o=c(t,"META",{name:!0,content:!0}),t.forEach(n),y=a(e),r=c(e,"P",{}),F(r).forEach(n),m=a(e),u(v.$$.fragment,e),p=a(e),u(G.$$.fragment,e),Be=a(e),D=c(e,"P",{"data-svelte-h":!0}),T(D)!=="svelte-1twrbfe"&&(D.innerHTML=Ht),Ee=a(e),K=c(e,"P",{"data-svelte-h":!0}),T(K)!=="svelte-vfdo9a"&&(K.textContent=Nt),Pe=a(e),ee=c(e,"P",{"data-svelte-h":!0}),T(ee)!=="svelte-57yk8e"&&(ee.innerHTML=Bt),Se=a(e),te=c(e,"P",{"data-svelte-h":!0}),T(te)!=="svelte-axv494"&&(te.textContent=Et),Re=a(e),oe=c(e,"UL",{"data-svelte-h":!0}),T(oe)!=="svelte-1mv2pve"&&(oe.innerHTML=Pt),Ve=a(e),u(ne.$$.fragment,e),Xe=a(e),se=c(e,"P",{"data-svelte-h":!0}),T(se)!=="svelte-8dtdop"&&(se.innerHTML=St),Ae=a(e),ae=c(e,"P",{"data-svelte-h":!0}),T(ae)!=="svelte-grk520"&&(ae.innerHTML=Rt),Qe=a(e),u(re.$$.fragment,e),Ye=a(e),u(ie.$$.fragment,e),Oe=a(e),J=c(e,"DIV",{class:!0});var H=F(J);u(le.$$.fragment,H),pt=a(H),Me=c(H,"P",{"data-svelte-h":!0}),T(Me)!=="svelte-1ivjwwq"&&(Me.innerHTML=Vt),mt=a(H),u(P.$$.fragment,H),H.forEach(n),De=a(e),u(de.$$.fragment,e),Ke=a(e),$=c(e,"DIV",{class:!0});var z=F($);u(ce.$$.fragment,z),ht=a(z),Ge=c(z,"P",{"data-svelte-h":!0}),T(Ge)!=="svelte-ihq7uf"&&(Ge.innerHTML=Xt),ut=a(z),$e=c(z,"P",{"data-svelte-h":!0}),T($e)!=="svelte-hswkmf"&&($e.innerHTML=At),ft=a(z),Ce=c(z,"P",{"data-svelte-h":!0}),T(Ce)!=="svelte-13xhxxj"&&(Ce.innerHTML=Qt),gt=a(z),U=c(z,"DIV",{class:!0});var N=F(U);u(pe.$$.fragment,N),_t=a(N),ze=c(N,"P",{"data-svelte-h":!0}),T(ze)!=="svelte-95bv2s"&&(ze.innerHTML=Yt),bt=a(N),u(S.$$.fragment,N),N.forEach(n),z.forEach(n),et=a(e),u(me.$$.fragment,e),tt=a(e),q=c(e,"DIV",{class:!0});var ve=F(q);u(he.$$.fragment,ve),Tt=a(ve),x=c(ve,"DIV",{class:!0});var j=F(x);u(ue.$$.fragment,j),yt=a(j),xe=c(j,"P",{"data-svelte-h":!0}),T(xe)!=="svelte-2qtvp4"&&(xe.innerHTML=Ot),kt=a(j),u(R.$$.fragment,j),vt=a(j),u(V.$$.fragment,j),j.forEach(n),ve.forEach(n),ot=a(e),u(fe.$$.fragment,e),nt=a(e),k=c(e,"DIV",{class:!0});var w=F(k);u(ge.$$.fragment,w),wt=a(w),Ie=c(w,"P",{"data-svelte-h":!0}),T(Ie)!=="svelte-dtb9he"&&(Ie.textContent=Dt),Mt=a(w),Je=c(w,"P",{"data-svelte-h":!0}),T(Je)!=="svelte-2odptb"&&(Je.innerHTML=Kt),Gt=a(w),je=c(w,"P",{"data-svelte-h":!0}),T(je)!=="svelte-10ugs3m"&&(je.innerHTML=eo),$t=a(w),Fe=c(w,"P",{"data-svelte-h":!0}),T(Fe)!=="svelte-az7ywp"&&(Fe.innerHTML=to),Ct=a(w),Le=c(w,"P",{"data-svelte-h":!0}),T(Le)!=="svelte-hswkmf"&&(Le.innerHTML=oo),zt=a(w),W=c(w,"DIV",{class:!0});var B=F(W);u(_e.$$.fragment,B),xt=a(B),Ue=c(B,"P",{"data-svelte-h":!0}),T(Ue)!=="svelte-1px3dae"&&(Ue.innerHTML=no),It=a(B),u(X.$$.fragment,B),B.forEach(n),w.forEach(n),st=a(e),u(be.$$.fragment,e),at=a(e),C=c(e,"DIV",{class:!0});var Z=F(C);u(Te.$$.fragment,Z),Jt=a(Z),We=c(Z,"P",{"data-svelte-h":!0}),T(We)!=="svelte-su1m9x"&&(We.textContent=so),jt=a(Z),Ze=c(Z,"P",{"data-svelte-h":!0}),T(Ze)!=="svelte-az7ywp"&&(Ze.innerHTML=ao),Ft=a(Z),qe=c(Z,"P",{"data-svelte-h":!0}),T(qe)!=="svelte-hswkmf"&&(qe.innerHTML=ro),Lt=a(Z),I=c(Z,"DIV",{class:!0});var Y=F(I);u(ye.$$.fragment,Y),Ut=a(Y),He=c(Y,"P",{"data-svelte-h":!0}),T(He)!=="svelte-183g9o6"&&(He.innerHTML=io),Wt=a(Y),u(A.$$.fragment,Y),Zt=a(Y),u(Q.$$.fragment,Y),Y.forEach(n),Z.forEach(n),rt=a(e),u(ke.$$.fragment,e),it=a(e),Ne=c(e,"P",{}),F(Ne).forEach(n),this.h()},h(){L(o,"name","hf:doc:metadata"),L(o,"content",wo),L(J,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(U,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L($,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(x,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(q,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(W,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(k,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(I,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),L(C,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(e,t){i(document.head,o),l(e,y,t),l(e,r,t),l(e,m,t),f(v,e,t),l(e,p,t),f(G,e,t),l(e,Be,t),l(e,D,t),l(e,Ee,t),l(e,K,t),l(e,Pe,t),l(e,ee,t),l(e,Se,t),l(e,te,t),l(e,Re,t),l(e,oe,t),l(e,Ve,t),f(ne,e,t),l(e,Xe,t),l(e,se,t),l(e,Ae,t),l(e,ae,t),l(e,Qe,t),f(re,e,t),l(e,Ye,t),f(ie,e,t),l(e,Oe,t),l(e,J,t),f(le,J,null),i(J,pt),i(J,Me),i(J,mt),f(P,J,null),l(e,De,t),f(de,e,t),l(e,Ke,t),l(e,$,t),f(ce,$,null),i($,ht),i($,Ge),i($,ut),i($,$e),i($,ft),i($,Ce),i($,gt),i($,U),f(pe,U,null),i(U,_t),i(U,ze),i(U,bt),f(S,U,null),l(e,et,t),f(me,e,t),l(e,tt,t),l(e,q,t),f(he,q,null),i(q,Tt),i(q,x),f(ue,x,null),i(x,yt),i(x,xe),i(x,kt),f(R,x,null),i(x,vt),f(V,x,null),l(e,ot,t),f(fe,e,t),l(e,nt,t),l(e,k,t),f(ge,k,null),i(k,wt),i(k,Ie),i(k,Mt),i(k,Je),i(k,Gt),i(k,je),i(k,$t),i(k,Fe),i(k,Ct),i(k,Le),i(k,zt),i(k,W),f(_e,W,null),i(W,xt),i(W,Ue),i(W,It),f(X,W,null),l(e,st,t),f(be,e,t),l(e,at,t),l(e,C,t),f(Te,C,null),i(C,Jt),i(C,We),i(C,jt),i(C,Ze),i(C,Ft),i(C,qe),i(C,Lt),i(C,I),f(ye,I,null),i(I,Ut),i(I,He),i(I,Wt),f(A,I,null),i(I,Zt),f(Q,I,null),l(e,rt,t),f(ke,e,t),l(e,it,t),l(e,Ne,t),lt=!0},p(e,[t]){const H={};t&2&&(H.$$scope={dirty:t,ctx:e}),P.$set(H);const z={};t&2&&(z.$$scope={dirty:t,ctx:e}),S.$set(z);const N={};t&2&&(N.$$scope={dirty:t,ctx:e}),R.$set(N);const ve={};t&2&&(ve.$$scope={dirty:t,ctx:e}),V.$set(ve);const j={};t&2&&(j.$$scope={dirty:t,ctx:e}),X.$set(j);const w={};t&2&&(w.$$scope={dirty:t,ctx:e}),A.$set(w);const B={};t&2&&(B.$$scope={dirty:t,ctx:e}),Q.$set(B)},i(e){lt||(g(v.$$.fragment,e),g(G.$$.fragment,e),g(ne.$$.fragment,e),g(re.$$.fragment,e),g(ie.$$.fragment,e),g(le.$$.fragment,e),g(P.$$.fragment,e),g(de.$$.fragment,e),g(ce.$$.fragment,e),g(pe.$$.fragment,e),g(S.$$.fragment,e),g(me.$$.fragment,e),g(he.$$.fragment,e),g(ue.$$.fragment,e),g(R.$$.fragment,e),g(V.$$.fragment,e),g(fe.$$.fragment,e),g(ge.$$.fragment,e),g(_e.$$.fragment,e),g(X.$$.fragment,e),g(be.$$.fragment,e),g(Te.$$.fragment,e),g(ye.$$.fragment,e),g(A.$$.fragment,e),g(Q.$$.fragment,e),g(ke.$$.fragment,e),lt=!0)},o(e){_(v.$$.fragment,e),_(G.$$.fragment,e),_(ne.$$.fragment,e),_(re.$$.fragment,e),_(ie.$$.fragment,e),_(le.$$.fragment,e),_(P.$$.fragment,e),_(de.$$.fragment,e),_(ce.$$.fragment,e),_(pe.$$.fragment,e),_(S.$$.fragment,e),_(me.$$.fragment,e),_(he.$$.fragment,e),_(ue.$$.fragment,e),_(R.$$.fragment,e),_(V.$$.fragment,e),_(fe.$$.fragment,e),_(ge.$$.fragment,e),_(_e.$$.fragment,e),_(X.$$.fragment,e),_(be.$$.fragment,e),_(Te.$$.fragment,e),_(ye.$$.fragment,e),_(A.$$.fragment,e),_(Q.$$.fragment,e),_(ke.$$.fragment,e),lt=!1},d(e){e&&(n(y),n(r),n(m),n(p),n(Be),n(D),n(Ee),n(K),n(Pe),n(ee),n(Se),n(te),n(Re),n(oe),n(Ve),n(Xe),n(se),n(Ae),n(ae),n(Qe),n(Ye),n(Oe),n(J),n(De),n(Ke),n($),n(et),n(tt),n(q),n(ot),n(nt),n(k),n(st),n(at),n(C),n(rt),n(it),n(Ne)),n(o),b(v,e),b(G,e),b(ne,e),b(re,e),b(ie,e),b(le),b(P),b(de,e),b(ce),b(pe),b(S),b(me,e),b(he),b(ue),b(R),b(V),b(fe,e),b(ge),b(_e),b(X),b(be,e),b(Te),b(ye),b(A),b(Q),b(ke,e)}}}const wo='{"title":"GLM","local":"glm","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"Usage tips","local":"usage-tips","sections":[],"depth":2},{"title":"GlmConfig","local":"transformers.GlmConfig","sections":[],"depth":2},{"title":"GlmModel","local":"transformers.GlmModel","sections":[],"depth":2},{"title":"GlmForCausalLM","local":"transformers.GlmForCausalLM","sections":[],"depth":2},{"title":"GlmForSequenceClassification","local":"transformers.GlmForSequenceClassification","sections":[],"depth":2},{"title":"GlmForTokenClassification","local":"transformers.GlmForTokenClassification","sections":[],"depth":2}],"depth":1}';function Mo(M){return co(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class jo extends po{constructor(o){super(),mo(this,o,Mo,vo,lo,{})}}export{jo as component}; | |
Xet Storage Details
- Size:
- 81 kB
- Xet hash:
- 465c174f074aa4e24f87222242f7f32c47bb275b9a588fac53e5caf0d81b1745
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.