Buckets:
| import{s as Ct,o as wt,n as ee}from"../chunks/scheduler.31fdf58d.js";import{S as Ut,i as bt,e as m,s as a,c as g,h as vt,a as h,d as r,b as i,f as F,j as C,g as f,k,l,m as p,n as _,t as y,o as M,p as T}from"../chunks/index.2f76fdf0.js";import{T as Mt}from"../chunks/Tip.8d349121.js";import{C as Jt}from"../chunks/CopyLLMTxtMenu.53b607bf.js";import{D as A}from"../chunks/Docstring.7acc6835.js";import{C as Le}from"../chunks/CodeBlock.e52df5d6.js";import{E as _o}from"../chunks/ExampleCodeBlock.f9704f52.js";import{H as K,E as It}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.08750ec0.js";import{H as jt,a as Tt}from"../chunks/HfOption.fb051768.js";function $t(U){let o,u;return o=new Le({props:{code:"JTBBZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Nb2RlbEZvckltYWdlVGV4dFRvVGV4dCUyQyUyMEF1dG9Qcm9jZXNzb3IlMEElMEElMEFtb2RlbF9pZCUyMCUzRCUyMCUyMkNvaGVyZUxhYnMlMkZjb21tYW5kLWEtdmlzaW9uLTA3LTIwMjUlMjIlMEElMEFwcm9jZXNzb3IlMjAlM0QlMjBBdXRvUHJvY2Vzc29yLmZyb21fcHJldHJhaW5lZChtb2RlbF9pZCklMEFtb2RlbCUyMCUzRCUyMEF1dG9Nb2RlbEZvckltYWdlVGV4dFRvVGV4dC5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwbW9kZWxfaWQlMkMlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMiUwQSklMEElMEElMjMlMjBGb3JtYXQlMjBtZXNzYWdlJTIwd2l0aCUyMHRoZSUyMENvbW1hbmQtQS1WaXNpb24lMjBjaGF0JTIwdGVtcGxhdGUlMEFtZXNzYWdlcyUyMCUzRCUyMCU1QiUwQSUyMCUyMCUyMCUyMCU3QiUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMnJvbGUlMjIlM0ElMjAlMjJ1c2VyJTIyJTJDJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIyY29udGVudCUyMiUzQSUyMCU1QiUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCU3QiUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMnR5cGUlMjIlM0ElMjAlMjJpbWFnZSUyMiUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMnVybCUyMiUzQSUyMCUyMmh0dHBzJTNBJTJGJTJGaW1hZ2VzLnBleGVscy5jb20lMkZwaG90b3MlMkYxMTA4MDk5JTJGcGV4ZWxzLXBob3RvLTExMDgwOTkuanBlZyUyMiUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCU3RCUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCU3QiUyMnR5cGUlMjIlM0ElMjAlMjJ0ZXh0JTIyJTJDJTIwJTIydGV4dCUyMiUzQSUyMCUyMndoYXQlMjBpcyUyMGluJTIwdGhpcyUyMGltYWdlJTNGJTIyJTdEJTJDJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTVEJTJDJTBBJTIwJTIwJTIwJTIwJTdEJTJDJTBBJTVEJTBBJTBBaW5wdXRzJTIwJTNEJTIwcHJvY2Vzc29yLmFwcGx5X2NoYXRfdGVtcGxhdGUoJTBBJTIwJTIwJTIwJTIwbWVzc2FnZXMlMkMlMEElMjAlMjAlMjAlMjBwYWRkaW5nJTNEVHJ1ZSUyQyUwQSUyMCUyMCUyMCUyMGFkZF9nZW5lcmF0aW9uX3Byb21wdCUzRFRydWUlMkMlMEElMjAlMjAlMjAlMjB0b2tlbml6ZSUzRFRydWUlMkMlMEElMjAlMjAlMjAlMjByZXR1cm5fZGljdCUzRFRydWUlMkMlMEElMjAlMjAlMjAlMjByZXR1cm5fdGVuc29ycyUzRCUyMnB0JTIyJTJDJTBBKS50byhtb2RlbC5kZXZpY2UpJTBBJTBBZ2VuX3Rva2VucyUyMCUzRCUyMG1vZGVsLmdlbmVyYXRlKCUwQSUyMCUyMCUyMCUyMCoqaW5wdXRzJTJDJTBBJTIwJTIwJTIwJTIwbWF4X25ld190b2tlbnMlM0QzMDAlMkMlMEElMjAlMjAlMjAlMjBkb19zYW1wbGUlM0RUcnVlJTJDJTBBJTIwJTIwJTIwJTIwdGVtcGVyYXR1cmUlM0QwLjMlMkMlMEEpJTBBJTBBcHJpbnQoJTBBJTIwJTIwJTIwJTIwcHJvY2Vzc29yLnRva2VuaXplci5kZWNvZGUoJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwZ2VuX3Rva2VucyU1QjAlNUQlNUJpbnB1dHMuaW5wdXRfaWRzLnNoYXBlJTVCMSU1RCUyMCUzQSU1RCUyQyUyMHNraXBfc3BlY2lhbF90b2tlbnMlM0RUcnVlJTBBJTIwJTIwJTIwJTIwKSUwQSk=",highlighted:` | |
| <span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoModelForImageTextToText, AutoProcessor | |
| model_id = <span class="hljs-string">"CohereLabs/command-a-vision-07-2025"</span> | |
| processor = AutoProcessor.from_pretrained(model_id) | |
| model = AutoModelForImageTextToText.from_pretrained( | |
| model_id, device_map=<span class="hljs-string">"auto"</span> | |
| ) | |
| <span class="hljs-comment"># Format message with the Command-A-Vision chat template</span> | |
| messages = [ | |
| { | |
| <span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, | |
| <span class="hljs-string">"content"</span>: [ | |
| { | |
| <span class="hljs-string">"type"</span>: <span class="hljs-string">"image"</span>, | |
| <span class="hljs-string">"url"</span>: <span class="hljs-string">"https://images.pexels.com/photos/1108099/pexels-photo-1108099.jpeg"</span>, | |
| }, | |
| {<span class="hljs-string">"type"</span>: <span class="hljs-string">"text"</span>, <span class="hljs-string">"text"</span>: <span class="hljs-string">"what is in this image?"</span>}, | |
| ], | |
| }, | |
| ] | |
| inputs = processor.apply_chat_template( | |
| messages, | |
| padding=<span class="hljs-literal">True</span>, | |
| add_generation_prompt=<span class="hljs-literal">True</span>, | |
| tokenize=<span class="hljs-literal">True</span>, | |
| return_dict=<span class="hljs-literal">True</span>, | |
| return_tensors=<span class="hljs-string">"pt"</span>, | |
| ).to(model.device) | |
| gen_tokens = model.generate( | |
| **inputs, | |
| max_new_tokens=<span class="hljs-number">300</span>, | |
| do_sample=<span class="hljs-literal">True</span>, | |
| temperature=<span class="hljs-number">0.3</span>, | |
| ) | |
| <span class="hljs-built_in">print</span>( | |
| processor.tokenizer.decode( | |
| gen_tokens[<span class="hljs-number">0</span>][inputs.input_ids.shape[<span class="hljs-number">1</span>] :], skip_special_tokens=<span class="hljs-literal">True</span> | |
| ) | |
| )`,lang:"python",wrap:!1}}),{c(){g(o.$$.fragment)},l(s){f(o.$$.fragment,s)},m(s,c){_(o,s,c),u=!0},p:ee,i(s){u||(y(o.$$.fragment,s),u=!0)},o(s){M(o.$$.fragment,s),u=!1},d(s){T(o,s)}}}function kt(U){let o,u;return o=new Le({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMHBpcGVsaW5lJTBBJTBBJTBBcGlwZSUyMCUzRCUyMHBpcGVsaW5lKG1vZGVsJTNEJTIyQ29oZXJlTGFicyUyRmNvbW1hbmQtYS12aXNpb24tMDctMjAyNSUyMiUyQyUyMHRhc2slM0QlMjJpbWFnZS10ZXh0LXRvLXRleHQlMjIlMkMlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMiklMEElMEFtZXNzYWdlcyUyMCUzRCUyMCU1QiUwQSUyMCUyMCUyMCUyMCU3QiUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMnJvbGUlMjIlM0ElMjAlMjJ1c2VyJTIyJTJDJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIyY29udGVudCUyMiUzQSUyMCU1QiUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCU3QiUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMnR5cGUlMjIlM0ElMjAlMjJpbWFnZSUyMiUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMnVybCUyMiUzQSUyMCUyMmh0dHBzJTNBJTJGJTJGbWVkaWEuaXN0b2NrcGhvdG8uY29tJTJGaWQlMkY0NTgwMTIwNTclMkZwaG90byUyRmlzdGFuYnVsLXR1cmtleS5qcGclM0ZzJTNENjEyeDYxMiUyNnclM0QwJTI2ayUzRDIwJTI2YyUzRHFvZ0FPVnZrcGZVeXFMVU1yX1hKUXlxLUhrQUNYeVlVU1piS2hCbFByeG8lM0QlMjIlMkMlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlN0QlMkMlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlN0IlMjJ0eXBlJTIyJTNBJTIwJTIydGV4dCUyMiUyQyUyMCUyMnRleHQlMjIlM0ElMjAlMjJXaGVyZSUyMHdhcyUyMHRoaXMlMjB0YWtlbiUyMCUzRiUyMiU3RCUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCU1RCUyQyUwQSUyMCUyMCUyMCUyMCU3RCUyQyUwQSU1RCUwQSUwQW91dHB1dHMlMjAlM0QlMjBwaXBlKHRleHQlM0RtZXNzYWdlcyUyQyUyMG1heF9uZXdfdG9rZW5zJTNEMzAwJTJDJTIwcmV0dXJuX2Z1bGxfdGV4dCUzREZhbHNlKSUwQSUwQXByaW50KG91dHB1dHMp",highlighted:`<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> pipeline | |
| pipe = pipeline(model=<span class="hljs-string">"CohereLabs/command-a-vision-07-2025"</span>, task=<span class="hljs-string">"image-text-to-text"</span>, device_map=<span class="hljs-string">"auto"</span>) | |
| messages = [ | |
| { | |
| <span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, | |
| <span class="hljs-string">"content"</span>: [ | |
| { | |
| <span class="hljs-string">"type"</span>: <span class="hljs-string">"image"</span>, | |
| <span class="hljs-string">"url"</span>: <span class="hljs-string">"https://media.istockphoto.com/id/458012057/photo/istanbul-turkey.jpg?s=612x612&w=0&k=20&c=qogAOVvkpfUyqLUMr_XJQyq-HkACXyYUSZbKhBlPrxo="</span>, | |
| }, | |
| {<span class="hljs-string">"type"</span>: <span class="hljs-string">"text"</span>, <span class="hljs-string">"text"</span>: <span class="hljs-string">"Where was this taken ?"</span>}, | |
| ], | |
| }, | |
| ] | |
| outputs = pipe(text=messages, max_new_tokens=<span class="hljs-number">300</span>, return_full_text=<span class="hljs-literal">False</span>) | |
| <span class="hljs-built_in">print</span>(outputs)`,lang:"python",wrap:!1}}),{c(){g(o.$$.fragment)},l(s){f(o.$$.fragment,s)},m(s,c){_(o,s,c),u=!0},p:ee,i(s){u||(y(o.$$.fragment,s),u=!0)},o(s){M(o.$$.fragment,s),u=!1},d(s){T(o,s)}}}function Vt(U){let o,u,s,c;return o=new Tt({props:{id:"usage",option:"AutoModel",$$slots:{default:[$t]},$$scope:{ctx:U}}}),s=new Tt({props:{id:"usage",option:"Pipeline",$$slots:{default:[kt]},$$scope:{ctx:U}}}),{c(){g(o.$$.fragment),u=a(),g(s.$$.fragment)},l(d){f(o.$$.fragment,d),u=i(d),f(s.$$.fragment,d)},m(d,t){_(o,d,t),p(d,u,t),_(s,d,t),c=!0},p(d,t){const w={};t&2&&(w.$$scope={dirty:t,ctx:d}),o.$set(w);const N={};t&2&&(N.$$scope={dirty:t,ctx:d}),s.$set(N)},i(d){c||(y(o.$$.fragment,d),y(s.$$.fragment,d),c=!0)},o(d){M(o.$$.fragment,d),M(s.$$.fragment,d),c=!1},d(d){d&&r(u),T(o,d),T(s,d)}}}function xt(U){let o,u=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=m("p"),o.innerHTML=u},l(s){o=h(s,"P",{"data-svelte-h":!0}),C(o)!=="svelte-fincs2"&&(o.innerHTML=u)},m(s,c){p(s,o,c)},p:ee,d(s){s&&r(o)}}}function zt(U){let o,u="Example:",s,c,d;return c=new Le({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Qcm9jZXNzb3IlMkMlMjBDb2hlcmUyVmlzaW9uRm9yQ29uZGl0aW9uYWxHZW5lcmF0aW9uJTBBaW1wb3J0JTIwdG9yY2glMEElMEFwcm9jZXNzb3IlMjAlM0QlMjBBdXRvUHJvY2Vzc29yLmZyb21fcHJldHJhaW5lZCglMjJDb2hlcmVMYWJzJTJGY29tbWFuZC1hLXZpc2lvbi0wNy0yMDI1JTIyJTJDJTIwdXNlX2Zhc3QlM0RUcnVlKSUwQW1vZGVsJTIwJTNEJTIwQ29oZXJlMlZpc2lvbkZvckNvbmRpdGlvbmFsR2VuZXJhdGlvbi5mcm9tX3ByZXRyYWluZWQoJTIyQ29oZXJlTGFicyUyRmNvbW1hbmQtYS12aXNpb24tMDctMjAyNSUyMiUyQyUyMGRldmljZV9tYXAlM0QlMjJhdXRvJTIyKSUwQSUwQW1lc3NhZ2VzJTIwJTNEJTIwJTVCJTBBJTIwJTIwJTIwJTIwJTdCJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIycm9sZSUyMiUzQSUyMCUyMnVzZXIlMjIlMkMlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjJjb250ZW50JTIyJTNBJTIwJTVCJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTdCJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIydHlwZSUyMiUzQSUyMCUyMmltYWdlJTIyJTJDJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIydXJsJTIyJTNBJTIwJTIyaHR0cHMlM0ElMkYlMkZpbWFnZXMucGV4ZWxzLmNvbSUyRnBob3RvcyUyRjExMDgwOTklMkZwZXhlbHMtcGhvdG8tMTEwODA5OS5qcGVnJTIyJTJDJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTdEJTJDJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTdCJTIydHlwZSUyMiUzQSUyMCUyMnRleHQlMjIlMkMlMjAlMjJ0ZXh0JTIyJTNBJTIwJTIyd2hhdCUyMGlzJTIwaW4lMjB0aGlzJTIwaW1hZ2UlM0YlMjIlN0QlMkMlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlNUQlMkMlMEElMjAlMjAlMjAlMjAlN0QlMkMlMEElNUQlMEElMEFpbnB1dHMlMjAlM0QlMjBwcm9jZXNzb3IuYXBwbHlfY2hhdF90ZW1wbGF0ZSglMEElMjAlMjAlMjAlMjBtZXNzYWdlcyUyQyUyMHBhZGRpbmclM0RUcnVlJTJDJTIwYWRkX2dlbmVyYXRpb25fcHJvbXB0JTNEVHJ1ZSUyQyUyMHRva2VuaXplJTNEVHJ1ZSUyQyUyMHJldHVybl9kaWN0JTNEVHJ1ZSUyQyUyMHJldHVybl90ZW5zb3JzJTNEJTIycHQlMjIlMkMlMEEpLnRvKG1vZGVsLmRldmljZSklMEElMEFnZW5fdG9rZW5zJTIwJTNEJTIwbW9kZWwuZ2VuZXJhdGUoKippbnB1dHMlMkMlMjBtYXhfbmV3X3Rva2VucyUzRDMwMCUyQyUyMGRvX3NhbXBsZSUzRFRydWUlMkMlMjB0ZW1wZXJhdHVyZSUzRDAuMyklMEFwcm9jZXNzb3IudG9rZW5pemVyLmRlY29kZShnZW5fdG9rZW5zJTVCMCU1RCU1QmlucHV0cy5pbnB1dF9pZHMuc2hhcGUlNUIxJTVEJTNBJTVEJTJDJTIwc2tpcF9zcGVjaWFsX3Rva2VucyUzRFRydWUp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoProcessor, Cohere2VisionForConditionalGeneration | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span>processor = AutoProcessor.from_pretrained(<span class="hljs-string">"CohereLabs/command-a-vision-07-2025"</span>, use_fast=<span class="hljs-literal">True</span>) | |
| <span class="hljs-meta">>>> </span>model = Cohere2VisionForConditionalGeneration.from_pretrained(<span class="hljs-string">"CohereLabs/command-a-vision-07-2025"</span>, device_map=<span class="hljs-string">"auto"</span>) | |
| <span class="hljs-meta">>>> </span>messages = [ | |
| <span class="hljs-meta">... </span> { | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"content"</span>: [ | |
| <span class="hljs-meta">... </span> { | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"type"</span>: <span class="hljs-string">"image"</span>, | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"url"</span>: <span class="hljs-string">"https://images.pexels.com/photos/1108099/pexels-photo-1108099.jpeg"</span>, | |
| <span class="hljs-meta">... </span> }, | |
| <span class="hljs-meta">... </span> {<span class="hljs-string">"type"</span>: <span class="hljs-string">"text"</span>, <span class="hljs-string">"text"</span>: <span class="hljs-string">"what is in this image?"</span>}, | |
| <span class="hljs-meta">... </span> ], | |
| <span class="hljs-meta">... </span> }, | |
| <span class="hljs-meta">... </span>] | |
| <span class="hljs-meta">>>> </span>inputs = processor.apply_chat_template( | |
| <span class="hljs-meta">... </span> messages, padding=<span class="hljs-literal">True</span>, add_generation_prompt=<span class="hljs-literal">True</span>, tokenize=<span class="hljs-literal">True</span>, return_dict=<span class="hljs-literal">True</span>, return_tensors=<span class="hljs-string">"pt"</span>, | |
| <span class="hljs-meta">... </span>).to(model.device) | |
| <span class="hljs-meta">>>> </span>gen_tokens = model.generate(**inputs, max_new_tokens=<span class="hljs-number">300</span>, do_sample=<span class="hljs-literal">True</span>, temperature=<span class="hljs-number">0.3</span>) | |
| <span class="hljs-meta">>>> </span>processor.tokenizer.decode(gen_tokens[<span class="hljs-number">0</span>][inputs.input_ids.shape[<span class="hljs-number">1</span>]:], skip_special_tokens=<span class="hljs-literal">True</span>)`,lang:"python",wrap:!1}}),{c(){o=m("p"),o.textContent=u,s=a(),g(c.$$.fragment)},l(t){o=h(t,"P",{"data-svelte-h":!0}),C(o)!=="svelte-11lpom8"&&(o.textContent=u),s=i(t),f(c.$$.fragment,t)},m(t,w){p(t,o,w),p(t,s,w),_(c,t,w),d=!0},p:ee,i(t){d||(y(c.$$.fragment,t),d=!0)},o(t){M(c.$$.fragment,t),d=!1},d(t){t&&(r(o),r(s)),T(c,t)}}}function Ft(U){let o,u="Example:",s,c,d;return c=new Le({props:{code:"ZnJvbSUyMFBJTCUyMGltcG9ydCUyMEltYWdlJTBBZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Qcm9jZXNzb3IlMkMlMjBDb2hlcmUyVmlzaW9uRm9yQ29uZGl0aW9uYWxHZW5lcmF0aW9uJTBBJTBBbW9kZWwlMjAlM0QlMjBDb2hlcmUyVmlzaW9uRm9yQ29uZGl0aW9uYWxHZW5lcmF0aW9uLmZyb21fcHJldHJhaW5lZCglMjJDb2hlcmVMYWJzJTJGY29tbWFuZC1hLXZpc2lvbi0wNy0yMDI1JTIyKSUwQXByb2Nlc3NvciUyMCUzRCUyMEF1dG9Qcm9jZXNzb3IuZnJvbV9wcmV0cmFpbmVkKCUyMkNvaGVyZUxhYnMlMkZjb21tYW5kLWEtdmlzaW9uLTA3LTIwMjUlMjIpJTBBJTBBbWVzc2FnZXMlMjAlM0QlMjAlNUIlMEElMjAlMjAlMjAlMjAlN0IlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjJyb2xlJTIyJTNBJTIwJTIydXNlciUyMiUyQyUyMCUyMmNvbnRlbnQlMjIlM0ElMjAlNUIlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlN0IlMjJ0eXBlJTIyJTNBJTIwJTIyaW1hZ2UlMjIlMkMlMjAlMjJ1cmwlMjIlM0ElMjAlMjJodHRwcyUzQSUyRiUyRmh1Z2dpbmdmYWNlLmNvJTJGZGF0YXNldHMlMkZodWdnaW5nZmFjZSUyRmRvY3VtZW50YXRpb24taW1hZ2VzJTJGcmVzb2x2ZSUyRm1haW4lMkZwaXBlbGluZS1jYXQtY2hvbmsuanBlZyUyMiU3RCUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCU3QiUyMnR5cGUlMjIlM0ElMjAlMjJ0ZXh0JTIyJTJDJTIwJTIydGV4dCUyMiUzQSUyMCUyMldoZXJlJTIwaXMlMjB0aGUlMjBjYXQlMjBzdGFuZGluZyUzRiUyMiU3RCUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCU1RCUwQSUyMCUyMCUyMCUyMCU3RCUyQyUwQSU1RCUwQSUwQWlucHV0cyUyMCUzRCUyMHByb2Nlc3Nvci5hcHBseV9jaGF0X3RlbXBsYXRlKCUwQSUyMCUyMCUyMCUyMG1lc3NhZ2VzJTJDJTBBJTIwJTIwJTIwJTIwdG9rZW5pemUlM0RUcnVlJTJDJTBBJTIwJTIwJTIwJTIwcmV0dXJuX2RpY3QlM0RUcnVlJTJDJTBBJTIwJTIwJTIwJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiUyQyUwQSUyMCUyMCUyMCUyMGFkZF9nZW5lcmF0aW9uX3Byb21wdCUzRFRydWUlMEEpJTBBJTIzJTIwR2VuZXJhdGUlMEFnZW5lcmF0ZV9pZHMlMjAlM0QlMjBtb2RlbC5nZW5lcmF0ZSgqKmlucHV0cyklMEFwcm9jZXNzb3IuYmF0Y2hfZGVjb2RlKGdlbmVyYXRlX2lkcyUyQyUyMHNraXBfc3BlY2lhbF90b2tlbnMlM0RUcnVlKSU1QjAlNUQ=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoProcessor, Cohere2VisionForConditionalGeneration | |
| <span class="hljs-meta">>>> </span>model = Cohere2VisionForConditionalGeneration.from_pretrained(<span class="hljs-string">"CohereLabs/command-a-vision-07-2025"</span>) | |
| <span class="hljs-meta">>>> </span>processor = AutoProcessor.from_pretrained(<span class="hljs-string">"CohereLabs/command-a-vision-07-2025"</span>) | |
| <span class="hljs-meta">>>> </span>messages = [ | |
| <span class="hljs-meta">... </span> { | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, <span class="hljs-string">"content"</span>: [ | |
| <span class="hljs-meta">... </span> {<span class="hljs-string">"type"</span>: <span class="hljs-string">"image"</span>, <span class="hljs-string">"url"</span>: <span class="hljs-string">"https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/pipeline-cat-chonk.jpeg"</span>}, | |
| <span class="hljs-meta">... </span> {<span class="hljs-string">"type"</span>: <span class="hljs-string">"text"</span>, <span class="hljs-string">"text"</span>: <span class="hljs-string">"Where is the cat standing?"</span>}, | |
| <span class="hljs-meta">... </span> ] | |
| <span class="hljs-meta">... </span> }, | |
| <span class="hljs-meta">... </span>] | |
| <span class="hljs-meta">>>> </span>inputs = processor.apply_chat_template( | |
| <span class="hljs-meta">... </span> messages, | |
| <span class="hljs-meta">... </span> tokenize=<span class="hljs-literal">True</span>, | |
| <span class="hljs-meta">... </span> return_dict=<span class="hljs-literal">True</span>, | |
| <span class="hljs-meta">... </span> return_tensors=<span class="hljs-string">"pt"</span>, | |
| <span class="hljs-meta">... </span> add_generation_prompt=<span class="hljs-literal">True</span> | |
| <span class="hljs-meta">... </span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Generate</span> | |
| <span class="hljs-meta">>>> </span>generate_ids = model.generate(**inputs) | |
| <span class="hljs-meta">>>> </span>processor.batch_decode(generate_ids, skip_special_tokens=<span class="hljs-literal">True</span>)[<span class="hljs-number">0</span>]`,lang:"python",wrap:!1}}),{c(){o=m("p"),o.textContent=u,s=a(),g(c.$$.fragment)},l(t){o=h(t,"P",{"data-svelte-h":!0}),C(o)!=="svelte-11lpom8"&&(o.textContent=u),s=i(t),f(c.$$.fragment,t)},m(t,w){p(t,o,w),p(t,s,w),_(c,t,w),d=!0},p:ee,i(t){d||(y(c.$$.fragment,t),d=!0)},o(t){M(c.$$.fragment,t),d=!1},d(t){t&&(r(o),r(s)),T(c,t)}}}function Bt(U){let o,u=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){o=m("p"),o.innerHTML=u},l(s){o=h(s,"P",{"data-svelte-h":!0}),C(o)!=="svelte-fincs2"&&(o.innerHTML=u)},m(s,c){p(s,o,c)},p:ee,d(s){s&&r(o)}}}function qt(U){let o,u="Example:",s,c,d;return c=new Le({props:{code:"",highlighted:"",lang:"python",wrap:!1}}),{c(){o=m("p"),o.textContent=u,s=a(),g(c.$$.fragment)},l(t){o=h(t,"P",{"data-svelte-h":!0}),C(o)!=="svelte-11lpom8"&&(o.textContent=u),s=i(t),f(c.$$.fragment,t)},m(t,w){p(t,o,w),p(t,s,w),_(c,t,w),d=!0},p:ee,i(t){d||(y(c.$$.fragment,t),d=!0)},o(t){M(c.$$.fragment,t),d=!1},d(t){t&&(r(o),r(s)),T(c,t)}}}function Pt(U){let o,u="Example:",s,c,d;return c=new Le({props:{code:"",highlighted:"",lang:"python",wrap:!1}}),{c(){o=m("p"),o.textContent=u,s=a(),g(c.$$.fragment)},l(t){o=h(t,"P",{"data-svelte-h":!0}),C(o)!=="svelte-11lpom8"&&(o.textContent=u),s=i(t),f(c.$$.fragment,t)},m(t,w){p(t,o,w),p(t,s,w),_(c,t,w),d=!0},p:ee,i(t){d||(y(c.$$.fragment,t),d=!0)},o(t){M(c.$$.fragment,t),d=!1},d(t){t&&(r(o),r(s)),T(c,t)}}}function Zt(U){let o,u,s,c,d,t="<em>This model was contributed to Hugging Face Transformers on 2025-07-31.</em>",w,N,Xe,oe,Se,R,Xo='<img alt="FlashAttention" src="https://img.shields.io/badge/%E2%9A%A1%EF%B8%8E%20FlashAttention-eae0c8?style=flat"/> <img alt="SDPA" src="https://img.shields.io/badge/SDPA-DE3412?style=flat&logo=pytorch&logoColor=white"/> <img alt="Tensor parallelism" src="https://img.shields.io/badge/Tensor%20parallelism-06b6d4?style=flat&logoColor=white"/>',De,te,Ye,se,So='Command A Vision (<a href="https://cohere.com/blog/command-a-vision" rel="nofollow">blog post</a>) is a state-of-the-art multimodal model designed to seamlessly integrate visual and textual information for a wide range of applications. By combining advanced computer vision techniques with natural language processing capabilities, Command A Vision enables users to analyze, understand, and generate insights from both visual and textual data.',Oe,ne,Do="The model excels at tasks including image captioning, visual question answering, document understanding, and chart understanding. This makes it a versatile tool for AI practitioners. Its ability to process complex visual and textual inputs makes it useful in settings where text-only representations are imprecise or unavailable, like real-world image understanding and graphics-heavy document processing.",Ke,re,Yo="Command A Vision is built upon a robust architecture that leverages the latest advancements in VLMs. It’s highly performant and efficient, even when dealing with large-scale datasets. The model’s flexibility makes it suitable for a wide range of use cases, from content moderation and image search to medical imaging analysis and robotics.",eo,ae,oo,ie,Oo="The model and image processor can be loaded as follows:",to,Q,so,le,no,q,ce,yo,ve,Ko=`This is the configuration class to store the configuration of a Cohere2VisionModel. It is used to instantiate a Cohere2 Vision | |
| model according to the specified arguments, defining the model architecture. Instantiating a configuration with the | |
| defaults will yield a similar configuration to that of the <a href="https://huggingface.co/CohereLabs/command-a-vision-07-2025" rel="nofollow">CohereLabs/command-a-vision-07-2025</a>`,Mo,Je,et=`Configuration objects inherit from <a href="/docs/transformers/pr_43265/en/main_classes/configuration#transformers.PreTrainedConfig">PreTrainedConfig</a> and can be used to control the model outputs. Read the | |
| documentation from <a href="/docs/transformers/pr_43265/en/main_classes/configuration#transformers.PreTrainedConfig">PreTrainedConfig</a> for more information.`,ro,de,ao,b,pe,To,Ie,ot="The COHERE2_VISION model which consists of a vision backbone and a language model.",Co,je,tt=`This model inherits from <a href="/docs/transformers/pr_43265/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,wo,$e,st=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,Uo,j,me,bo,ke,nt='The <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionForConditionalGeneration">Cohere2VisionForConditionalGeneration</a> forward method, overrides the <code>__call__</code> special method.',vo,L,Jo,Ve,rt=`<li><p><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> is provided) — Language modeling loss (for next-token prediction).</p></li> <li><p><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, config.vocab_size)</code>) — Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).</p></li> <li><p><strong>past_key_values</strong> (<code>Cache</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — It is a <a href="/docs/transformers/pr_43265/en/internal/generation_utils#transformers.Cache">Cache</a> instance. For more details, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>.</p> <p>Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see | |
| <code>past_key_values</code> input) to speed up sequential decoding.</p></li> <li><p><strong>hidden_states</strong> (<code>tuple[torch.FloatTensor]</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>attentions</strong> (<code>tuple[torch.FloatTensor]</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p></li> <li><p><strong>image_hidden_states</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) — A <code>torch.FloatTensor</code> of size <code>(batch_size, num_images, sequence_length, hidden_size)</code>. | |
| image_hidden_states of the model produced by the vision encoder and after projecting the last hidden state.</p></li>`,Io,H,jo,G,he,$o,xe,at=`<li><p><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — Sequence of hidden-states at the output of the last layer of the model.</p></li> <li><p><strong>pooler_output</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, hidden_size)</code>) — Last layer hidden-state of the first token of the sequence (classification token) after further processing | |
| through the layers used for the auxiliary pretraining task. E.g. for BERT-family of models, this returns | |
| the classification token after processing through a linear layer and a tanh activation function. The linear | |
| layer weights are trained from the next sentence prediction (classification) objective during pretraining.</p></li> <li><p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p></li>`,ko,X,io,ue,lo,v,ge,Vo,ze,it="The Cohere2Vision model which consists of a vision backbone and a language model, without a language modeling head.",xo,Fe,lt=`This model inherits from <a href="/docs/transformers/pr_43265/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,zo,Be,ct=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,Fo,$,fe,Bo,qe,dt='The <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionModel">Cohere2VisionModel</a> forward method, overrides the <code>__call__</code> special method.',qo,S,Po,Pe,pt=`<li><p><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — Sequence of hidden-states at the output of the last layer of the model.</p> <p>If <code>past_key_values</code> is used only the last hidden-state of the sequences of shape <code>(batch_size, 1, hidden_size)</code> is output.</p></li> <li><p><strong>past_key_values</strong> (<code>Cache</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — It is a <a href="/docs/transformers/pr_43265/en/internal/generation_utils#transformers.Cache">Cache</a> instance. For more details, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>.</p> <p>Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see | |
| <code>past_key_values</code> input) to speed up sequential decoding.</p></li> <li><p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p></li> <li><p><strong>image_hidden_states</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) — A <code>torch.FloatTensor</code> of size <code>(batch_size, num_images, sequence_length, hidden_size)</code>. | |
| image_hidden_states of the model produced by the vision encoder and after projecting the last hidden state.</p></li>`,Zo,D,Ao,B,_e,No,Ze,mt="Obtains image last hidden states from the vision tower and apply multimodal projection.",Go,Ae,ht=`<li><p><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — Sequence of hidden-states at the output of the last layer of the model.</p></li> <li><p><strong>pooler_output</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, hidden_size)</code>) — Last layer hidden-state of the first token of the sequence (classification token) after further processing | |
| through the layers used for the auxiliary pretraining task. E.g. for BERT-family of models, this returns | |
| the classification token after processing through a linear layer and a tanh activation function. The linear | |
| layer weights are trained from the next sentence prediction (classification) objective during pretraining.</p></li> <li><p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p></li>`,Eo,Y,co,ye,po,P,Me,Wo,Ne,ut="Constructs a Cohere2VisionImageProcessor image processor.",Ro,Ge,Te,mo,Ce,ho,V,we,Qo,Ee,gt="Constructs a Cohere2VisionProcessor which wraps a image processor and a tokenizer into a single processor.",Lo,We,ft=`<a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionProcessor">Cohere2VisionProcessor</a> offers all the functionalities of <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a> and <code>tokenizer_class</code>. See the | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">~Cohere2VisionImageProcessor</a> and <code>~tokenizer_class</code> for more information.`,Ho,Re,Ue,uo,be,go,He,fo;return N=new Jt({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),oe=new K({props:{title:"Command A Vision",local:"command-a-vision",headingTag:"h1"}}),te=new K({props:{title:"Overview",local:"overview",headingTag:"h2"}}),ae=new K({props:{title:"Usage tips",local:"usage-tips",headingTag:"h2"}}),Q=new jt({props:{id:"usage",options:["AutoModel","Pipeline"],$$slots:{default:[Vt]},$$scope:{ctx:U}}}),le=new K({props:{title:"Cohere2VisionConfig",local:"transformers.Cohere2VisionConfig",headingTag:"h2"}}),ce=new A({props:{name:"class transformers.Cohere2VisionConfig",anchor:"transformers.Cohere2VisionConfig",parameters:[{name:"transformers_version",val:": str | None = None"},{name:"architectures",val:": list[str] | None = None"},{name:"output_hidden_states",val:": bool | None = False"},{name:"return_dict",val:": bool | None = True"},{name:"dtype",val:": typing.Union[str, ForwardRef('torch.dtype'), NoneType] = None"},{name:"chunk_size_feed_forward",val:": int = 0"},{name:"is_encoder_decoder",val:": bool = False"},{name:"id2label",val:": dict[int, str] | dict[str, str] | None = None"},{name:"label2id",val:": dict[str, int] | dict[str, str] | None = None"},{name:"problem_type",val:": typing.Optional[typing.Literal['regression', 'single_label_classification', 'multi_label_classification']] = None"},{name:"vision_config",val:": dict | transformers.configuration_utils.PreTrainedConfig | None = None"},{name:"text_config",val:": dict | transformers.configuration_utils.PreTrainedConfig | None = None"},{name:"downsample_factor",val:": int = 2"},{name:"image_token_id",val:": int = 255036"},{name:"alignment_intermediate_size",val:": int = 36864"},{name:"tie_word_embeddings",val:": bool = True"}],parametersDescription:[{anchor:"transformers.Cohere2VisionConfig.vision_config",description:`<strong>vision_config</strong> (<code>Union[dict, ~configuration_utils.PreTrainedConfig]</code>, <em>optional</em>) — | |
| The config object or dictionary of the vision backbone.`,name:"vision_config"},{anchor:"transformers.Cohere2VisionConfig.text_config",description:`<strong>text_config</strong> (<code>Union[dict, ~configuration_utils.PreTrainedConfig]</code>, <em>optional</em>) — | |
| The config object or dictionary of the text backbone.`,name:"text_config"},{anchor:"transformers.Cohere2VisionConfig.downsample_factor",description:`<strong>downsample_factor</strong> (<code>int</code>, <em>optional</em>, defaults to 2) — | |
| The factor by which to downsample the input image.`,name:"downsample_factor"},{anchor:"transformers.Cohere2VisionConfig.image_token_id",description:`<strong>image_token_id</strong> (<code>int</code>, <em>optional</em>, defaults to <code>255036</code>) — | |
| The image token index used as a placeholder for input images.`,name:"image_token_id"},{anchor:"transformers.Cohere2VisionConfig.alignment_intermediate_size",description:`<strong>alignment_intermediate_size</strong> (<code>int</code>, <em>optional</em>, defaults to 36864) — | |
| The size of the intermediate layer for alignment.`,name:"alignment_intermediate_size"},{anchor:"transformers.Cohere2VisionConfig.tie_word_embeddings",description:`<strong>tie_word_embeddings</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to tie weight embeddings according to model’s <code>tied_weights_keys</code> mapping.`,name:"tie_word_embeddings"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/configuration_cohere2_vision.py#L25"}}),de=new K({props:{title:"Cohere2VisionForConditionalGeneration",local:"transformers.Cohere2VisionForConditionalGeneration",headingTag:"h2"}}),pe=new A({props:{name:"class transformers.Cohere2VisionForConditionalGeneration",anchor:"transformers.Cohere2VisionForConditionalGeneration",parameters:[{name:"config",val:": Cohere2VisionConfig"}],parametersDescription:[{anchor:"transformers.Cohere2VisionForConditionalGeneration.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionConfig">Cohere2VisionConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_43265/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/modeling_cohere2_vision.py#L242"}}),me=new A({props:{name:"forward",anchor:"transformers.Cohere2VisionForConditionalGeneration.forward",parameters:[{name:"input_ids",val:": torch.LongTensor | None = None"},{name:"pixel_values",val:": torch.FloatTensor | None = None"},{name:"attention_mask",val:": torch.Tensor | None = None"},{name:"position_ids",val:": torch.LongTensor | None = None"},{name:"past_key_values",val:": transformers.cache_utils.Cache | None = None"},{name:"inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"labels",val:": torch.LongTensor | None = None"},{name:"use_cache",val:": bool | None = None"},{name:"logits_to_keep",val:": int | torch.Tensor = 0"},{name:"image_sizes",val:": torch.Tensor | None = None"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.utils.generic.TransformersKwargs]"}],parametersDescription:[{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_43265/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_43265/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_43265/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, image_size, image_size)</code>, <em>optional</em>) — | |
| The tensors corresponding to the input images. Pixel values can be obtained using | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a>. See <code>Cohere2VisionImageProcessor.__call__()</code> for details (<a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionProcessor">Cohere2VisionProcessor</a> uses | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a> for processing images).`,name:"pixel_values"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"attention_mask"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>~cache_utils.Cache</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Only <a href="/docs/transformers/pr_43265/en/internal/generation_utils#transformers.Cache">Cache</a> instance is allowed as input, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>. | |
| If no <code>past_key_values</code> are passed, <a href="/docs/transformers/pr_43265/en/internal/generation_utils#transformers.DynamicCache">DynamicCache</a> will be initialized by default.</p> | |
| <p>The model will output the same cache format that is fed as input.</p> | |
| <p>If <code>past_key_values</code> are used, the user is expected to input only unprocessed <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, unprocessed_length)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.labels",description:`<strong>labels</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Labels for computing the masked language modeling loss. Indices should either be in <code>[0, ..., config.vocab_size]</code> or -100 (see <code>input_ids</code> docstring). Tokens with indices set to <code>-100</code> are ignored | |
| (masked), the loss is only computed for the tokens with labels in <code>[0, ..., config.vocab_size]</code>.`,name:"labels"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.logits_to_keep",description:`<strong>logits_to_keep</strong> (<code>Union[int, torch.Tensor]</code>, <em>optional</em>, defaults to <code>0</code>) — | |
| If an <code>int</code>, compute logits for the last <code>logits_to_keep</code> tokens. If <code>0</code>, calculate logits for all | |
| <code>input_ids</code> (special case). Only last token logits are needed for generation, and calculating them only for that | |
| token can save memory, which becomes pretty significant for long sequences or large vocabulary size. | |
| If a <code>torch.Tensor</code>, must be 1D corresponding to the indices to keep in the sequence length dimension. | |
| This is useful when using packed tensor format (single dimension for batch and sequence length).`,name:"logits_to_keep"},{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.image_sizes",description:`<strong>image_sizes</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, 2)</code>, <em>optional</em>) — | |
| The sizes of the images in the batch, being (height, width) for each image.`,name:"image_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/modeling_cohere2_vision.py#L260",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>Cohere2VisionCausalLMOutputWithPast</code> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionConfig" | |
| >Cohere2VisionConfig</a>) and inputs.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>Cohere2VisionCausalLMOutputWithPast</code> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),L=new Mt({props:{$$slots:{default:[xt]},$$scope:{ctx:U}}}),H=new _o({props:{anchor:"transformers.Cohere2VisionForConditionalGeneration.forward.example",$$slots:{default:[zt]},$$scope:{ctx:U}}}),he=new A({props:{name:"get_image_features",anchor:"transformers.Cohere2VisionForConditionalGeneration.get_image_features",parameters:[{name:"pixel_values",val:": FloatTensor"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.utils.generic.TransformersKwargs]"}],parametersDescription:[{anchor:"transformers.Cohere2VisionForConditionalGeneration.get_image_features.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, image_size, image_size)</code>) — | |
| The tensors corresponding to the input images. Pixel values can be obtained using | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a>. See <code>Cohere2VisionImageProcessor.__call__()</code> for details (<a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionProcessor">Cohere2VisionProcessor</a> uses | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a> for processing images).`,name:"pixel_values"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/modeling_cohere2_vision.py#L254",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_43265/en/main_classes/output#transformers.modeling_outputs.BaseModelOutputWithPooling" | |
| >BaseModelOutputWithPooling</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionConfig" | |
| >Cohere2VisionConfig</a>) and inputs.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_43265/en/main_classes/output#transformers.modeling_outputs.BaseModelOutputWithPooling" | |
| >BaseModelOutputWithPooling</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),X=new _o({props:{anchor:"transformers.Cohere2VisionForConditionalGeneration.get_image_features.example",$$slots:{default:[Ft]},$$scope:{ctx:U}}}),ue=new K({props:{title:"Cohere2VisionModel",local:"transformers.Cohere2VisionModel",headingTag:"h2"}}),ge=new A({props:{name:"class transformers.Cohere2VisionModel",anchor:"transformers.Cohere2VisionModel",parameters:[{name:"config",val:": Cohere2VisionConfig"}],parametersDescription:[{anchor:"transformers.Cohere2VisionModel.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionConfig">Cohere2VisionConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_43265/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/modeling_cohere2_vision.py#L146"}}),fe=new A({props:{name:"forward",anchor:"transformers.Cohere2VisionModel.forward",parameters:[{name:"input_ids",val:": torch.LongTensor | None = None"},{name:"pixel_values",val:": torch.FloatTensor | None = None"},{name:"attention_mask",val:": torch.Tensor | None = None"},{name:"position_ids",val:": torch.LongTensor | None = None"},{name:"past_key_values",val:": transformers.cache_utils.Cache | None = None"},{name:"inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"use_cache",val:": bool | None = None"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.modeling_flash_attention_utils.FlashAttentionKwargs]"}],parametersDescription:[{anchor:"transformers.Cohere2VisionModel.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_43265/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_43265/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_43265/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.Cohere2VisionModel.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, image_size, image_size)</code>, <em>optional</em>) — | |
| The tensors corresponding to the input images. Pixel values can be obtained using | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a>. See <code>Cohere2VisionImageProcessor.__call__()</code> for details (<a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionProcessor">Cohere2VisionProcessor</a> uses | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a> for processing images).`,name:"pixel_values"},{anchor:"transformers.Cohere2VisionModel.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"attention_mask"},{anchor:"transformers.Cohere2VisionModel.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>.</p> | |
| <p><a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.Cohere2VisionModel.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>~cache_utils.Cache</code>, <em>optional</em>) — | |
| Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used to speed up sequential decoding. This typically consists in the <code>past_key_values</code> | |
| returned by the model at a previous stage of decoding, when <code>use_cache=True</code> or <code>config.use_cache=True</code>.</p> | |
| <p>Only <a href="/docs/transformers/pr_43265/en/internal/generation_utils#transformers.Cache">Cache</a> instance is allowed as input, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>. | |
| If no <code>past_key_values</code> are passed, <a href="/docs/transformers/pr_43265/en/internal/generation_utils#transformers.DynamicCache">DynamicCache</a> will be initialized by default.</p> | |
| <p>The model will output the same cache format that is fed as input.</p> | |
| <p>If <code>past_key_values</code> are used, the user is expected to input only unprocessed <code>input_ids</code> (those that don’t | |
| have their past key value states given to this model) of shape <code>(batch_size, unprocessed_length)</code> instead of all <code>input_ids</code> | |
| of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.Cohere2VisionModel.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.Cohere2VisionModel.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/modeling_cohere2_vision.py#L192",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>Cohere2VisionModelOutputWithPast</code> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionConfig" | |
| >Cohere2VisionConfig</a>) and inputs.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>Cohere2VisionModelOutputWithPast</code> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),S=new Mt({props:{$$slots:{default:[Bt]},$$scope:{ctx:U}}}),D=new _o({props:{anchor:"transformers.Cohere2VisionModel.forward.example",$$slots:{default:[qt]},$$scope:{ctx:U}}}),_e=new A({props:{name:"get_image_features",anchor:"transformers.Cohere2VisionModel.get_image_features",parameters:[{name:"pixel_values",val:": FloatTensor"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.utils.generic.TransformersKwargs]"}],parametersDescription:[{anchor:"transformers.Cohere2VisionModel.get_image_features.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, image_size, image_size)</code>) — | |
| The tensors corresponding to the input images. Pixel values can be obtained using | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a>. See <code>Cohere2VisionImageProcessor.__call__()</code> for details (<a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionProcessor">Cohere2VisionProcessor</a> uses | |
| <a href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionImageProcessor">Cohere2VisionImageProcessor</a> for processing images).`,name:"pixel_values"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/modeling_cohere2_vision.py#L155",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_43265/en/main_classes/output#transformers.modeling_outputs.BaseModelOutputWithPooling" | |
| >BaseModelOutputWithPooling</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_43265/en/model_doc/cohere2_vision#transformers.Cohere2VisionConfig" | |
| >Cohere2VisionConfig</a>) and inputs.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_43265/en/main_classes/output#transformers.modeling_outputs.BaseModelOutputWithPooling" | |
| >BaseModelOutputWithPooling</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),Y=new _o({props:{anchor:"transformers.Cohere2VisionModel.get_image_features.example",$$slots:{default:[Pt]},$$scope:{ctx:U}}}),ye=new K({props:{title:"Cohere2VisionImageProcessor",local:"transformers.Cohere2VisionImageProcessor",headingTag:"h2"}}),Me=new A({props:{name:"class transformers.Cohere2VisionImageProcessor",anchor:"transformers.Cohere2VisionImageProcessor",parameters:[{name:"**kwargs",val:": typing_extensions.Unpack[transformers.models.cohere2_vision.image_processing_cohere2_vision.Cohere2VisionImageProcessorKwargs]"}],parametersDescription:[{anchor:"transformers.Cohere2VisionImageProcessor.crop_to_patches",description:`<strong>crop_to_patches</strong> (<code>bool</code>, <em>kwargs</em>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to crop the image to patches. Can be overridden by the <code>crop_to_patches</code> parameter in the | |
| <code>preprocess</code> method.`,name:"crop_to_patches"},{anchor:"transformers.Cohere2VisionImageProcessor.min_patches",description:`<strong>min_patches</strong> (<code>int</code>, <em>kwargs</em>, <em>optional</em>, defaults to 1) — | |
| The minimum number of patches to be extracted from the image. Only has an effect if <code>crop_to_patches</code> is | |
| set to <code>True</code>. Can be overridden by the <code>min_patches</code> parameter in the <code>preprocess</code> method.`,name:"min_patches"},{anchor:"transformers.Cohere2VisionImageProcessor.max_patches",description:`<strong>max_patches</strong> (<code>int</code>, <em>kwargs</em>, <em>optional</em>, defaults to 12) — | |
| The maximum number of patches to be extracted from the image. Only has an effect if <code>crop_to_patches</code> is | |
| set to <code>True</code>. Can be overridden by the <code>max_patches</code> parameter in the <code>preprocess</code> method.`,name:"max_patches"},{anchor:"transformers.Cohere2VisionImageProcessor.*kwargs",description:`*<strong>*kwargs</strong> (<a href="/docs/transformers/pr_43265/en/main_classes/processors#transformers.ImagesKwargs">ImagesKwargs</a>, <em>optional</em>) — | |
| Additional image preprocessing options. Model-specific kwargs are listed above; see the TypedDict class | |
| for the complete list of supported arguments.`,name:"*kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/image_processing_cohere2_vision.py#L110"}}),Te=new A({props:{name:"preprocess",anchor:"transformers.Cohere2VisionImageProcessor.preprocess",parameters:[{name:"images",val:": typing.Union[ForwardRef('PIL.Image.Image'), numpy.ndarray, ForwardRef('torch.Tensor'), list['PIL.Image.Image'], list[numpy.ndarray], list['torch.Tensor']]"},{name:"*args",val:""},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.processing_utils.ImagesKwargs]"}],parametersDescription:[{anchor:"transformers.Cohere2VisionImageProcessor.preprocess.images",description:`<strong>images</strong> (<code>Union[PIL.Image.Image, numpy.ndarray, torch.Tensor, list[PIL.Image.Image], list[numpy.ndarray], list[torch.Tensor]]</code>) — | |
| Image to preprocess. Expects a single or batch of images with pixel values ranging from 0 to 255. If | |
| passing in images with pixel values between 0 and 1, set <code>do_rescale=False</code>.`,name:"images"},{anchor:"transformers.Cohere2VisionImageProcessor.preprocess.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_43265/en/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| Returns stacked tensors if set to <code>'pt'</code>, otherwise returns a list of tensors.`,name:"return_tensors"},{anchor:"transformers.Cohere2VisionImageProcessor.preprocess.*kwargs",description:`*<strong>*kwargs</strong> (<a href="/docs/transformers/pr_43265/en/main_classes/processors#transformers.ImagesKwargs">ImagesKwargs</a>, <em>optional</em>) — | |
| Additional image preprocessing options. Model-specific kwargs are listed above; see the TypedDict class | |
| for the complete list of supported arguments.`,name:"*kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/image_processing_utils.py#L382",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <ul> | |
| <li><strong>data</strong> (<code>dict</code>) — Dictionary of lists/arrays/tensors returned by the <strong>call</strong> method (‘pixel_values’, etc.).</li> | |
| <li><strong>tensor_type</strong> (<code>Union[None, str, TensorType]</code>, <em>optional</em>) — You can give a tensor_type here to convert the lists of integers in PyTorch/Numpy Tensors at | |
| initialization.</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>~image_processing_base.BatchFeature</code></p> | |
| `}}),Ce=new K({props:{title:"Cohere2VisionProcessor",local:"transformers.Cohere2VisionProcessor",headingTag:"h2"}}),we=new A({props:{name:"class transformers.Cohere2VisionProcessor",anchor:"transformers.Cohere2VisionProcessor",parameters:[{name:"image_processor",val:" = None"},{name:"tokenizer",val:" = None"},{name:"chat_template",val:" = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.Cohere2VisionProcessor.image_processor",description:`<strong>image_processor</strong> (<code>Cohere2VisionImageProcessor</code>) — | |
| The image processor is a required input.`,name:"image_processor"},{anchor:"transformers.Cohere2VisionProcessor.tokenizer",description:`<strong>tokenizer</strong> (<code>tokenizer_class</code>) — | |
| The tokenizer is a required input.`,name:"tokenizer"},{anchor:"transformers.Cohere2VisionProcessor.chat_template",description:`<strong>chat_template</strong> (<code>str</code>) — | |
| A Jinja template to convert lists of messages in a chat into a tokenizable string.`,name:"chat_template"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/processing_cohere2_vision.py#L34"}}),Ue=new A({props:{name:"__call__",anchor:"transformers.Cohere2VisionProcessor.__call__",parameters:[{name:"images",val:": typing.Union[ForwardRef('PIL.Image.Image'), numpy.ndarray, ForwardRef('torch.Tensor'), list['PIL.Image.Image'], list[numpy.ndarray], list['torch.Tensor'], NoneType] = None"},{name:"text",val:": str | list[str] | list[list[str]] | None = None"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.models.cohere2_vision.processing_cohere2_vision.Cohere2VisionProcessorKwargs]"}],parametersDescription:[{anchor:"transformers.Cohere2VisionProcessor.__call__.images",description:`<strong>images</strong> (<code>Union[PIL.Image.Image, numpy.ndarray, torch.Tensor, list[PIL.Image.Image], list[numpy.ndarray], list[torch.Tensor]]</code>, <em>optional</em>) — | |
| Image to preprocess. Expects a single or batch of images with pixel values ranging from 0 to 255. If | |
| passing in images with pixel values between 0 and 1, set <code>do_rescale=False</code>.`,name:"images"},{anchor:"transformers.Cohere2VisionProcessor.__call__.text",description:`<strong>text</strong> (<code>Union[str, list[str], list[list[str]]]</code>, <em>optional</em>) — | |
| The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings | |
| (pretokenized string). If you pass a pretokenized input, set <code>is_split_into_words=True</code> to avoid ambiguity with batched inputs.`,name:"text"},{anchor:"transformers.Cohere2VisionProcessor.__call__.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <a href="/docs/transformers/pr_43265/en/internal/file_utils#transformers.TensorType">TensorType</a>, <em>optional</em>) — | |
| If set, will return tensors of a particular framework. Acceptable values are:</p> | |
| <ul> | |
| <li><code>'pt'</code>: Return PyTorch <code>torch.Tensor</code> objects.</li> | |
| <li><code>'np'</code>: Return NumPy <code>np.ndarray</code> objects.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.Cohere2VisionProcessor.__call__.*kwargs",description:`*<strong>*kwargs</strong> (<a href="/docs/transformers/pr_43265/en/main_classes/processors#transformers.ProcessingKwargs">ProcessingKwargs</a>, <em>optional</em>) — | |
| Additional processing options for each modality (text, images, videos, audio). Model-specific parameters | |
| are listed above; see the TypedDict class for the complete list of supported arguments.`,name:"*kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_43265/src/transformers/models/cohere2_vision/processing_cohere2_vision.py#L62",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_43265/en/main_classes/feature_extractor#transformers.BatchFeature" | |
| >BatchFeature</a> with the following fields:</p> | |
| <ul> | |
| <li><strong>input_ids</strong> — List of token ids to be fed to a model. Returned when <code>text</code> is not <code>None</code>.</li> | |
| <li><strong>attention_mask</strong> — List of indices specifying which tokens should be attended to by the model (when | |
| <code>return_attention_mask=True</code> or if <em>“attention_mask”</em> is in <code>self.model_input_names</code> and if <code>text</code> is not | |
| <code>None</code>).</li> | |
| <li><strong>pixel_values</strong> — Pixel values to be fed to a model. Returned when <code>images</code> is not <code>None</code>.</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_43265/en/main_classes/feature_extractor#transformers.BatchFeature" | |
| >BatchFeature</a></p> | |
| `}}),be=new It({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/cohere2_vision.md"}}),{c(){o=m("meta"),u=a(),s=m("p"),c=a(),d=m("p"),d.innerHTML=t,w=a(),g(N.$$.fragment),Xe=a(),g(oe.$$.fragment),Se=a(),R=m("div"),R.innerHTML=Xo,De=a(),g(te.$$.fragment),Ye=a(),se=m("p"),se.innerHTML=So,Oe=a(),ne=m("p"),ne.textContent=Do,Ke=a(),re=m("p"),re.textContent=Yo,eo=a(),g(ae.$$.fragment),oo=a(),ie=m("p"),ie.textContent=Oo,to=a(),g(Q.$$.fragment),so=a(),g(le.$$.fragment),no=a(),q=m("div"),g(ce.$$.fragment),yo=a(),ve=m("p"),ve.innerHTML=Ko,Mo=a(),Je=m("p"),Je.innerHTML=et,ro=a(),g(de.$$.fragment),ao=a(),b=m("div"),g(pe.$$.fragment),To=a(),Ie=m("p"),Ie.textContent=ot,Co=a(),je=m("p"),je.innerHTML=tt,wo=a(),$e=m("p"),$e.innerHTML=st,Uo=a(),j=m("div"),g(me.$$.fragment),bo=a(),ke=m("p"),ke.innerHTML=nt,vo=a(),g(L.$$.fragment),Jo=a(),Ve=m("ul"),Ve.innerHTML=rt,Io=a(),g(H.$$.fragment),jo=a(),G=m("div"),g(he.$$.fragment),$o=a(),xe=m("ul"),xe.innerHTML=at,ko=a(),g(X.$$.fragment),io=a(),g(ue.$$.fragment),lo=a(),v=m("div"),g(ge.$$.fragment),Vo=a(),ze=m("p"),ze.textContent=it,xo=a(),Fe=m("p"),Fe.innerHTML=lt,zo=a(),Be=m("p"),Be.innerHTML=ct,Fo=a(),$=m("div"),g(fe.$$.fragment),Bo=a(),qe=m("p"),qe.innerHTML=dt,qo=a(),g(S.$$.fragment),Po=a(),Pe=m("ul"),Pe.innerHTML=pt,Zo=a(),g(D.$$.fragment),Ao=a(),B=m("div"),g(_e.$$.fragment),No=a(),Ze=m("p"),Ze.textContent=mt,Go=a(),Ae=m("ul"),Ae.innerHTML=ht,Eo=a(),g(Y.$$.fragment),co=a(),g(ye.$$.fragment),po=a(),P=m("div"),g(Me.$$.fragment),Wo=a(),Ne=m("p"),Ne.textContent=ut,Ro=a(),Ge=m("div"),g(Te.$$.fragment),mo=a(),g(Ce.$$.fragment),ho=a(),V=m("div"),g(we.$$.fragment),Qo=a(),Ee=m("p"),Ee.textContent=gt,Lo=a(),We=m("p"),We.innerHTML=ft,Ho=a(),Re=m("div"),g(Ue.$$.fragment),uo=a(),g(be.$$.fragment),go=a(),He=m("p"),this.h()},l(e){const n=vt("svelte-u9bgzb",document.head);o=h(n,"META",{name:!0,content:!0}),n.forEach(r),u=i(e),s=h(e,"P",{}),F(s).forEach(r),c=i(e),d=h(e,"P",{"data-svelte-h":!0}),C(d)!=="svelte-dcyqwm"&&(d.innerHTML=t),w=i(e),f(N.$$.fragment,e),Xe=i(e),f(oe.$$.fragment,e),Se=i(e),R=h(e,"DIV",{class:!0,"data-svelte-h":!0}),C(R)!=="svelte-1iyo9lh"&&(R.innerHTML=Xo),De=i(e),f(te.$$.fragment,e),Ye=i(e),se=h(e,"P",{"data-svelte-h":!0}),C(se)!=="svelte-mp5n58"&&(se.innerHTML=So),Oe=i(e),ne=h(e,"P",{"data-svelte-h":!0}),C(ne)!=="svelte-18ri1fu"&&(ne.textContent=Do),Ke=i(e),re=h(e,"P",{"data-svelte-h":!0}),C(re)!=="svelte-2q66tf"&&(re.textContent=Yo),eo=i(e),f(ae.$$.fragment,e),oo=i(e),ie=h(e,"P",{"data-svelte-h":!0}),C(ie)!=="svelte-1a3wb26"&&(ie.textContent=Oo),to=i(e),f(Q.$$.fragment,e),so=i(e),f(le.$$.fragment,e),no=i(e),q=h(e,"DIV",{class:!0});var E=F(q);f(ce.$$.fragment,E),yo=i(E),ve=h(E,"P",{"data-svelte-h":!0}),C(ve)!=="svelte-dk9sv8"&&(ve.innerHTML=Ko),Mo=i(E),Je=h(E,"P",{"data-svelte-h":!0}),C(Je)!=="svelte-1e8815j"&&(Je.innerHTML=et),E.forEach(r),ro=i(e),f(de.$$.fragment,e),ao=i(e),b=h(e,"DIV",{class:!0});var J=F(b);f(pe.$$.fragment,J),To=i(J),Ie=h(J,"P",{"data-svelte-h":!0}),C(Ie)!=="svelte-hiv3do"&&(Ie.textContent=ot),Co=i(J),je=h(J,"P",{"data-svelte-h":!0}),C(je)!=="svelte-1gb3c10"&&(je.innerHTML=tt),wo=i(J),$e=h(J,"P",{"data-svelte-h":!0}),C($e)!=="svelte-hswkmf"&&($e.innerHTML=st),Uo=i(J),j=h(J,"DIV",{class:!0});var x=F(j);f(me.$$.fragment,x),bo=i(x),ke=h(x,"P",{"data-svelte-h":!0}),C(ke)!=="svelte-8p4026"&&(ke.innerHTML=nt),vo=i(x),f(L.$$.fragment,x),Jo=i(x),Ve=h(x,"UL",{"data-svelte-h":!0}),C(Ve)!=="svelte-1iewcof"&&(Ve.innerHTML=rt),Io=i(x),f(H.$$.fragment,x),x.forEach(r),jo=i(J),G=h(J,"DIV",{class:!0});var W=F(G);f(he.$$.fragment,W),$o=i(W),xe=h(W,"UL",{"data-svelte-h":!0}),C(xe)!=="svelte-1vt4ztn"&&(xe.innerHTML=at),ko=i(W),f(X.$$.fragment,W),W.forEach(r),J.forEach(r),io=i(e),f(ue.$$.fragment,e),lo=i(e),v=h(e,"DIV",{class:!0});var I=F(v);f(ge.$$.fragment,I),Vo=i(I),ze=h(I,"P",{"data-svelte-h":!0}),C(ze)!=="svelte-1kgnlur"&&(ze.textContent=it),xo=i(I),Fe=h(I,"P",{"data-svelte-h":!0}),C(Fe)!=="svelte-1gb3c10"&&(Fe.innerHTML=lt),zo=i(I),Be=h(I,"P",{"data-svelte-h":!0}),C(Be)!=="svelte-hswkmf"&&(Be.innerHTML=ct),Fo=i(I),$=h(I,"DIV",{class:!0});var z=F($);f(fe.$$.fragment,z),Bo=i(z),qe=h(z,"P",{"data-svelte-h":!0}),C(qe)!=="svelte-k6k5h8"&&(qe.innerHTML=dt),qo=i(z),f(S.$$.fragment,z),Po=i(z),Pe=h(z,"UL",{"data-svelte-h":!0}),C(Pe)!=="svelte-1boxntb"&&(Pe.innerHTML=pt),Zo=i(z),f(D.$$.fragment,z),z.forEach(r),Ao=i(I),B=h(I,"DIV",{class:!0});var Z=F(B);f(_e.$$.fragment,Z),No=i(Z),Ze=h(Z,"P",{"data-svelte-h":!0}),C(Ze)!=="svelte-1vzo9k5"&&(Ze.textContent=mt),Go=i(Z),Ae=h(Z,"UL",{"data-svelte-h":!0}),C(Ae)!=="svelte-1vt4ztn"&&(Ae.innerHTML=ht),Eo=i(Z),f(Y.$$.fragment,Z),Z.forEach(r),I.forEach(r),co=i(e),f(ye.$$.fragment,e),po=i(e),P=h(e,"DIV",{class:!0});var Qe=F(P);f(Me.$$.fragment,Qe),Wo=i(Qe),Ne=h(Qe,"P",{"data-svelte-h":!0}),C(Ne)!=="svelte-13trhbl"&&(Ne.textContent=ut),Ro=i(Qe),Ge=h(Qe,"DIV",{class:!0});var _t=F(Ge);f(Te.$$.fragment,_t),_t.forEach(r),Qe.forEach(r),mo=i(e),f(Ce.$$.fragment,e),ho=i(e),V=h(e,"DIV",{class:!0});var O=F(V);f(we.$$.fragment,O),Qo=i(O),Ee=h(O,"P",{"data-svelte-h":!0}),C(Ee)!=="svelte-p24co5"&&(Ee.textContent=gt),Lo=i(O),We=h(O,"P",{"data-svelte-h":!0}),C(We)!=="svelte-127hg95"&&(We.innerHTML=ft),Ho=i(O),Re=h(O,"DIV",{class:!0});var yt=F(Re);f(Ue.$$.fragment,yt),yt.forEach(r),O.forEach(r),uo=i(e),f(be.$$.fragment,e),go=i(e),He=h(e,"P",{}),F(He).forEach(r),this.h()},h(){k(o,"name","hf:doc:metadata"),k(o,"content",At),k(R,"class","flex flex-wrap space-x-1"),k(q,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(j,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(G,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(b,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k($,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(B,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(v,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(Ge,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(P,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(Re,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),k(V,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(e,n){l(document.head,o),p(e,u,n),p(e,s,n),p(e,c,n),p(e,d,n),p(e,w,n),_(N,e,n),p(e,Xe,n),_(oe,e,n),p(e,Se,n),p(e,R,n),p(e,De,n),_(te,e,n),p(e,Ye,n),p(e,se,n),p(e,Oe,n),p(e,ne,n),p(e,Ke,n),p(e,re,n),p(e,eo,n),_(ae,e,n),p(e,oo,n),p(e,ie,n),p(e,to,n),_(Q,e,n),p(e,so,n),_(le,e,n),p(e,no,n),p(e,q,n),_(ce,q,null),l(q,yo),l(q,ve),l(q,Mo),l(q,Je),p(e,ro,n),_(de,e,n),p(e,ao,n),p(e,b,n),_(pe,b,null),l(b,To),l(b,Ie),l(b,Co),l(b,je),l(b,wo),l(b,$e),l(b,Uo),l(b,j),_(me,j,null),l(j,bo),l(j,ke),l(j,vo),_(L,j,null),l(j,Jo),l(j,Ve),l(j,Io),_(H,j,null),l(b,jo),l(b,G),_(he,G,null),l(G,$o),l(G,xe),l(G,ko),_(X,G,null),p(e,io,n),_(ue,e,n),p(e,lo,n),p(e,v,n),_(ge,v,null),l(v,Vo),l(v,ze),l(v,xo),l(v,Fe),l(v,zo),l(v,Be),l(v,Fo),l(v,$),_(fe,$,null),l($,Bo),l($,qe),l($,qo),_(S,$,null),l($,Po),l($,Pe),l($,Zo),_(D,$,null),l(v,Ao),l(v,B),_(_e,B,null),l(B,No),l(B,Ze),l(B,Go),l(B,Ae),l(B,Eo),_(Y,B,null),p(e,co,n),_(ye,e,n),p(e,po,n),p(e,P,n),_(Me,P,null),l(P,Wo),l(P,Ne),l(P,Ro),l(P,Ge),_(Te,Ge,null),p(e,mo,n),_(Ce,e,n),p(e,ho,n),p(e,V,n),_(we,V,null),l(V,Qo),l(V,Ee),l(V,Lo),l(V,We),l(V,Ho),l(V,Re),_(Ue,Re,null),p(e,uo,n),_(be,e,n),p(e,go,n),p(e,He,n),fo=!0},p(e,[n]){const E={};n&2&&(E.$$scope={dirty:n,ctx:e}),Q.$set(E);const J={};n&2&&(J.$$scope={dirty:n,ctx:e}),L.$set(J);const x={};n&2&&(x.$$scope={dirty:n,ctx:e}),H.$set(x);const W={};n&2&&(W.$$scope={dirty:n,ctx:e}),X.$set(W);const I={};n&2&&(I.$$scope={dirty:n,ctx:e}),S.$set(I);const z={};n&2&&(z.$$scope={dirty:n,ctx:e}),D.$set(z);const Z={};n&2&&(Z.$$scope={dirty:n,ctx:e}),Y.$set(Z)},i(e){fo||(y(N.$$.fragment,e),y(oe.$$.fragment,e),y(te.$$.fragment,e),y(ae.$$.fragment,e),y(Q.$$.fragment,e),y(le.$$.fragment,e),y(ce.$$.fragment,e),y(de.$$.fragment,e),y(pe.$$.fragment,e),y(me.$$.fragment,e),y(L.$$.fragment,e),y(H.$$.fragment,e),y(he.$$.fragment,e),y(X.$$.fragment,e),y(ue.$$.fragment,e),y(ge.$$.fragment,e),y(fe.$$.fragment,e),y(S.$$.fragment,e),y(D.$$.fragment,e),y(_e.$$.fragment,e),y(Y.$$.fragment,e),y(ye.$$.fragment,e),y(Me.$$.fragment,e),y(Te.$$.fragment,e),y(Ce.$$.fragment,e),y(we.$$.fragment,e),y(Ue.$$.fragment,e),y(be.$$.fragment,e),fo=!0)},o(e){M(N.$$.fragment,e),M(oe.$$.fragment,e),M(te.$$.fragment,e),M(ae.$$.fragment,e),M(Q.$$.fragment,e),M(le.$$.fragment,e),M(ce.$$.fragment,e),M(de.$$.fragment,e),M(pe.$$.fragment,e),M(me.$$.fragment,e),M(L.$$.fragment,e),M(H.$$.fragment,e),M(he.$$.fragment,e),M(X.$$.fragment,e),M(ue.$$.fragment,e),M(ge.$$.fragment,e),M(fe.$$.fragment,e),M(S.$$.fragment,e),M(D.$$.fragment,e),M(_e.$$.fragment,e),M(Y.$$.fragment,e),M(ye.$$.fragment,e),M(Me.$$.fragment,e),M(Te.$$.fragment,e),M(Ce.$$.fragment,e),M(we.$$.fragment,e),M(Ue.$$.fragment,e),M(be.$$.fragment,e),fo=!1},d(e){e&&(r(u),r(s),r(c),r(d),r(w),r(Xe),r(Se),r(R),r(De),r(Ye),r(se),r(Oe),r(ne),r(Ke),r(re),r(eo),r(oo),r(ie),r(to),r(so),r(no),r(q),r(ro),r(ao),r(b),r(io),r(lo),r(v),r(co),r(po),r(P),r(mo),r(ho),r(V),r(uo),r(go),r(He)),r(o),T(N,e),T(oe,e),T(te,e),T(ae,e),T(Q,e),T(le,e),T(ce),T(de,e),T(pe),T(me),T(L),T(H),T(he),T(X),T(ue,e),T(ge),T(fe),T(S),T(D),T(_e),T(Y),T(ye,e),T(Me),T(Te),T(Ce,e),T(we),T(Ue),T(be,e)}}}const At='{"title":"Command A Vision","local":"command-a-vision","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"Usage tips","local":"usage-tips","sections":[],"depth":2},{"title":"Cohere2VisionConfig","local":"transformers.Cohere2VisionConfig","sections":[],"depth":2},{"title":"Cohere2VisionForConditionalGeneration","local":"transformers.Cohere2VisionForConditionalGeneration","sections":[],"depth":2},{"title":"Cohere2VisionModel","local":"transformers.Cohere2VisionModel","sections":[],"depth":2},{"title":"Cohere2VisionImageProcessor","local":"transformers.Cohere2VisionImageProcessor","sections":[],"depth":2},{"title":"Cohere2VisionProcessor","local":"transformers.Cohere2VisionProcessor","sections":[],"depth":2}],"depth":1}';function Nt(U){return wt(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class Dt extends Ut{constructor(o){super(),bt(this,o,Nt,Zt,Ct,{})}}export{Dt as component}; | |
Xet Storage Details
- Size:
- 80.3 kB
- Xet hash:
- 4ea8304457d0c471ec1258e01a9e55dfcf420864d752b41cdfabc56fa1d06403
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.