Buckets:
| import{s as No,o as wo,n as ro}from"../chunks/scheduler.25b97de1.js";import{S as Lo,i as jo,g as d,s as n,r as p,A as Jo,h as c,f as t,c as a,j as z,u as g,x as b,k as C,y as l,a as s,v as h,d as f,t as u,w as _}from"../chunks/index.d9030fc9.js";import{T as Io}from"../chunks/Tip.baa67368.js";import{D as q}from"../chunks/Docstring.ffac8efa.js";import{C as Re}from"../chunks/CodeBlock.e6cd0d95.js";import{E as Mo}from"../chunks/ExampleCodeBlock.22dfe688.js";import{H as be,E as zo}from"../chunks/EditOnGithub.91d95064.js";function Co($){let i,M="Example:",v,m,y;return m=new Re({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMExsYXZhTmV4dEZvckNvbmRpdGlvbmFsR2VuZXJhdGlvbiUyQyUyMExsYXZhTmV4dENvbmZpZyUyQyUyMENMSVBWaXNpb25Db25maWclMkMlMjBMbGFtYUNvbmZpZyUwQSUwQSUyMyUyMEluaXRpYWxpemluZyUyMGElMjBDTElQLXZpc2lvbiUyMGNvbmZpZyUwQXZpc2lvbl9jb25maWclMjAlM0QlMjBDTElQVmlzaW9uQ29uZmlnKCklMEElMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwTGxhbWElMjBjb25maWclMEF0ZXh0X2NvbmZpZyUyMCUzRCUyMExsYW1hQ29uZmlnKCklMEElMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwTGxhdmEtTmV4dCUyMGxsYXZhLWhmJTJGbGxhdmEtdjEuNi1taXN0cmFsLTdiLWhmJTIwc3R5bGUlMjBjb25maWd1cmF0aW9uJTBBY29uZmlndXJhdGlvbiUyMCUzRCUyMExsYXZhTmV4dENvbmZpZyh2aXNpb25fY29uZmlnJTJDJTIwdGV4dF9jb25maWcpJTBBJTBBJTIzJTIwSW5pdGlhbGl6aW5nJTIwYSUyMG1vZGVsJTIwZnJvbSUyMHRoZSUyMGxsYXZhLWhmJTJGbGxhdmEtdjEuNi1taXN0cmFsLTdiLWhmJTIwc3R5bGUlMjBjb25maWd1cmF0aW9uJTBBbW9kZWwlMjAlM0QlMjBMbGF2YU5leHRGb3JDb25kaXRpb25hbEdlbmVyYXRpb24oY29uZmlndXJhdGlvbiklMEElMEElMjMlMjBBY2Nlc3NpbmclMjB0aGUlMjBtb2RlbCUyMGNvbmZpZ3VyYXRpb24lMEFjb25maWd1cmF0aW9uJTIwJTNEJTIwbW9kZWwuY29uZmln",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> LlavaNextForConditionalGeneration, LlavaNextConfig, CLIPVisionConfig, LlamaConfig | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a CLIP-vision config</span> | |
| <span class="hljs-meta">>>> </span>vision_config = CLIPVisionConfig() | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a Llama config</span> | |
| <span class="hljs-meta">>>> </span>text_config = LlamaConfig() | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a Llava-Next llava-hf/llava-v1.6-mistral-7b-hf style configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = LlavaNextConfig(vision_config, text_config) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a model from the llava-hf/llava-v1.6-mistral-7b-hf style configuration</span> | |
| <span class="hljs-meta">>>> </span>model = LlavaNextForConditionalGeneration(configuration) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Accessing the model configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = model.config`,wrap:!1}}),{c(){i=d("p"),i.textContent=M,v=n(),p(m.$$.fragment)},l(r){i=c(r,"P",{"data-svelte-h":!0}),b(i)!=="svelte-11lpom8"&&(i.textContent=M),v=a(r),g(m.$$.fragment,r)},m(r,T){s(r,i,T),s(r,v,T),h(m,r,T),y=!0},p:ro,i(r){y||(f(m.$$.fragment,r),y=!0)},o(r){u(m.$$.fragment,r),y=!1},d(r){r&&(t(i),t(v)),_(m,r)}}}function Uo($){let i,M=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){i=d("p"),i.innerHTML=M},l(v){i=c(v,"P",{"data-svelte-h":!0}),b(i)!=="svelte-fincs2"&&(i.innerHTML=M)},m(v,m){s(v,i,m)},p:ro,d(v){v&&t(i)}}}function $o($){let i,M="Example:",v,m,y;return m=new Re({props:{code:"ZnJvbSUyMFBJTCUyMGltcG9ydCUyMEltYWdlJTBBaW1wb3J0JTIwcmVxdWVzdHMlMEFmcm9tJTIwdHJhbnNmb3JtZXJzJTIwaW1wb3J0JTIwQXV0b1Byb2Nlc3NvciUyQyUyMExsYXZhTmV4dEZvckNvbmRpdGlvbmFsR2VuZXJhdGlvbiUwQSUwQW1vZGVsJTIwJTNEJTIwTGxhdmFOZXh0Rm9yQ29uZGl0aW9uYWxHZW5lcmF0aW9uLmZyb21fcHJldHJhaW5lZCglMjJsbGF2YS1oZiUyRmxsYXZhLXYxLjYtbWlzdHJhbC03Yi1oZiUyMiklMEFwcm9jZXNzb3IlMjAlM0QlMjBBdXRvUHJvY2Vzc29yLmZyb21fcHJldHJhaW5lZCglMjJsbGF2YS1oZiUyRmxsYXZhLXYxLjYtbWlzdHJhbC03Yi1oZiUyMiklMEElMEFwcm9tcHQlMjAlM0QlMjAlMjIlNUJJTlNUJTVEJTIwJTNDaW1hZ2UlM0UlNUNuV2hhdCUyMGlzJTIwc2hvd24lMjBpbiUyMHRoaXMlMjBpbWFnZSUzRiUyMCU1QiUyRklOU1QlNUQlMjIlMEF1cmwlMjAlM0QlMjAlMjJodHRwcyUzQSUyRiUyRnd3dy5pbGFua2VsbWFuLm9yZyUyRnN0b3BzaWducyUyRmF1c3RyYWxpYS5qcGclMjIlMEFpbWFnZSUyMCUzRCUyMEltYWdlLm9wZW4ocmVxdWVzdHMuZ2V0KHVybCUyQyUyMHN0cmVhbSUzRFRydWUpLnJhdyklMEElMEFpbnB1dHMlMjAlM0QlMjBwcm9jZXNzb3IoaW1hZ2VzJTNEaW1hZ2UlMkMlMjB0ZXh0JTNEcHJvbXB0JTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiklMEElMEElMjMlMjBHZW5lcmF0ZSUwQWdlbmVyYXRlX2lkcyUyMCUzRCUyMG1vZGVsLmdlbmVyYXRlKCoqaW5wdXRzJTJDJTIwbWF4X2xlbmd0aCUzRDMwKSUwQXByb2Nlc3Nvci5iYXRjaF9kZWNvZGUoZ2VuZXJhdGVfaWRzJTJDJTIwc2tpcF9zcGVjaWFsX3Rva2VucyUzRFRydWUlMkMlMjBjbGVhbl91cF90b2tlbml6YXRpb25fc3BhY2VzJTNERmFsc2UpJTVCMCU1RA==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> requests | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoProcessor, LlavaNextForConditionalGeneration | |
| <span class="hljs-meta">>>> </span>model = LlavaNextForConditionalGeneration.from_pretrained(<span class="hljs-string">"llava-hf/llava-v1.6-mistral-7b-hf"</span>) | |
| <span class="hljs-meta">>>> </span>processor = AutoProcessor.from_pretrained(<span class="hljs-string">"llava-hf/llava-v1.6-mistral-7b-hf"</span>) | |
| <span class="hljs-meta">>>> </span>prompt = <span class="hljs-string">"[INST] <image>\\nWhat is shown in this image? [/INST]"</span> | |
| <span class="hljs-meta">>>> </span>url = <span class="hljs-string">"https://www.ilankelman.org/stopsigns/australia.jpg"</span> | |
| <span class="hljs-meta">>>> </span>image = Image.<span class="hljs-built_in">open</span>(requests.get(url, stream=<span class="hljs-literal">True</span>).raw) | |
| <span class="hljs-meta">>>> </span>inputs = processor(images=image, text=prompt, return_tensors=<span class="hljs-string">"pt"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Generate</span> | |
| <span class="hljs-meta">>>> </span>generate_ids = model.generate(**inputs, max_length=<span class="hljs-number">30</span>) | |
| <span class="hljs-meta">>>> </span>processor.batch_decode(generate_ids, skip_special_tokens=<span class="hljs-literal">True</span>, clean_up_tokenization_spaces=<span class="hljs-literal">False</span>)[<span class="hljs-number">0</span>] | |
| <span class="hljs-string">"[INST] \\nWhat is shown in this image? [/INST] The image appears to be a radar chart, which is a type of multi-dimensional plot (...)"</span>`,wrap:!1}}),{c(){i=d("p"),i.textContent=M,v=n(),p(m.$$.fragment)},l(r){i=c(r,"P",{"data-svelte-h":!0}),b(i)!=="svelte-11lpom8"&&(i.textContent=M),v=a(r),g(m.$$.fragment,r)},m(r,T){s(r,i,T),s(r,v,T),h(m,r,T),y=!0},p:ro,i(r){y||(f(m.$$.fragment,r),y=!0)},o(r){u(m.$$.fragment,r),y=!1},d(r){r&&(t(i),t(v)),_(m,r)}}}function ko($){let i,M,v,m,y,r,T,xe,R,io='The Granite Vision model is a variant of <a href="llava_next">LLaVA-NeXT</a>, leveraging a <a href="granite">Granite</a> language model alongside a <a href="SigLIP">SigLIP</a> visual encoder. It utilizes multiple concatenated vision hidden states as its image features, similar to <a href="vipllava">VipLlava</a>. It also uses a larger set of image grid pinpoints than the original LlaVa-NeXT models to support additional aspect ratios.',Me,B,lo="Tips:",Ne,V,co='<li><p>This model is loaded into Transformers as an instance of LlaVA-Next. The usage and tips from <a href="llava_next">LLaVA-NeXT</a> apply to this model as well.</p></li> <li><p>You can apply the chat template on the tokenizer / processor in the same way as well. Example chat format:</p></li>',we,X,Le,H,mo="Sample inference:",je,A,Je,S,po='This model was contributed by <a href="https://huggingface.co/abrooks9944" rel="nofollow">Alexander Brooks</a>.',Ie,Y,ze,N,D,Be,le,go=`This is the configuration class to store the configuration of a <a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextForConditionalGeneration">LlavaNextForConditionalGeneration</a>. It is used to instantiate an | |
| Llava-NeXT model according to the specified arguments, defining the model architecture. Instantiating a configuration | |
| with the defaults will yield a similar configuration to that of the <a href="https://huggingface.co/llava-hf/llava-v1.6-mistral-7b-hf" rel="nofollow">llava-hf/llava-v1.6-mistral-7b-hf</a> | |
| model.`,Ve,de,ho=`Configuration objects inherit from <a href="/docs/transformers/pr_36049/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> and can be used to control the model outputs. Read the | |
| documentation from <a href="/docs/transformers/pr_36049/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> for more information.`,Xe,k,Ce,Q,Ue,j,O,He,ce,fo=`Constructs a LLaVa-NeXT image processor. Based on <a href="/docs/transformers/pr_36049/en/model_doc/clip#transformers.CLIPImageProcessor">CLIPImageProcessor</a> with incorporation of additional techniques | |
| for processing high resolution images as explained in the <a href="https://arxiv.org/abs/2310.03744" rel="nofollow">LLaVa paper</a>.`,Ae,me,K,$e,ee,ke,x,oe,Se,pe,uo="Constructs a LLaVa-NeXT processor which wraps a LLaVa-NeXT image processor and a LLaMa tokenizer into a single processor.",Ye,ge,_o=`<a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextProcessor">LlavaNextProcessor</a> offers all the functionalities of <a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextImageProcessor">LlavaNextImageProcessor</a> and <a href="/docs/transformers/pr_36049/en/model_doc/llama#transformers.LlamaTokenizerFast">LlamaTokenizerFast</a>. See the | |
| <code>__call__()</code> and <a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextProcessor.decode">decode()</a> for more information.`,De,F,te,Qe,he,vo=`This method forwards all its arguments to LlamaTokenizerFast’s <a href="/docs/transformers/pr_36049/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.batch_decode">batch_decode()</a>. Please | |
| refer to the docstring of this method for more information.`,Oe,P,ne,Ke,fe,bo=`This method forwards all its arguments to LlamaTokenizerFast’s <a href="/docs/transformers/pr_36049/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.decode">decode()</a>. Please refer to | |
| the docstring of this method for more information.`,Fe,ae,Pe,w,se,eo,ue,yo=`The LLAVA-NeXT model which consists of a vision backbone and a language model. | |
| This model inherits from <a href="/docs/transformers/pr_36049/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,oo,_e,To=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,to,L,re,no,ve,xo='The <a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextForConditionalGeneration">LlavaNextForConditionalGeneration</a> forward method, overrides the <code>__call__</code> special method.',ao,G,so,W,Ge,ie,We,ye,Ze;return y=new be({props:{title:"Granite Vision",local:"granite-vision",headingTag:"h1"}}),T=new be({props:{title:"Overview",local:"overview",headingTag:"h2"}}),X=new Re({props:{code:"JTIyJTNDJTdDdXNlciU3QyUzRSU1Q25XaGF0JUUyJTgwJTk5cyUyMHNob3duJTIwaW4lMjB0aGlzJTIwaW1hZ2UlM0YlNUNuJTNDJTdDYXNzaXN0YW50JTdDJTNFJTVDblRoaXMlMjBpbWFnZSUyMHNob3dzJTIwYSUyMHJlZCUyMHN0b3AlMjBzaWduLiUzQyU3Q2VuZF9vZl90ZXh0JTdDJTNFJTNDJTdDdXNlciU3QyUzRSU1Q25EZXNjcmliZSUyMHRoZSUyMGltYWdlJTIwaW4lMjBtb3JlJTIwZGV0YWlscy4lNUNuJTNDJTdDYXNzaXN0YW50JTdDJTNFJTVDbiUyMg==",highlighted:'<span class="hljs-string">"<|user|>\\nWhat’s shown in this image?\\n<|assistant|>\\nThis image shows a red stop sign.<|end_of_text|><|user|>\\nDescribe the image in more details.\\n<|assistant|>\\n"</span>',wrap:!1}}),A=new Re({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMExsYXZhTmV4dFByb2Nlc3NvciUyQyUyMExsYXZhTmV4dEZvckNvbmRpdGlvbmFsR2VuZXJhdGlvbiUwQSUwQW1vZGVsX3BhdGglMjAlM0QlMjAlMjJpYm0tZ3Jhbml0ZSUyRmdyYW5pdGUtdmlzaW9uLTMuMS0yYi1wcmV2aWV3JTIyJTBBcHJvY2Vzc29yJTIwJTNEJTIwTGxhdmFOZXh0UHJvY2Vzc29yLmZyb21fcHJldHJhaW5lZChtb2RlbF9wYXRoKSUwQSUwQW1vZGVsJTIwJTNEJTIwTGxhdmFOZXh0Rm9yQ29uZGl0aW9uYWxHZW5lcmF0aW9uLmZyb21fcHJldHJhaW5lZChtb2RlbF9wYXRoKS50byglMjJjdWRhJTIyKSUwQSUwQSUyMyUyMHByZXBhcmUlMjBpbWFnZSUyMGFuZCUyMHRleHQlMjBwcm9tcHQlMkMlMjB1c2luZyUyMHRoZSUyMGFwcHJvcHJpYXRlJTIwcHJvbXB0JTIwdGVtcGxhdGUlMEF1cmwlMjAlM0QlMjAlMjJodHRwcyUzQSUyRiUyRmdpdGh1Yi5jb20lMkZoYW90aWFuLWxpdSUyRkxMYVZBJTJGYmxvYiUyRjFhOTFmYzI3NGQ3YzM1YTliNTBiM2NiMjljNDI0N2FlNTgzN2NlMzklMkZpbWFnZXMlMkZsbGF2YV92MV81X3JhZGFyLmpwZyUzRnJhdyUzRHRydWUlMjIlMEElMEFjb252ZXJzYXRpb24lMjAlM0QlMjAlNUIlMEElMjAlMjAlMjAlMjAlN0IlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjJyb2xlJTIyJTNBJTIwJTIydXNlciUyMiUyQyUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMmNvbnRlbnQlMjIlM0ElMjAlNUIlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlN0IlMjJ0eXBlJTIyJTNBJTIwJTIyaW1hZ2UlMjIlMkMlMjAlMjJ1cmwlMjIlM0ElMjB1cmwlN0QlMkMlMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjAlN0IlMjJ0eXBlJTIyJTNBJTIwJTIydGV4dCUyMiUyQyUyMCUyMnRleHQlMjIlM0ElMjAlMjJXaGF0JTIwaXMlMjBzaG93biUyMGluJTIwdGhpcyUyMGltYWdlJTNGJTIyJTdEJTJDJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTVEJTJDJTBBJTIwJTIwJTIwJTIwJTdEJTJDJTBBJTVEJTBBaW5wdXRzJTIwJTNEJTIwcHJvY2Vzc29yLmFwcGx5X2NoYXRfdGVtcGxhdGUoJTBBJTIwJTIwJTIwJTIwY29udmVyc2F0aW9uJTJDJTBBJTIwJTIwJTIwJTIwYWRkX2dlbmVyYXRpb25fcHJvbXB0JTNEVHJ1ZSUyQyUwQSUyMCUyMCUyMCUyMHRva2VuaXplJTNEVHJ1ZSUyQyUwQSUyMCUyMCUyMCUyMHJldHVybl9kaWN0JTNEVHJ1ZSUyQyUwQSUyMCUyMCUyMCUyMHJldHVybl90ZW5zb3JzJTNEJTIycHQlMjIlMEEpLnRvKCUyMmN1ZGElMjIpJTBBJTBBJTBBJTIzJTIwYXV0b3JlZ3Jlc3NpdmVseSUyMGNvbXBsZXRlJTIwcHJvbXB0JTBBb3V0cHV0JTIwJTNEJTIwbW9kZWwuZ2VuZXJhdGUoKippbnB1dHMlMkMlMjBtYXhfbmV3X3Rva2VucyUzRDEwMCklMEElMEFwcmludChwcm9jZXNzb3IuZGVjb2RlKG91dHB1dCU1QjAlNUQlMkMlMjBza2lwX3NwZWNpYWxfdG9rZW5zJTNEVHJ1ZSkp",highlighted:`<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> LlavaNextProcessor, LlavaNextForConditionalGeneration | |
| model_path = <span class="hljs-string">"ibm-granite/granite-vision-3.1-2b-preview"</span> | |
| processor = LlavaNextProcessor.from_pretrained(model_path) | |
| model = LlavaNextForConditionalGeneration.from_pretrained(model_path).to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-comment"># prepare image and text prompt, using the appropriate prompt template</span> | |
| url = <span class="hljs-string">"https://github.com/haotian-liu/LLaVA/blob/1a91fc274d7c35a9b50b3cb29c4247ae5837ce39/images/llava_v1_5_radar.jpg?raw=true"</span> | |
| conversation = [ | |
| { | |
| <span class="hljs-string">"role"</span>: <span class="hljs-string">"user"</span>, | |
| <span class="hljs-string">"content"</span>: [ | |
| {<span class="hljs-string">"type"</span>: <span class="hljs-string">"image"</span>, <span class="hljs-string">"url"</span>: url}, | |
| {<span class="hljs-string">"type"</span>: <span class="hljs-string">"text"</span>, <span class="hljs-string">"text"</span>: <span class="hljs-string">"What is shown in this image?"</span>}, | |
| ], | |
| }, | |
| ] | |
| inputs = processor.apply_chat_template( | |
| conversation, | |
| add_generation_prompt=<span class="hljs-literal">True</span>, | |
| tokenize=<span class="hljs-literal">True</span>, | |
| return_dict=<span class="hljs-literal">True</span>, | |
| return_tensors=<span class="hljs-string">"pt"</span> | |
| ).to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-comment"># autoregressively complete prompt</span> | |
| output = model.generate(**inputs, max_new_tokens=<span class="hljs-number">100</span>) | |
| <span class="hljs-built_in">print</span>(processor.decode(output[<span class="hljs-number">0</span>], skip_special_tokens=<span class="hljs-literal">True</span>))`,wrap:!1}}),Y=new be({props:{title:"LlavaNextConfig",local:"transformers.LlavaNextConfig",headingTag:"h2"}}),D=new q({props:{name:"class transformers.LlavaNextConfig",anchor:"transformers.LlavaNextConfig",parameters:[{name:"vision_config",val:" = None"},{name:"text_config",val:" = None"},{name:"ignore_index",val:" = -100"},{name:"image_token_index",val:" = 32000"},{name:"projector_hidden_act",val:" = 'gelu'"},{name:"vision_feature_select_strategy",val:" = 'default'"},{name:"vision_feature_layer",val:" = -2"},{name:"image_grid_pinpoints",val:" = None"},{name:"tie_word_embeddings",val:" = False"},{name:"image_seq_length",val:" = 576"},{name:"multimodal_projector_bias",val:" = True"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.LlavaNextConfig.vision_config",description:`<strong>vision_config</strong> (<code>Union[AutoConfig, dict]</code>, <em>optional</em>, defaults to <code>CLIPVisionConfig</code>) — | |
| The config object or dictionary of the vision backbone.`,name:"vision_config"},{anchor:"transformers.LlavaNextConfig.text_config",description:`<strong>text_config</strong> (<code>Union[AutoConfig, dict]</code>, <em>optional</em>, defaults to <code>LlamaConfig</code>) — | |
| The config object or dictionary of the text backbone.`,name:"text_config"},{anchor:"transformers.LlavaNextConfig.ignore_index",description:`<strong>ignore_index</strong> (<code>int</code>, <em>optional</em>, defaults to -100) — | |
| The ignore index for the loss function.`,name:"ignore_index"},{anchor:"transformers.LlavaNextConfig.image_token_index",description:`<strong>image_token_index</strong> (<code>int</code>, <em>optional</em>, defaults to 32000) — | |
| The image token index to encode the image prompt.`,name:"image_token_index"},{anchor:"transformers.LlavaNextConfig.projector_hidden_act",description:`<strong>projector_hidden_act</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"gelu"</code>) — | |
| The activation function used by the multimodal projector.`,name:"projector_hidden_act"},{anchor:"transformers.LlavaNextConfig.vision_feature_select_strategy",description:`<strong>vision_feature_select_strategy</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"default"</code>) — | |
| The feature selection strategy used to select the vision feature from the vision backbone. | |
| Can be one of <code>"default"</code> or <code>"full"</code>. If <code>"default"</code>, the CLS token is removed from the vision features. | |
| If <code>"full"</code>, the full vision features are used.`,name:"vision_feature_select_strategy"},{anchor:"transformers.LlavaNextConfig.vision_feature_layer",description:`<strong>vision_feature_layer</strong> (<code>Union[int, List[int]]</code>, <em>optional</em>, defaults to -2) — | |
| The index of the layer to select the vision feature. If multiple indices are provided, | |
| the vision feature of the corresponding indices will be concatenated to form the | |
| vision features.`,name:"vision_feature_layer"},{anchor:"transformers.LlavaNextConfig.image_grid_pinpoints",description:`<strong>image_grid_pinpoints</strong> (<code>List</code>, <em>optional</em>, defaults to <code>[[336, 672], [672, 336], [672, 672], [1008, 336], [336, 1008]]</code>) — | |
| A list of possible resolutions to use for processing high resolution images. Each item in the list should be a tuple or list | |
| of the form <code>(height, width)</code>.`,name:"image_grid_pinpoints"},{anchor:"transformers.LlavaNextConfig.tie_word_embeddings",description:`<strong>tie_word_embeddings</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether the model’s input and output word embeddings should be tied.`,name:"tie_word_embeddings"},{anchor:"transformers.LlavaNextConfig.image_seq_length",description:`<strong>image_seq_length</strong> (<code>int</code>, <em>optional</em>, defaults to 576) — | |
| Sequence length of one image embedding.`,name:"image_seq_length"},{anchor:"transformers.LlavaNextConfig.multimodal_projector_bias",description:`<strong>multimodal_projector_bias</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to use bias in the multimodal projector.`,name:"multimodal_projector_bias"}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/configuration_llava_next.py#L24"}}),k=new Mo({props:{anchor:"transformers.LlavaNextConfig.example",$$slots:{default:[Co]},$$scope:{ctx:$}}}),Q=new be({props:{title:"LlavaNextImageProcessor",local:"transformers.LlavaNextImageProcessor",headingTag:"h2"}}),O=new q({props:{name:"class transformers.LlavaNextImageProcessor",anchor:"transformers.LlavaNextImageProcessor",parameters:[{name:"do_resize",val:": bool = True"},{name:"size",val:": typing.Dict[str, int] = None"},{name:"image_grid_pinpoints",val:": typing.List = None"},{name:"resample",val:": Resampling = <Resampling.BICUBIC: 3>"},{name:"do_center_crop",val:": bool = True"},{name:"crop_size",val:": typing.Dict[str, int] = None"},{name:"do_rescale",val:": bool = True"},{name:"rescale_factor",val:": typing.Union[int, float] = 0.00392156862745098"},{name:"do_normalize",val:": bool = True"},{name:"image_mean",val:": typing.Union[float, typing.List[float], NoneType] = None"},{name:"image_std",val:": typing.Union[float, typing.List[float], NoneType] = None"},{name:"do_pad",val:": typing.Optional[bool] = True"},{name:"do_convert_rgb",val:": bool = True"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.LlavaNextImageProcessor.do_resize",description:`<strong>do_resize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to resize the image’s (height, width) dimensions to the specified <code>size</code>. Can be overridden by | |
| <code>do_resize</code> in the <code>preprocess</code> method.`,name:"do_resize"},{anchor:"transformers.LlavaNextImageProcessor.size",description:`<strong>size</strong> (<code>Dict[str, int]</code> <em>optional</em>, defaults to <code>{"shortest_edge" -- 224}</code>): | |
| Size of the image after resizing. The shortest edge of the image is resized to size[“shortest_edge”], with | |
| the longest edge resized to keep the input aspect ratio. Can be overridden by <code>size</code> in the <code>preprocess</code> | |
| method.`,name:"size"},{anchor:"transformers.LlavaNextImageProcessor.image_grid_pinpoints",description:`<strong>image_grid_pinpoints</strong> (<code>List</code> <em>optional</em>, defaults to <code>[[672, 336], [336, 672], [672, 672], [336, 1008], [1008, 336]]</code>) — | |
| A list of possible resolutions to use for processing high resolution images. The best resolution is selected | |
| based on the original size of the image. Can be overridden by <code>image_grid_pinpoints</code> in the <code>preprocess</code> | |
| method.`,name:"image_grid_pinpoints"},{anchor:"transformers.LlavaNextImageProcessor.resample",description:`<strong>resample</strong> (<code>PILImageResampling</code>, <em>optional</em>, defaults to <code>Resampling.BICUBIC</code>) — | |
| Resampling filter to use if resizing the image. Can be overridden by <code>resample</code> in the <code>preprocess</code> method.`,name:"resample"},{anchor:"transformers.LlavaNextImageProcessor.do_center_crop",description:`<strong>do_center_crop</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to center crop the image to the specified <code>crop_size</code>. Can be overridden by <code>do_center_crop</code> in the | |
| <code>preprocess</code> method.`,name:"do_center_crop"},{anchor:"transformers.LlavaNextImageProcessor.crop_size",description:`<strong>crop_size</strong> (<code>Dict[str, int]</code> <em>optional</em>, defaults to 224) — | |
| Size of the output image after applying <code>center_crop</code>. Can be overridden by <code>crop_size</code> in the <code>preprocess</code> | |
| method.`,name:"crop_size"},{anchor:"transformers.LlavaNextImageProcessor.do_rescale",description:`<strong>do_rescale</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to rescale the image by the specified scale <code>rescale_factor</code>. Can be overridden by <code>do_rescale</code> in | |
| the <code>preprocess</code> method.`,name:"do_rescale"},{anchor:"transformers.LlavaNextImageProcessor.rescale_factor",description:`<strong>rescale_factor</strong> (<code>int</code> or <code>float</code>, <em>optional</em>, defaults to <code>1/255</code>) — | |
| Scale factor to use if rescaling the image. Can be overridden by <code>rescale_factor</code> in the <code>preprocess</code> | |
| method.`,name:"rescale_factor"},{anchor:"transformers.LlavaNextImageProcessor.do_normalize",description:`<strong>do_normalize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to normalize the image. Can be overridden by <code>do_normalize</code> in the <code>preprocess</code> method.`,name:"do_normalize"},{anchor:"transformers.LlavaNextImageProcessor.image_mean",description:`<strong>image_mean</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>[0.48145466, 0.4578275, 0.40821073]</code>) — | |
| Mean to use if normalizing the image. This is a float or list of floats the length of the number of | |
| channels in the image. Can be overridden by the <code>image_mean</code> parameter in the <code>preprocess</code> method.`,name:"image_mean"},{anchor:"transformers.LlavaNextImageProcessor.image_std",description:`<strong>image_std</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>[0.26862954, 0.26130258, 0.27577711]</code>) — | |
| Standard deviation to use if normalizing the image. This is a float or list of floats the length of the | |
| number of channels in the image. Can be overridden by the <code>image_std</code> parameter in the <code>preprocess</code> method. | |
| Can be overridden by the <code>image_std</code> parameter in the <code>preprocess</code> method.`,name:"image_std"},{anchor:"transformers.LlavaNextImageProcessor.do_pad",description:`<strong>do_pad</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to pad the image. If <code>True</code>, will pad the patch dimension of the images in the batch to the largest | |
| number of patches in the batch. Padding will be applied to the bottom and right with zeros.`,name:"do_pad"},{anchor:"transformers.LlavaNextImageProcessor.do_convert_rgb",description:`<strong>do_convert_rgb</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to convert the image to RGB.`,name:"do_convert_rgb"}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/image_processing_llava_next.py#L119"}}),K=new q({props:{name:"preprocess",anchor:"transformers.LlavaNextImageProcessor.preprocess",parameters:[{name:"images",val:": typing.Union[ForwardRef('PIL.Image.Image'), numpy.ndarray, ForwardRef('torch.Tensor'), typing.List[ForwardRef('PIL.Image.Image')], typing.List[numpy.ndarray], typing.List[ForwardRef('torch.Tensor')]]"},{name:"do_resize",val:": bool = None"},{name:"size",val:": typing.Dict[str, int] = None"},{name:"image_grid_pinpoints",val:": typing.List = None"},{name:"resample",val:": Resampling = None"},{name:"do_center_crop",val:": bool = None"},{name:"crop_size",val:": int = None"},{name:"do_rescale",val:": bool = None"},{name:"rescale_factor",val:": float = None"},{name:"do_normalize",val:": bool = None"},{name:"image_mean",val:": typing.Union[float, typing.List[float], NoneType] = None"},{name:"image_std",val:": typing.Union[float, typing.List[float], NoneType] = None"},{name:"do_pad",val:": typing.Optional[bool] = None"},{name:"do_convert_rgb",val:": bool = None"},{name:"return_tensors",val:": typing.Union[str, transformers.utils.generic.TensorType, NoneType] = None"},{name:"data_format",val:": typing.Optional[transformers.image_utils.ChannelDimension] = <ChannelDimension.FIRST: 'channels_first'>"},{name:"input_data_format",val:": typing.Union[str, transformers.image_utils.ChannelDimension, NoneType] = None"}],parametersDescription:[{anchor:"transformers.LlavaNextImageProcessor.preprocess.images",description:`<strong>images</strong> (<code>ImageInput</code>) — | |
| Image to preprocess. Expects a single or batch of images with pixel values ranging from 0 to 255. If | |
| passing in images with pixel values between 0 and 1, set <code>do_rescale=False</code>.`,name:"images"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.do_resize",description:`<strong>do_resize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_resize</code>) — | |
| Whether to resize the image.`,name:"do_resize"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.size",description:`<strong>size</strong> (<code>Dict[str, int]</code>, <em>optional</em>, defaults to <code>self.size</code>) — | |
| Size of the image after resizing. Shortest edge of the image is resized to size[“shortest_edge”], with | |
| the longest edge resized to keep the input aspect ratio.`,name:"size"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.image_grid_pinpoints",description:`<strong>image_grid_pinpoints</strong> (<code>List</code> <em>optional</em>, defaults to <code>self.image_grid_pinpoints</code>) — | |
| A list of possible resolutions to use for processing high resolution images. The best resolution is | |
| selected based on the original size of the image.`,name:"image_grid_pinpoints"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.resample",description:`<strong>resample</strong> (<code>int</code>, <em>optional</em>, defaults to <code>self.resample</code>) — | |
| Resampling filter to use if resizing the image. This can be one of the enum <code>PILImageResampling</code>. Only | |
| has an effect if <code>do_resize</code> is set to <code>True</code>.`,name:"resample"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.do_center_crop",description:`<strong>do_center_crop</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_center_crop</code>) — | |
| Whether to center crop the image.`,name:"do_center_crop"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.crop_size",description:`<strong>crop_size</strong> (<code>Dict[str, int]</code>, <em>optional</em>, defaults to <code>self.crop_size</code>) — | |
| Size of the center crop. Only has an effect if <code>do_center_crop</code> is set to <code>True</code>.`,name:"crop_size"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.do_rescale",description:`<strong>do_rescale</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_rescale</code>) — | |
| Whether to rescale the image.`,name:"do_rescale"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.rescale_factor",description:`<strong>rescale_factor</strong> (<code>float</code>, <em>optional</em>, defaults to <code>self.rescale_factor</code>) — | |
| Rescale factor to rescale the image by if <code>do_rescale</code> is set to <code>True</code>.`,name:"rescale_factor"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.do_normalize",description:`<strong>do_normalize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_normalize</code>) — | |
| Whether to normalize the image.`,name:"do_normalize"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.image_mean",description:`<strong>image_mean</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>self.image_mean</code>) — | |
| Image mean to use for normalization. Only has an effect if <code>do_normalize</code> is set to <code>True</code>.`,name:"image_mean"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.image_std",description:`<strong>image_std</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>self.image_std</code>) — | |
| Image standard deviation to use for normalization. Only has an effect if <code>do_normalize</code> is set to | |
| <code>True</code>.`,name:"image_std"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.do_pad",description:`<strong>do_pad</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_pad</code>) — | |
| Whether to pad the image. If <code>True</code>, will pad the patch dimension of the images in the batch to the largest | |
| number of patches in the batch. Padding will be applied to the bottom and right with zeros.`,name:"do_pad"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.do_convert_rgb",description:`<strong>do_convert_rgb</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_convert_rgb</code>) — | |
| Whether to convert the image to RGB.`,name:"do_convert_rgb"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <code>TensorType</code>, <em>optional</em>) — | |
| The type of tensors to return. Can be one of:<ul> | |
| <li>Unset: Return a list of <code>np.ndarray</code>.</li> | |
| <li><code>TensorType.TENSORFLOW</code> or <code>'tf'</code>: Return a batch of type <code>tf.Tensor</code>.</li> | |
| <li><code>TensorType.PYTORCH</code> or <code>'pt'</code>: Return a batch of type <code>torch.Tensor</code>.</li> | |
| <li><code>TensorType.NUMPY</code> or <code>'np'</code>: Return a batch of type <code>np.ndarray</code>.</li> | |
| <li><code>TensorType.JAX</code> or <code>'jax'</code>: Return a batch of type <code>jax.numpy.ndarray</code>.</li> | |
| </ul>`,name:"return_tensors"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.data_format",description:`<strong>data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>, defaults to <code>ChannelDimension.FIRST</code>) — | |
| The channel dimension format for the output image. Can be one of:<ul> | |
| <li><code>"channels_first"</code> or <code>ChannelDimension.FIRST</code>: image in (num_channels, height, width) format.</li> | |
| <li><code>"channels_last"</code> or <code>ChannelDimension.LAST</code>: image in (height, width, num_channels) format.</li> | |
| <li>Unset: Use the channel dimension format of the input image.</li> | |
| </ul>`,name:"data_format"},{anchor:"transformers.LlavaNextImageProcessor.preprocess.input_data_format",description:`<strong>input_data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>) — | |
| The channel dimension format for the input image. If unset, the channel dimension format is inferred | |
| from the input image. Can be one of:<ul> | |
| <li><code>"channels_first"</code> or <code>ChannelDimension.FIRST</code>: image in (num_channels, height, width) format.</li> | |
| <li><code>"channels_last"</code> or <code>ChannelDimension.LAST</code>: image in (height, width, num_channels) format.</li> | |
| <li><code>"none"</code> or <code>ChannelDimension.NONE</code>: image in (height, width) format.</li> | |
| </ul>`,name:"input_data_format"}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/image_processing_llava_next.py#L558"}}),ee=new be({props:{title:"LlavaNextProcessor",local:"transformers.LlavaNextProcessor",headingTag:"h2"}}),oe=new q({props:{name:"class transformers.LlavaNextProcessor",anchor:"transformers.LlavaNextProcessor",parameters:[{name:"image_processor",val:" = None"},{name:"tokenizer",val:" = None"},{name:"patch_size",val:" = None"},{name:"vision_feature_select_strategy",val:" = None"},{name:"chat_template",val:" = None"},{name:"image_token",val:" = '<image>'"},{name:"num_additional_image_tokens",val:" = 0"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.LlavaNextProcessor.image_processor",description:`<strong>image_processor</strong> (<a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextImageProcessor">LlavaNextImageProcessor</a>, <em>optional</em>) — | |
| The image processor is a required input.`,name:"image_processor"},{anchor:"transformers.LlavaNextProcessor.tokenizer",description:`<strong>tokenizer</strong> (<a href="/docs/transformers/pr_36049/en/model_doc/llama#transformers.LlamaTokenizerFast">LlamaTokenizerFast</a>, <em>optional</em>) — | |
| The tokenizer is a required input.`,name:"tokenizer"},{anchor:"transformers.LlavaNextProcessor.patch_size",description:`<strong>patch_size</strong> (<code>int</code>, <em>optional</em>) — | |
| Patch size from the vision tower.`,name:"patch_size"},{anchor:"transformers.LlavaNextProcessor.vision_feature_select_strategy",description:`<strong>vision_feature_select_strategy</strong> (<code>str</code>, <em>optional</em>) — | |
| The feature selection strategy used to select the vision feature from the vision backbone. | |
| Shoudl be same as in model’s config`,name:"vision_feature_select_strategy"},{anchor:"transformers.LlavaNextProcessor.chat_template",description:`<strong>chat_template</strong> (<code>str</code>, <em>optional</em>) — A Jinja template which will be used to convert lists of messages | |
| in a chat into a tokenizable string.`,name:"chat_template"},{anchor:"transformers.LlavaNextProcessor.image_token",description:`<strong>image_token</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"<image>"</code>) — | |
| Special token used to denote image location.`,name:"image_token"},{anchor:"transformers.LlavaNextProcessor.num_additional_image_tokens",description:`<strong>num_additional_image_tokens</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of additional tokens added to the image embeddings, such as CLS (+1). If the backbone has no CLS or other | |
| extra tokens appended, no need to set this arg.`,name:"num_additional_image_tokens"}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/processing_llava_next.py#L43"}}),te=new q({props:{name:"batch_decode",anchor:"transformers.LlavaNextProcessor.batch_decode",parameters:[{name:"*args",val:""},{name:"**kwargs",val:""}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/processing_llava_next.py#L216"}}),ne=new q({props:{name:"decode",anchor:"transformers.LlavaNextProcessor.decode",parameters:[{name:"*args",val:""},{name:"**kwargs",val:""}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/processing_llava_next.py#L224"}}),ae=new be({props:{title:"LlavaNextForConditionalGeneration",local:"transformers.LlavaNextForConditionalGeneration",headingTag:"h2"}}),se=new q({props:{name:"class transformers.LlavaNextForConditionalGeneration",anchor:"transformers.LlavaNextForConditionalGeneration",parameters:[{name:"config",val:": LlavaNextConfig"}],parametersDescription:[{anchor:"transformers.LlavaNextForConditionalGeneration.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextConfig">LlavaNextConfig</a> or <code>LlavaNextVisionConfig</code>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_36049/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/modeling_llava_next.py#L352"}}),re=new q({props:{name:"forward",anchor:"transformers.LlavaNextForConditionalGeneration.forward",parameters:[{name:"input_ids",val:": LongTensor = None"},{name:"pixel_values",val:": FloatTensor = None"},{name:"image_sizes",val:": typing.Optional[torch.LongTensor] = None"},{name:"attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"position_ids",val:": typing.Optional[torch.LongTensor] = None"},{name:"past_key_values",val:": typing.Optional[typing.List[torch.FloatTensor]] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"vision_feature_layer",val:": typing.Union[int, typing.List[int], NoneType] = None"},{name:"vision_feature_select_strategy",val:": typing.Optional[str] = None"},{name:"labels",val:": typing.Optional[torch.LongTensor] = None"},{name:"use_cache",val:": typing.Optional[bool] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"},{name:"cache_position",val:": typing.Optional[torch.LongTensor] = None"},{name:"logits_to_keep",val:": typing.Union[int, torch.Tensor] = 0"}],parametersDescription:[{anchor:"transformers.LlavaNextForConditionalGeneration.forward.input_ids",description:`<strong>input_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>) — | |
| Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide | |
| it.</p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_36049/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_36049/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_36049/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p><a href="../glossary#input-ids">What are input IDs?</a>`,name:"input_ids"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape \`(batch_size, num_channels, image_size, image_size)) — | |
| The tensors corresponding to the input images. Pixel values can be obtained using | |
| <a href="/docs/transformers/pr_36049/en/model_doc/auto#transformers.AutoImageProcessor">AutoImageProcessor</a>. See <a href="/docs/transformers/pr_36049/en/model_doc/deit#transformers.DeiTFeatureExtractor.__call__">LlavaNextImageProcessor.<strong>call</strong>()</a> for details. <a href="/docs/transformers/pr_36049/en/model_doc/llava#transformers.LlavaProcessor">LlavaProcessor</a> uses | |
| <a href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextImageProcessor">LlavaNextImageProcessor</a> for processing images.`,name:"pixel_values"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.image_sizes",description:`<strong>image_sizes</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, 2)</code>, <em>optional</em>) — | |
| The sizes of the images in the batch, being (height, width) for each image.`,name:"image_sizes"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.attention_mask",description:`<strong>attention_mask</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding token indices. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for tokens that are <strong>not masked</strong>,</li> | |
| <li>0 for tokens that are <strong>masked</strong>.</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a></p> | |
| <p>Indices can be obtained using <a href="/docs/transformers/pr_36049/en/model_doc/auto#transformers.AutoTokenizer">AutoTokenizer</a>. See <a href="/docs/transformers/pr_36049/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.encode">PreTrainedTokenizer.encode()</a> and | |
| <a href="/docs/transformers/pr_36049/en/internal/tokenization_utils#transformers.PreTrainedTokenizerBase.__call__">PreTrainedTokenizer.<strong>call</strong>()</a> for details.</p> | |
| <p>If <code>past_key_values</code> is used, optionally only the last <code>decoder_input_ids</code> have to be input (see | |
| <code>past_key_values</code>).</p> | |
| <p>If you want to change padding behavior, you should read <code>modeling_opt._prepare_decoder_attention_mask</code> | |
| and modify to your needs. See diagram 1 in <a href="https://arxiv.org/abs/1910.13461" rel="nofollow">the paper</a> for more | |
| information on the default strategy.</p> | |
| <ul> | |
| <li>1 indicates the head is <strong>not masked</strong>,</li> | |
| <li>0 indicates the head is <strong>masked</strong>.</li> | |
| </ul>`,name:"attention_mask"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>) — | |
| Indices of positions of each input sequence tokens in the position embeddings. Selected in the range <code>[0, config.n_positions - 1]</code>. <a href="../glossary#position-ids">What are position IDs?</a>`,name:"position_ids"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.past_key_values",description:`<strong>past_key_values</strong> (<code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — | |
| Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of shape | |
| <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>) and 2 additional tensors of shape | |
| <code>(batch_size, num_heads, encoder_sequence_length, embed_size_per_head)</code>.</p> | |
| <p>Contains pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention | |
| blocks) that can be used (see <code>past_key_values</code> input) to speed up sequential decoding.</p> | |
| <p>If <code>past_key_values</code> are used, the user can optionally input only the last <code>decoder_input_ids</code> (those that | |
| don’t have their past key value states given to this model) of shape <code>(batch_size, 1)</code> instead of all | |
| <code>decoder_input_ids</code> of shape <code>(batch_size, sequence_length)</code>.`,name:"past_key_values"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing <code>input_ids</code> you can choose to directly pass an embedded representation. This | |
| is useful if you want more control over how to convert <code>input_ids</code> indices into associated vectors than the | |
| model’s internal embedding lookup matrix.`,name:"inputs_embeds"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.vision_feature_layer",description:`<strong>vision_feature_layer</strong> (<code>Union[int, List[int]], *optional*, defaults to -2</code>) — | |
| The index of the layer to select the vision feature. If multiple indices are provided, | |
| the vision feature of the corresponding indices will be concatenated to form the | |
| vision features.`,name:"vision_feature_layer"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.vision_feature_select_strategy",description:`<strong>vision_feature_select_strategy</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"default"</code>) — | |
| The feature selection strategy used to select the vision feature from the vision backbone. | |
| Can be one of <code>"default"</code> or <code>"full"</code>. If <code>"default"</code>, the CLS token is removed from the vision features. | |
| If <code>"full"</code>, the full vision features are used.`,name:"vision_feature_select_strategy"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.use_cache",description:`<strong>use_cache</strong> (<code>bool</code>, <em>optional</em>) — | |
| If set to <code>True</code>, <code>past_key_values</code> key value states are returned and can be used to speed up decoding (see | |
| <code>past_key_values</code>).`,name:"use_cache"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_36049/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.cache_position",description:`<strong>cache_position</strong> (<code>torch.LongTensor</code> of shape <code>(sequence_length)</code>, <em>optional</em>) — | |
| Indices depicting the position of the input sequence tokens in the sequence. Contrarily to <code>position_ids</code>, | |
| this tensor is not affected by padding. It is used to update the cache in the correct position and to infer | |
| the complete sequence length.`,name:"cache_position"},{anchor:"transformers.LlavaNextForConditionalGeneration.forward.Args",description:`<strong>Args</strong> — | |
| labels (<code>torch.LongTensor</code> of shape <code>(batch_size, sequence_length)</code>, <em>optional</em>): | |
| Labels for computing the masked language modeling loss. Indices should either be in <code>[0, ..., config.vocab_size]</code> or -100 (see <code>input_ids</code> docstring). Tokens with indices set to <code>-100</code> are ignored | |
| (masked), the loss is only computed for the tokens with labels in <code>[0, ..., config.vocab_size]</code>.</p> | |
| <p>logits_to_keep (<code>int</code> or <code>torch.Tensor</code>, <em>optional</em>): | |
| If an <code>int</code>, compute logits for the last <code>logits_to_keep</code> tokens. If <code>0</code>, calculate logits for all | |
| <code>input_ids</code> (special case). Only last token logits are needed for generation, and calculating them only for that | |
| token can save memory, which becomes pretty significant for long sequences or large vocabulary size. | |
| If a <code>torch.Tensor</code>, must be 1D corresponding to the indices to keep in the sequence length dimension. | |
| This is useful when using packed tensor format (single dimension for batch and sequence length).`,name:"Args"}],source:"https://github.com/huggingface/transformers/blob/vr_36049/src/transformers/models/llava_next/modeling_llava_next.py#L776",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <code>transformers.models.llava_next.modeling_llava_next.LlavaNextCausalLMOutputWithPast</code> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_36049/en/model_doc/llava_next#transformers.LlavaNextConfig" | |
| >LlavaNextConfig</a>) and inputs.</p> | |
| <ul> | |
| <li> | |
| <p><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> is provided) — Language modeling loss (for next-token prediction).</p> | |
| </li> | |
| <li> | |
| <p><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, config.vocab_size)</code>) — Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).</p> | |
| </li> | |
| <li> | |
| <p><strong>past_key_values</strong> (<code>tuple(tuple(torch.FloatTensor))</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — Tuple of <code>tuple(torch.FloatTensor)</code> of length <code>config.n_layers</code>, with each tuple having 2 tensors of shape | |
| <code>(batch_size, num_heads, sequence_length, embed_size_per_head)</code>)</p> | |
| <p>Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see | |
| <code>past_key_values</code> input) to speed up sequential decoding.</p> | |
| </li> | |
| <li> | |
| <p><strong>hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, + | |
| one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> | |
| <p>Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.</p> | |
| </li> | |
| <li> | |
| <p><strong>attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> | |
| <p>Attentions weights after the attention softmax, used to compute the weighted average in the self-attention | |
| heads.</p> | |
| </li> | |
| <li> | |
| <p><strong>image_hidden_states</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) — A <code>torch.FloatTensor</code> of size (batch_size * num_patches, num_images, sequence_length, hidden_size)\`. | |
| image_hidden_states of the model produced by the vision encoder and after projecting the last hidden state.</p> | |
| </li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>transformers.models.llava_next.modeling_llava_next.LlavaNextCausalLMOutputWithPast</code> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),G=new Io({props:{$$slots:{default:[Uo]},$$scope:{ctx:$}}}),W=new Mo({props:{anchor:"transformers.LlavaNextForConditionalGeneration.forward.example",$$slots:{default:[$o]},$$scope:{ctx:$}}}),ie=new zo({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/granitevision.md"}}),{c(){i=d("meta"),M=n(),v=d("p"),m=n(),p(y.$$.fragment),r=n(),p(T.$$.fragment),xe=n(),R=d("p"),R.innerHTML=io,Me=n(),B=d("p"),B.textContent=lo,Ne=n(),V=d("ul"),V.innerHTML=co,we=n(),p(X.$$.fragment),Le=n(),H=d("p"),H.textContent=mo,je=n(),p(A.$$.fragment),Je=n(),S=d("p"),S.innerHTML=po,Ie=n(),p(Y.$$.fragment),ze=n(),N=d("div"),p(D.$$.fragment),Be=n(),le=d("p"),le.innerHTML=go,Ve=n(),de=d("p"),de.innerHTML=ho,Xe=n(),p(k.$$.fragment),Ce=n(),p(Q.$$.fragment),Ue=n(),j=d("div"),p(O.$$.fragment),He=n(),ce=d("p"),ce.innerHTML=fo,Ae=n(),me=d("div"),p(K.$$.fragment),$e=n(),p(ee.$$.fragment),ke=n(),x=d("div"),p(oe.$$.fragment),Se=n(),pe=d("p"),pe.textContent=uo,Ye=n(),ge=d("p"),ge.innerHTML=_o,De=n(),F=d("div"),p(te.$$.fragment),Qe=n(),he=d("p"),he.innerHTML=vo,Oe=n(),P=d("div"),p(ne.$$.fragment),Ke=n(),fe=d("p"),fe.innerHTML=bo,Fe=n(),p(ae.$$.fragment),Pe=n(),w=d("div"),p(se.$$.fragment),eo=n(),ue=d("p"),ue.innerHTML=yo,oo=n(),_e=d("p"),_e.innerHTML=To,to=n(),L=d("div"),p(re.$$.fragment),no=n(),ve=d("p"),ve.innerHTML=xo,ao=n(),p(G.$$.fragment),so=n(),p(W.$$.fragment),Ge=n(),p(ie.$$.fragment),We=n(),ye=d("p"),this.h()},l(e){const o=Jo("svelte-u9bgzb",document.head);i=c(o,"META",{name:!0,content:!0}),o.forEach(t),M=a(e),v=c(e,"P",{}),z(v).forEach(t),m=a(e),g(y.$$.fragment,e),r=a(e),g(T.$$.fragment,e),xe=a(e),R=c(e,"P",{"data-svelte-h":!0}),b(R)!=="svelte-135xyyt"&&(R.innerHTML=io),Me=a(e),B=c(e,"P",{"data-svelte-h":!0}),b(B)!=="svelte-axv494"&&(B.textContent=lo),Ne=a(e),V=c(e,"UL",{"data-svelte-h":!0}),b(V)!=="svelte-1wvs10l"&&(V.innerHTML=co),we=a(e),g(X.$$.fragment,e),Le=a(e),H=c(e,"P",{"data-svelte-h":!0}),b(H)!=="svelte-bwa11d"&&(H.textContent=mo),je=a(e),g(A.$$.fragment,e),Je=a(e),S=c(e,"P",{"data-svelte-h":!0}),b(S)!=="svelte-jh9aly"&&(S.innerHTML=po),Ie=a(e),g(Y.$$.fragment,e),ze=a(e),N=c(e,"DIV",{class:!0});var J=z(N);g(D.$$.fragment,J),Be=a(J),le=c(J,"P",{"data-svelte-h":!0}),b(le)!=="svelte-1jnw21"&&(le.innerHTML=go),Ve=a(J),de=c(J,"P",{"data-svelte-h":!0}),b(de)!=="svelte-fh21ub"&&(de.innerHTML=ho),Xe=a(J),g(k.$$.fragment,J),J.forEach(t),Ce=a(e),g(Q.$$.fragment,e),Ue=a(e),j=c(e,"DIV",{class:!0});var U=z(j);g(O.$$.fragment,U),He=a(U),ce=c(U,"P",{"data-svelte-h":!0}),b(ce)!=="svelte-154s8nf"&&(ce.innerHTML=fo),Ae=a(U),me=c(U,"DIV",{class:!0});var Te=z(me);g(K.$$.fragment,Te),Te.forEach(t),U.forEach(t),$e=a(e),g(ee.$$.fragment,e),ke=a(e),x=c(e,"DIV",{class:!0});var I=z(x);g(oe.$$.fragment,I),Se=a(I),pe=c(I,"P",{"data-svelte-h":!0}),b(pe)!=="svelte-qcp8uy"&&(pe.textContent=uo),Ye=a(I),ge=c(I,"P",{"data-svelte-h":!0}),b(ge)!=="svelte-rbthza"&&(ge.innerHTML=_o),De=a(I),F=c(I,"DIV",{class:!0});var Ee=z(F);g(te.$$.fragment,Ee),Qe=a(Ee),he=c(Ee,"P",{"data-svelte-h":!0}),b(he)!=="svelte-1uf9cni"&&(he.innerHTML=vo),Ee.forEach(t),Oe=a(I),P=c(I,"DIV",{class:!0});var qe=z(P);g(ne.$$.fragment,qe),Ke=a(qe),fe=c(qe,"P",{"data-svelte-h":!0}),b(fe)!=="svelte-1hbbwi4"&&(fe.innerHTML=bo),qe.forEach(t),I.forEach(t),Fe=a(e),g(ae.$$.fragment,e),Pe=a(e),w=c(e,"DIV",{class:!0});var Z=z(w);g(se.$$.fragment,Z),eo=a(Z),ue=c(Z,"P",{"data-svelte-h":!0}),b(ue)!=="svelte-1ibtaz3"&&(ue.innerHTML=yo),oo=a(Z),_e=c(Z,"P",{"data-svelte-h":!0}),b(_e)!=="svelte-hswkmf"&&(_e.innerHTML=To),to=a(Z),L=c(Z,"DIV",{class:!0});var E=z(L);g(re.$$.fragment,E),no=a(E),ve=c(E,"P",{"data-svelte-h":!0}),b(ve)!=="svelte-1gzcm45"&&(ve.innerHTML=xo),ao=a(E),g(G.$$.fragment,E),so=a(E),g(W.$$.fragment,E),E.forEach(t),Z.forEach(t),Ge=a(e),g(ie.$$.fragment,e),We=a(e),ye=c(e,"P",{}),z(ye).forEach(t),this.h()},h(){C(i,"name","hf:doc:metadata"),C(i,"content",Fo),C(N,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),C(me,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),C(j,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),C(F,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),C(P,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),C(x,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),C(L,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),C(w,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(e,o){l(document.head,i),s(e,M,o),s(e,v,o),s(e,m,o),h(y,e,o),s(e,r,o),h(T,e,o),s(e,xe,o),s(e,R,o),s(e,Me,o),s(e,B,o),s(e,Ne,o),s(e,V,o),s(e,we,o),h(X,e,o),s(e,Le,o),s(e,H,o),s(e,je,o),h(A,e,o),s(e,Je,o),s(e,S,o),s(e,Ie,o),h(Y,e,o),s(e,ze,o),s(e,N,o),h(D,N,null),l(N,Be),l(N,le),l(N,Ve),l(N,de),l(N,Xe),h(k,N,null),s(e,Ce,o),h(Q,e,o),s(e,Ue,o),s(e,j,o),h(O,j,null),l(j,He),l(j,ce),l(j,Ae),l(j,me),h(K,me,null),s(e,$e,o),h(ee,e,o),s(e,ke,o),s(e,x,o),h(oe,x,null),l(x,Se),l(x,pe),l(x,Ye),l(x,ge),l(x,De),l(x,F),h(te,F,null),l(F,Qe),l(F,he),l(x,Oe),l(x,P),h(ne,P,null),l(P,Ke),l(P,fe),s(e,Fe,o),h(ae,e,o),s(e,Pe,o),s(e,w,o),h(se,w,null),l(w,eo),l(w,ue),l(w,oo),l(w,_e),l(w,to),l(w,L),h(re,L,null),l(L,no),l(L,ve),l(L,ao),h(G,L,null),l(L,so),h(W,L,null),s(e,Ge,o),h(ie,e,o),s(e,We,o),s(e,ye,o),Ze=!0},p(e,[o]){const J={};o&2&&(J.$$scope={dirty:o,ctx:e}),k.$set(J);const U={};o&2&&(U.$$scope={dirty:o,ctx:e}),G.$set(U);const Te={};o&2&&(Te.$$scope={dirty:o,ctx:e}),W.$set(Te)},i(e){Ze||(f(y.$$.fragment,e),f(T.$$.fragment,e),f(X.$$.fragment,e),f(A.$$.fragment,e),f(Y.$$.fragment,e),f(D.$$.fragment,e),f(k.$$.fragment,e),f(Q.$$.fragment,e),f(O.$$.fragment,e),f(K.$$.fragment,e),f(ee.$$.fragment,e),f(oe.$$.fragment,e),f(te.$$.fragment,e),f(ne.$$.fragment,e),f(ae.$$.fragment,e),f(se.$$.fragment,e),f(re.$$.fragment,e),f(G.$$.fragment,e),f(W.$$.fragment,e),f(ie.$$.fragment,e),Ze=!0)},o(e){u(y.$$.fragment,e),u(T.$$.fragment,e),u(X.$$.fragment,e),u(A.$$.fragment,e),u(Y.$$.fragment,e),u(D.$$.fragment,e),u(k.$$.fragment,e),u(Q.$$.fragment,e),u(O.$$.fragment,e),u(K.$$.fragment,e),u(ee.$$.fragment,e),u(oe.$$.fragment,e),u(te.$$.fragment,e),u(ne.$$.fragment,e),u(ae.$$.fragment,e),u(se.$$.fragment,e),u(re.$$.fragment,e),u(G.$$.fragment,e),u(W.$$.fragment,e),u(ie.$$.fragment,e),Ze=!1},d(e){e&&(t(M),t(v),t(m),t(r),t(xe),t(R),t(Me),t(B),t(Ne),t(V),t(we),t(Le),t(H),t(je),t(Je),t(S),t(Ie),t(ze),t(N),t(Ce),t(Ue),t(j),t($e),t(ke),t(x),t(Fe),t(Pe),t(w),t(Ge),t(We),t(ye)),t(i),_(y,e),_(T,e),_(X,e),_(A,e),_(Y,e),_(D),_(k),_(Q,e),_(O),_(K),_(ee,e),_(oe),_(te),_(ne),_(ae,e),_(se),_(re),_(G),_(W),_(ie,e)}}}const Fo='{"title":"Granite Vision","local":"granite-vision","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"LlavaNextConfig","local":"transformers.LlavaNextConfig","sections":[],"depth":2},{"title":"LlavaNextImageProcessor","local":"transformers.LlavaNextImageProcessor","sections":[],"depth":2},{"title":"LlavaNextProcessor","local":"transformers.LlavaNextProcessor","sections":[],"depth":2},{"title":"LlavaNextForConditionalGeneration","local":"transformers.LlavaNextForConditionalGeneration","sections":[],"depth":2}],"depth":1}';function Po($){return wo(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class Vo extends Lo{constructor(i){super(),jo(this,i,Po,ko,No,{})}}export{Vo as component}; | |
Xet Storage Details
- Size:
- 62.6 kB
- Xet hash:
- 8d5a98cfc6bed4b2a081c4c22366d9b220bbe9a4825a3ab52442a745bd7cdbfc
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.