Buckets:
| import"../chunks/DsnmJJEf.js";import{i as M,h as z,C as W,H as T,a as I,D as o,E as j,s as F}from"../chunks/BtE7mKSK.js";import{p as O,o as E,s as e,f as Z,a as L,b as G,c as t,d as x,n as r,r as n}from"../chunks/jDjavuwI.js";const R='{"title":"AutoencoderKLLTX2Video","local":"autoencoderklltx2video","sections":[{"title":"AutoencoderKLLTX2Video","local":"diffusers.AutoencoderKLLTX2Video","sections":[],"depth":2}],"depth":1}';var J=x('<meta name="hf:doc:metadata"/>'),B=x(`<p></p> <!> <!> <p>The 3D variational autoencoder (VAE) model with KL loss used in <a href="https://huggingface.co/Lightricks/LTX-2" rel="nofollow">LTX-2</a> was introduced by Lightricks.</p> <p>The model can be loaded with the following code snippet.</p> <!> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>A VAE model with KL loss for encoding images into latents and decoding latent representations into images. Used in <a href="https://huggingface.co/Lightricks/LTX-2" rel="nofollow">LTX-2</a>.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14178/en/api/models/overview#diffusers.ModelMixin">ModelMixin</a>. Check the superclass documentation for it’s generic methods implemented | |
| for all models (such as downloading or saving).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Decode a batch of images.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encode a batch of images into latents.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Enable tiled VAE decoding. When this option is enabled, the VAE will split the input tensor into tiles to | |
| compute decoding and encoding in several steps. This is useful for saving a large amount of memory and to allow | |
| processing larger images.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Decode a batch of images using a tiled decoder.</p></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encode a batch of images using a tiled encoder.</p></div></div> <!> <p></p>`,1);function C(w,V){O(V,!1),E(()=>{new URLSearchParams(window.location.search).get("fw")}),M();var u=B();z("t0b3it",b=>{var v=J();F(v,"content",R),L(b,v)});var p=e(Z(u),2);W(p,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var m=e(p,2);T(m,{title:"AutoencoderKLLTX2Video",local:"autoencoderklltx2video",headingTag:"h1"});var h=e(m,6);I(h,{code:"ZnJvbSUyMGRpZmZ1c2VycyUyMGltcG9ydCUyMEF1dG9lbmNvZGVyS0xMVFgyVmlkZW8lMEElMEF2YWUlMjAlM0QlMjBBdXRvZW5jb2RlcktMTFRYMlZpZGVvLmZyb21fcHJldHJhaW5lZCglMjJMaWdodHJpY2tzJTJGTFRYLTIlMjIlMkMlMjBzdWJmb2xkZXIlM0QlMjJ2YWUlMjIlMkMlMjB0b3JjaF9kdHlwZSUzRHRvcmNoLmZsb2F0MzIpLnRvKCUyMmN1ZGElMjIp",highlighted:`<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> AutoencoderKLLTX2Video | |
| vae = AutoencoderKLLTX2Video.from_pretrained(<span class="hljs-string">"Lightricks/LTX-2"</span>, subfolder=<span class="hljs-string">"vae"</span>, torch_dtype=torch.float32).to(<span class="hljs-string">"cuda"</span>)`,lang:"python",wrap:!1});var f=e(h,2);T(f,{title:"AutoencoderKLLTX2Video",local:"diffusers.AutoencoderKLLTX2Video",headingTag:"h2"});var a=e(f,2),_=t(a);o(_,{name:"class diffusers.AutoencoderKLLTX2Video",anchor:"diffusers.AutoencoderKLLTX2Video",source:"https://github.com/huggingface/diffusers/blob/vr_14178/src/diffusers/models/autoencoders/autoencoder_kl_ltx2.py#L1025",parameters:[{name:"in_channels",val:": int = 3"},{name:"out_channels",val:": int = 3"},{name:"latent_channels",val:": int = 128"},{name:"block_out_channels",val:": tuple = (256, 512, 1024, 2048)"},{name:"down_block_types",val:": tuple = ('LTX2VideoDownBlock3D', 'LTX2VideoDownBlock3D', 'LTX2VideoDownBlock3D', 'LTX2VideoDownBlock3D')"},{name:"decoder_block_out_channels",val:": tuple = (256, 512, 1024)"},{name:"layers_per_block",val:": tuple = (4, 6, 6, 2, 2)"},{name:"decoder_layers_per_block",val:": tuple = (5, 5, 5, 5)"},{name:"spatio_temporal_scaling",val:": bool | tuple[bool, ...] = (True, True, True, True)"},{name:"decoder_spatio_temporal_scaling",val:": bool | tuple[bool, ...] = (True, True, True)"},{name:"decoder_inject_noise",val:": bool | tuple[bool, ...] = (False, False, False, False)"},{name:"downsample_type",val:": tuple = ('spatial', 'temporal', 'spatiotemporal', 'spatiotemporal')"},{name:"upsample_type",val:": tuple = ('spatiotemporal', 'spatiotemporal', 'spatiotemporal')"},{name:"upsample_residual",val:": bool | tuple[bool, ...] = (True, True, True)"},{name:"upsample_factor",val:": tuple = (2, 2, 2)"},{name:"timestep_conditioning",val:": bool = False"},{name:"patch_size",val:": int = 4"},{name:"patch_size_t",val:": int = 1"},{name:"resnet_norm_eps",val:": float = 1e-06"},{name:"scaling_factor",val:": float = 1.0"},{name:"encoder_causal",val:": bool = True"},{name:"decoder_causal",val:": bool = True"},{name:"encoder_spatial_padding_mode",val:": str = 'zeros'"},{name:"decoder_spatial_padding_mode",val:": str = 'reflect'"},{name:"spatial_compression_ratio",val:": int = None"},{name:"temporal_compression_ratio",val:": int = None"}],parametersDescription:[{anchor:"diffusers.AutoencoderKLLTX2Video.in_channels",description:`<strong>in_channels</strong> (<code>int</code>, defaults to <code>3</code>) — | |
| Number of input channels.`,name:"in_channels"},{anchor:"diffusers.AutoencoderKLLTX2Video.out_channels",description:`<strong>out_channels</strong> (<code>int</code>, defaults to <code>3</code>) — | |
| Number of output channels.`,name:"out_channels"},{anchor:"diffusers.AutoencoderKLLTX2Video.latent_channels",description:`<strong>latent_channels</strong> (<code>int</code>, defaults to <code>128</code>) — | |
| Number of latent channels.`,name:"latent_channels"},{anchor:"diffusers.AutoencoderKLLTX2Video.block_out_channels",description:`<strong>block_out_channels</strong> (<code>tuple[int, ...]</code>, defaults to <code>(128, 256, 512, 512)</code>) — | |
| The number of output channels for each block.`,name:"block_out_channels"},{anchor:"diffusers.AutoencoderKLLTX2Video.spatio_temporal_scaling",description:"<strong>spatio_temporal_scaling</strong> (<code>tuple[bool, ...], defaults to </code>(True, True, True, False)` —\nWhether a block should contain spatio-temporal downscaling or not.",name:"spatio_temporal_scaling"},{anchor:"diffusers.AutoencoderKLLTX2Video.layers_per_block",description:`<strong>layers_per_block</strong> (<code>tuple[int, ...]</code>, defaults to <code>(4, 3, 3, 3, 4)</code>) — | |
| The number of layers per block.`,name:"layers_per_block"},{anchor:"diffusers.AutoencoderKLLTX2Video.patch_size",description:`<strong>patch_size</strong> (<code>int</code>, defaults to <code>4</code>) — | |
| The size of spatial patches.`,name:"patch_size"},{anchor:"diffusers.AutoencoderKLLTX2Video.patch_size_t",description:`<strong>patch_size_t</strong> (<code>int</code>, defaults to <code>1</code>) — | |
| The size of temporal patches.`,name:"patch_size_t"},{anchor:"diffusers.AutoencoderKLLTX2Video.resnet_norm_eps",description:`<strong>resnet_norm_eps</strong> (<code>float</code>, defaults to <code>1e-6</code>) — | |
| Epsilon value for ResNet normalization layers.`,name:"resnet_norm_eps"},{anchor:"diffusers.AutoencoderKLLTX2Video.scaling_factor",description:`<strong>scaling_factor</strong> (<code>float</code>, <em>optional</em>, defaults to <code>1.0</code>) — | |
| The component-wise standard deviation of the trained latent space computed using the first batch of the | |
| training set. This is used to scale the latent space to have unit variance when training the diffusion | |
| model. The latents are scaled with the formula <code>z = z * scaling_factor</code> before being passed to the | |
| diffusion model. When decoding, the latents are scaled back to the original scale with the formula: <code>z = 1 / scaling_factor * z</code>. For more details, refer to sections 4.3.2 and D.1 of the <a href="https://huggingface.co/papers/2112.10752" rel="nofollow">High-Resolution Image | |
| Synthesis with Latent Diffusion Models</a> paper.`,name:"scaling_factor"},{anchor:"diffusers.AutoencoderKLLTX2Video.encoder_causal",description:`<strong>encoder_causal</strong> (<code>bool</code>, defaults to <code>True</code>) — | |
| Whether the encoder should behave causally (future frames depend only on past frames) or not.`,name:"encoder_causal"},{anchor:"diffusers.AutoencoderKLLTX2Video.decoder_causal",description:`<strong>decoder_causal</strong> (<code>bool</code>, defaults to <code>False</code>) — | |
| Whether the decoder should behave causally (future frames depend only on past frames) or not.`,name:"decoder_causal"}]});var d=e(_,6),y=t(d);o(y,{name:"decode",anchor:"diffusers.AutoencoderKLLTX2Video.decode",source:"https://github.com/huggingface/diffusers/blob/vr_14178/src/diffusers/models/autoencoders/autoencoder_kl_ltx2.py#L1291",parameters:[{name:"z",val:": Tensor"},{name:"temb",val:": typing.Optional[torch.Tensor] = None"},{name:"causal",val:": bool | None = None"},{name:"return_dict",val:": bool = True"}],parametersDescription:[{anchor:"diffusers.AutoencoderKLLTX2Video.decode.z",description:"<strong>z</strong> (<code>torch.Tensor</code>) — Input batch of latent vectors.",name:"z"},{anchor:"diffusers.AutoencoderKLLTX2Video.decode.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to return a <code>~models.vae.DecoderOutput</code> instead of a plain tuple.`,name:"return_dict"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If return_dict is True, a <code>~models.vae.DecoderOutput</code> is returned, otherwise a plain <code>tuple</code> is | |
| returned.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>~models.vae.DecoderOutput</code> or <code>tuple</code></p> | |
| `}),r(2),n(d);var s=e(d,2),X=t(s);o(X,{name:"encode",anchor:"diffusers.AutoencoderKLLTX2Video.encode",source:"https://github.com/huggingface/diffusers/blob/vr_14178/src/diffusers/models/autoencoders/autoencoder_kl_ltx2.py#L1239",parameters:[{name:"x",val:": Tensor"},{name:"causal",val:": bool | None = None"},{name:"return_dict",val:": bool = True"}],parametersDescription:[{anchor:"diffusers.AutoencoderKLLTX2Video.encode.x",description:"<strong>x</strong> (<code>torch.Tensor</code>) — Input batch of images.",name:"x"},{anchor:"diffusers.AutoencoderKLLTX2Video.encode.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to return a <code>~models.autoencoder_kl.AutoencoderKLOutput</code> instead of a plain tuple.`,name:"return_dict"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The latent representations of the encoded videos. If <code>return_dict</code> is True, a | |
| <code>~models.autoencoder_kl.AutoencoderKLOutput</code> is returned, otherwise a plain <code>tuple</code> is returned.</p> | |
| `}),r(2),n(s);var i=e(s,2),A=t(i);o(A,{name:"enable_tiling",anchor:"diffusers.AutoencoderKLLTX2Video.enable_tiling",source:"https://github.com/huggingface/diffusers/blob/vr_14178/src/diffusers/models/autoencoders/autoencoder_kl_ltx2.py#L1192",parameters:[{name:"tile_sample_min_height",val:": int | None = None"},{name:"tile_sample_min_width",val:": int | None = None"},{name:"tile_sample_min_num_frames",val:": int | None = None"},{name:"tile_sample_stride_height",val:": float | None = None"},{name:"tile_sample_stride_width",val:": float | None = None"},{name:"tile_sample_stride_num_frames",val:": float | None = None"}],parametersDescription:[{anchor:"diffusers.AutoencoderKLLTX2Video.enable_tiling.tile_sample_min_height",description:`<strong>tile_sample_min_height</strong> (<code>int</code>, <em>optional</em>) — | |
| The minimum height required for a sample to be separated into tiles across the height dimension.`,name:"tile_sample_min_height"},{anchor:"diffusers.AutoencoderKLLTX2Video.enable_tiling.tile_sample_min_width",description:`<strong>tile_sample_min_width</strong> (<code>int</code>, <em>optional</em>) — | |
| The minimum width required for a sample to be separated into tiles across the width dimension.`,name:"tile_sample_min_width"},{anchor:"diffusers.AutoencoderKLLTX2Video.enable_tiling.tile_sample_stride_height",description:`<strong>tile_sample_stride_height</strong> (<code>int</code>, <em>optional</em>) — | |
| The minimum amount of overlap between two consecutive vertical tiles. This is to ensure that there are | |
| no tiling artifacts produced across the height dimension.`,name:"tile_sample_stride_height"},{anchor:"diffusers.AutoencoderKLLTX2Video.enable_tiling.tile_sample_stride_width",description:`<strong>tile_sample_stride_width</strong> (<code>int</code>, <em>optional</em>) — | |
| The stride between two consecutive horizontal tiles. This is to ensure that there are no tiling | |
| artifacts produced across the width dimension.`,name:"tile_sample_stride_width"}]}),r(2),n(i);var c=e(i,2),K=t(c);o(K,{name:"forward",anchor:"diffusers.AutoencoderKLLTX2Video.forward",source:"https://github.com/huggingface/diffusers/blob/vr_14178/src/diffusers/models/autoencoders/autoencoder_kl_ltx2.py#L1535",parameters:[{name:"sample",val:": Tensor"},{name:"temb",val:": typing.Optional[torch.Tensor] = None"},{name:"sample_posterior",val:": bool = False"},{name:"encoder_causal",val:": bool | None = None"},{name:"decoder_causal",val:": bool | None = None"},{name:"return_dict",val:": bool = True"},{name:"generator",val:": typing.Optional[torch.Generator] = None"}],parametersDescription:[{anchor:"diffusers.AutoencoderKLLTX2Video.forward.sample",description:"<strong>sample</strong> (<code>torch.Tensor</code>) — Input sample.",name:"sample"},{anchor:"diffusers.AutoencoderKLLTX2Video.forward.temb",description:`<strong>temb</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Optional timestep embedding tensor used to condition the decoder.`,name:"temb"},{anchor:"diffusers.AutoencoderKLLTX2Video.forward.sample_posterior",description:`<strong>sample_posterior</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to sample from the posterior.`,name:"sample_posterior"},{anchor:"diffusers.AutoencoderKLLTX2Video.forward.encoder_causal",description:`<strong>encoder_causal</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether the encoder should use causal convolutions. If <code>None</code>, falls back to the model default.`,name:"encoder_causal"},{anchor:"diffusers.AutoencoderKLLTX2Video.forward.decoder_causal",description:`<strong>decoder_causal</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether the decoder should use causal convolutions. If <code>None</code>, falls back to the model default.`,name:"decoder_causal"},{anchor:"diffusers.AutoencoderKLLTX2Video.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to return a <code>DecoderOutput</code> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.AutoencoderKLLTX2Video.forward.generator",description:`<strong>generator</strong> (<code>torch.Generator</code>, <em>optional</em>) — | |
| A <a href="https://pytorch.org/docs/stable/generated/torch.Generator.html" rel="nofollow"><code>torch.Generator</code></a> to make sampling | |
| deterministic.`,name:"generator"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If <code>return_dict</code> is True, a <code>~models.vae.DecoderOutput</code> is returned, otherwise a plain <code>tuple</code> is | |
| returned.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>~models.vae.DecoderOutput</code> or <code>tuple</code></p> | |
| `}),n(c);var l=e(c,2),k=t(l);o(k,{name:"tiled_decode",anchor:"diffusers.AutoencoderKLLTX2Video.tiled_decode",source:"https://github.com/huggingface/diffusers/blob/vr_14178/src/diffusers/models/autoencoders/autoencoder_kl_ltx2.py#L1405",parameters:[{name:"z",val:": Tensor"},{name:"temb",val:": typing.Optional[torch.Tensor]"},{name:"causal",val:": bool | None = None"},{name:"return_dict",val:": bool = True"}],parametersDescription:[{anchor:"diffusers.AutoencoderKLLTX2Video.tiled_decode.z",description:"<strong>z</strong> (<code>torch.Tensor</code>) — Input batch of latent vectors.",name:"z"},{anchor:"diffusers.AutoencoderKLLTX2Video.tiled_decode.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to return a <code>~models.vae.DecoderOutput</code> instead of a plain tuple.`,name:"return_dict"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If return_dict is True, a <code>~models.vae.DecoderOutput</code> is returned, otherwise a plain <code>tuple</code> is | |
| returned.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>~models.vae.DecoderOutput</code> or <code>tuple</code></p> | |
| `}),r(2),n(l);var g=e(l,2),N=t(g);o(N,{name:"tiled_encode",anchor:"diffusers.AutoencoderKLLTX2Video.tiled_encode",source:"https://github.com/huggingface/diffusers/blob/vr_14178/src/diffusers/models/autoencoders/autoencoder_kl_ltx2.py#L1353",parameters:[{name:"x",val:": Tensor"},{name:"causal",val:": bool | None = None"}],parametersDescription:[{anchor:"diffusers.AutoencoderKLLTX2Video.tiled_encode.x",description:"<strong>x</strong> (<code>torch.Tensor</code>) — Input batch of videos.",name:"x"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The latent representation of the encoded videos.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>torch.Tensor</code></p> | |
| `}),r(2),n(g),n(a);var D=e(a,2);j(D,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/models/autoencoderkl_ltx_2.md"}),r(2),L(w,u),G()}export{C as component}; | |
Xet Storage Details
- Size:
- 18.5 kB
- Xet hash:
- 9020a3be9e853cc802b29b786e0ccff87315feb09e10524f7b7a754362d9d740
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.