Buckets:

download
raw
50.9 kB
import"../chunks/DsnmJJEf.js";import{i as L,h as O,H as i,a,D as n,E as A,s as K}from"../chunks/BtE7mKSK.js";import{p as $,o as ee,s as e,f as b,a as p,b as ne,c as o,d as v,r as t,n as l}from"../chunks/jDjavuwI.js";import{E as z}from"../chunks/SrSJA0zO.js";const oe='{"title":"HunyuanVideo-1.5","local":"hunyuanvideo-15","sections":[{"title":"Notes","local":"notes","sections":[],"depth":2},{"title":"HunyuanVideo15Pipeline","local":"diffusers.HunyuanVideo15Pipeline","sections":[],"depth":2},{"title":"HunyuanVideo15ImageToVideoPipeline","local":"diffusers.HunyuanVideo15ImageToVideoPipeline","sections":[],"depth":2},{"title":"HunyuanVideo15PipelineOutput","local":"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput","sections":[],"depth":2}],"depth":1}';var te=v('<meta name="hf:doc:metadata"/>'),N=v("<p>Examples:</p> <!>",1),se=v(`<p></p> <!> <p>HunyuanVideo-1.5 is a lightweight yet powerful video generation model that achieves state-of-the-art visual quality and motion coherence with only 8.3 billion parameters, enabling efficient inference on consumer-grade GPUs. This achievement is built upon several key components, including meticulous data curation, an advanced DiT architecture with selective and sliding tile attention (SSTA), enhanced bilingual understanding through glyph-aware text encoding, progressive pre-training and post-training, and an efficient video super-resolution network. Leveraging these designs, we developed a unified framework capable of high-quality text-to-video and image-to-video generation across multiple durations and resolutions. Extensive experiments demonstrate that this compact and proficient model establishes a new state-of-the-art among open-source models.</p> <p>You can find all the original HunyuanVideo checkpoints under the <a href="https://huggingface.co/tencent" rel="nofollow">Tencent</a> organization.</p> <blockquote class="tip"><p>Click on the HunyuanVideo models in the right sidebar for more examples of video generation tasks.</p> <p>The examples below use a checkpoint from <a href="https://huggingface.co/hunyuanvideo-community" rel="nofollow">hunyuanvideo-community</a> because the weights are stored in a layout compatible with Diffusers.</p></blockquote> <p>The example below demonstrates how to generate a video optimized for memory or inference speed.</p> &#60;hfoptions id="usage">
&#60;hfoption id="memory"> <p>Refer to the <a href="../../optimization/memory">Reduce memory usage</a> guide for more details about the various memory saving techniques.</p> <!> <!> <ul><li><p>HunyuanVideo1.5 use attention masks with variable-length sequences. For best performance, we recommend using an attention backend that handles padding efficiently.</p> <ul><li><strong>H100/H800:</strong> <code>_flash_3_hub</code> or <code>_flash_3_varlen_hub</code></li> <li><strong>A100/A800/RTX 4090:</strong> <code>flash_hub</code> or <code>flash_varlen_hub</code></li> <li><strong>Other GPUs:</strong> <code>sage_hub</code></li></ul></li></ul> <p>Refer to the <a href="../../optimization/attention_backends">Attention backends</a> guide for more details about using a different backend.</p> <!> <ul><li><a href="/docs/diffusers/pr_14230/en/api/pipelines/hunyuan_video15#diffusers.HunyuanVideo15Pipeline">HunyuanVideo15Pipeline</a> use guider and does not take <code>guidance_scale</code> parameter at runtime.</li></ul> <p>You can check the default guider configuration using <code>pipe.guider</code>:</p> <!> <p>To update guider configuration, you can run <code>pipe.guider = pipe.guider.new(...)</code></p> <!> <p>Read more on Guider <a href="../../using-diffusers/guiders">here</a>.</p> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for text-to-video generation using HunyuanVideo1.5.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14230/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods
implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Prepare conditional latents and mask for t2v generation.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for image-to-video generation using HunyuanVideo1.5.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14230/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods
implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Prepare conditional latents and mask for t2v generation.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for HunyuanVideo1.5 pipelines.</p></div> <!> <p></p>`,1);function pe(C,X){$(X,!1),ee(()=>{new URLSearchParams(window.location.search).get("fw")}),L();var T=se();O("s12q98",s=>{var d=te();K(d,"content",oe),p(s,d)});var V=e(b(T),2);i(V,{title:"HunyuanVideo-1.5",local:"hunyuanvideo-15",headingTag:"h1"});var w=e(V,12);a(w,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwQXV0b01vZGVsJTJDJTIwSHVueXVhblZpZGVvMTVQaXBlbGluZSUwQWZyb20lMjBkaWZmdXNlcnMudXRpbHMlMjBpbXBvcnQlMjBleHBvcnRfdG9fdmlkZW8lMEElMEElMEFwaXBlbGluZSUyMCUzRCUyMEh1bnl1YW5WaWRlbzE1UGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMkh1bnl1YW5WaWRlby0xLjUtRGlmZnVzZXJzLTQ4MHBfdDJ2JTIyJTJDJTBBJTIwJTIwJTIwJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiUyQyUwQSklMEElMEElMjMlMjBtb2RlbC1vZmZsb2FkaW5nJTIwYW5kJTIwdGlsaW5nJTBBcGlwZWxpbmUuZW5hYmxlX21vZGVsX2NwdV9vZmZsb2FkKCklMEFwaXBlbGluZS52YWUuZW5hYmxlX3RpbGluZygpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMGZsdWZmeSUyMHRlZGR5JTIwYmVhciUyMHNpdHMlMjBvbiUyMGElMjBiZWQlMjBvZiUyMHNvZnQlMjBwaWxsb3dzJTIwc3Vycm91bmRlZCUyMGJ5JTIwY2hpbGRyZW4ncyUyMHRveXMuJTIyJTBBdmlkZW8lMjAlM0QlMjBwaXBlbGluZShwcm9tcHQlM0Rwcm9tcHQlMkMlMjBudW1fZnJhbWVzJTNENjElMkMlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNEMzApLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMTUp",highlighted:`<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> AutoModel, HunyuanVideo15Pipeline
<span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video
pipeline = HunyuanVideo15Pipeline.from_pretrained(
<span class="hljs-string">&quot;HunyuanVideo-1.5-Diffusers-480p_t2v&quot;</span>,
torch_dtype=torch.bfloat16,
)
<span class="hljs-comment"># model-offloading and tiling</span>
pipeline.enable_model_cpu_offload()
pipeline.vae.enable_tiling()
prompt = <span class="hljs-string">&quot;A fluffy teddy bear sits on a bed of soft pillows surrounded by children&#x27;s toys.&quot;</span>
video = pipeline(prompt=prompt, num_frames=<span class="hljs-number">61</span>, num_inference_steps=<span class="hljs-number">30</span>).frames[<span class="hljs-number">0</span>]
export_to_video(video, <span class="hljs-string">&quot;output.mp4&quot;</span>, fps=<span class="hljs-number">15</span>)`,lang:"py",wrap:!1});var M=e(w,2);i(M,{title:"Notes",local:"notes",headingTag:"h2"});var x=e(M,6);a(x,{code:"cGlwZS50cmFuc2Zvcm1lci5zZXRfYXR0ZW50aW9uX2JhY2tlbmQoJTIyZmxhc2hfaHViJTIyKSUyMCUyMCUyMyUyMG9yJTIweW91ciUyMHByZWZlcnJlZCUyMGJhY2tlbmQ=",highlighted:'pipe.transformer.set_attention_backend(<span class="hljs-string">&quot;flash_hub&quot;</span>) <span class="hljs-comment"># or your preferred backend</span>',lang:"py",wrap:!1});var k=e(x,6);a(k,{code:"cGlwZS5ndWlkZXIlMjAlMEE=",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.guider
ClassifierFreeGuidance {
<span class="hljs-string">&quot;_class_name&quot;</span>: <span class="hljs-string">&quot;ClassifierFreeGuidance&quot;</span>,
<span class="hljs-string">&quot;_diffusers_version&quot;</span>: <span class="hljs-string">&quot;0.36.0.dev0&quot;</span>,
<span class="hljs-string">&quot;enabled&quot;</span>: true,
<span class="hljs-string">&quot;guidance_rescale&quot;</span>: <span class="hljs-number">0.0</span>,
<span class="hljs-string">&quot;guidance_scale&quot;</span>: <span class="hljs-number">6.0</span>,
<span class="hljs-string">&quot;start&quot;</span>: <span class="hljs-number">0.0</span>,
<span class="hljs-string">&quot;stop&quot;</span>: <span class="hljs-number">1.0</span>,
<span class="hljs-string">&quot;use_original_formulation&quot;</span>: false
}
State:
step: <span class="hljs-literal">None</span>
num_inference_steps: <span class="hljs-literal">None</span>
timestep: <span class="hljs-literal">None</span>
count_prepared: <span class="hljs-number">0</span>
enabled: <span class="hljs-literal">True</span>
num_conditions: <span class="hljs-number">2</span>`,lang:"py",wrap:!1});var H=e(k,4);a(H,{code:"cGlwZS5ndWlkZXIlMjAlM0QlMjBwaXBlLmd1aWRlci5uZXcoZ3VpZGFuY2Vfc2NhbGUlM0Q1LjAp",highlighted:'pipe.guider = pipe.guider.new(guidance_scale=<span class="hljs-number">5.0</span>)',lang:"py",wrap:!1});var I=e(H,4);i(I,{title:"HunyuanVideo15Pipeline",local:"diffusers.HunyuanVideo15Pipeline",headingTag:"h2"});var c=e(I,2),J=o(c);n(J,{name:"class diffusers.HunyuanVideo15Pipeline",anchor:"diffusers.HunyuanVideo15Pipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L166",parameters:[{name:"text_encoder",val:": Qwen2_5_VLTextModel"},{name:"tokenizer",val:": Qwen2Tokenizer"},{name:"transformer",val:": HunyuanVideo15Transformer3DModel"},{name:"vae",val:": AutoencoderKLHunyuanVideo15"},{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"},{name:"text_encoder_2",val:": T5EncoderModel"},{name:"tokenizer_2",val:": ByT5Tokenizer"},{name:"guider",val:": ClassifierFreeGuidance"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/hunyuan_video15_transformer_3d#diffusers.HunyuanVideo15Transformer3DModel">HunyuanVideo15Transformer3DModel</a>) &#x2014;
Conditional Transformer (MMDiT) architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.HunyuanVideo15Pipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14230/en/api/schedulers/flow_match_euler_discrete#diffusers.FlowMatchEulerDiscreteScheduler">FlowMatchEulerDiscreteScheduler</a>) &#x2014;
A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents.`,name:"scheduler"},{anchor:"diffusers.HunyuanVideo15Pipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/autoencoder_kl_hunyuan_video15#diffusers.AutoencoderKLHunyuanVideo15">AutoencoderKLHunyuanVideo15</a>) &#x2014;
Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.HunyuanVideo15Pipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>Qwen2.5-VL-7B-Instruct</code>) &#x2014;
<a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a>, specifically the
<a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a> variant.`,name:"text_encoder"},{anchor:"diffusers.HunyuanVideo15Pipeline.tokenizer",description:"<strong>tokenizer</strong> (<code>Qwen2Tokenizer</code>) &#x2014; Tokenizer of class [Qwen2Tokenizer].",name:"tokenizer"},{anchor:"diffusers.HunyuanVideo15Pipeline.text_encoder_2",description:`<strong>text_encoder_2</strong> (<code>T5EncoderModel</code>) &#x2014;
<a href="https://huggingface.co/docs/transformers/en/model_doc/t5#transformers.T5EncoderModel" rel="nofollow">T5EncoderModel</a>
variant.`,name:"text_encoder_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.tokenizer_2",description:"<strong>tokenizer_2</strong> (<code>ByT5Tokenizer</code>) &#x2014; Tokenizer of class [ByT5Tokenizer]",name:"tokenizer_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14230/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>) &#x2014;
[ClassifierFreeGuidance]for classifier free guidance.`,name:"guider"}]});var m=e(J,6),j=o(m);n(j,{name:"__call__",anchor:"diffusers.HunyuanVideo15Pipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L542",parameters:[{name:"prompt",val:": str | list[str] = None"},{name:"negative_prompt",val:": str | list[str] = None"},{name:"height",val:": int | None = None"},{name:"width",val:": int | None = None"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"sigmas",val:": list = None"},{name:"num_videos_per_prompt",val:": int | None = 1"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": str | None = 'np'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": dict[str, typing.Any] | None = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts to guide the image generation. If not defined, one has to pass <code>prompt_embeds</code>
instead.`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts not to guide the image generation. If not defined, one has to pass
<code>negative_prompt_embeds</code> instead.`,name:"negative_prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, <em>optional</em>) &#x2014;
The height in pixels of the generated video.`,name:"height"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, <em>optional</em>) &#x2014;
The width in pixels of the generated video.`,name:"width"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) &#x2014;
The number of frames in the generated video.`,name:"num_frames"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, defaults to <code>50</code>) &#x2014;
The number of denoising steps. More denoising steps usually lead to a higher quality video at the
expense of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.sigmas",description:`<strong>sigmas</strong> (<code>list[float]</code>, <em>optional</em>) &#x2014;
Custom sigmas to use for the denoising process with schedulers which support a <code>sigmas</code> argument in
their <code>set_timesteps</code> method. If not defined, the default behavior when <code>num_inference_steps</code> is passed
will be used.`,name:"sigmas"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) &#x2014;
The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>list[torch.Generator]</code>, <em>optional</em>) &#x2014;
A <a href="https://pytorch.org/docs/stable/generated/torch.Generator.html" rel="nofollow"><code>torch.Generator</code></a> to make
generation deterministic.`,name:"generator"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated noisy latents sampled from a Gaussian distribution, to be used as inputs for video
generation. Can be used to tweak the same generation with different prompts. If not provided, a latents
tensor is generated by sampling using the supplied random <code>generator</code>.`,name:"latents"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings. Can be used to easily tweak text inputs (prompt weighting). If not
provided, text embeddings are generated from the <code>prompt</code> input argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for prompt embeddings.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt
weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input
argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds_mask",description:`<strong>negative_prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for negative prompt embeddings.`,name:"negative_prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings from the second text encoder. Can be used to easily tweak text inputs.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for prompt embeddings from the second text encoder.`,name:"prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds_2",description:`<strong>negative_prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings from the second text encoder.`,name:"negative_prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds_mask_2",description:`<strong>negative_prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for negative prompt embeddings from the second text encoder.`,name:"negative_prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>&quot;np&quot;</code>) &#x2014;
The output format of the generated video. Choose between &#x201C;np&#x201D;, &#x201C;pt&#x201D;, or &#x201C;latent&#x201D;.`,name:"output_type"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) &#x2014;
Whether or not to return a <code>HunyuanVideo15PipelineOutput</code> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) &#x2014;
A kwargs dictionary that if specified is passed along to the <code>AttentionProcessor</code> as defined under
<code>self.processor</code> in
<a href="https://github.com/huggingface/diffusers/blob/main/src/diffusers/models/attention_processor.py" rel="nofollow">diffusers.models.attention_processor</a>.`,name:"attention_kwargs"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>If <code>return_dict</code> is <code>True</code>, <code>HunyuanVideo15PipelineOutput</code> is returned, otherwise a <code>tuple</code> is
returned where the first element is a list with the generated videos.</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><code>~HunyuanVideo15PipelineOutput</code> or <code>tuple</code></p>
`});var Q=e(j,4);z(Q,{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.example",children:(s,d)=>{var r=N(),y=e(b(r),2);a(y,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSHVueXVhblZpZGVvMTVQaXBlbGluZSUwQWZyb20lMjBkaWZmdXNlcnMudXRpbHMlMjBpbXBvcnQlMjBleHBvcnRfdG9fdmlkZW8lMEElMEFtb2RlbF9pZCUyMCUzRCUyMCUyMmh1bnl1YW52aWRlby1jb21tdW5pdHklMkZIdW55dWFuVmlkZW8tMS41LTQ4MHBfdDJ2JTIyJTBBcGlwZSUyMCUzRCUyMEh1bnl1YW5WaWRlbzE1UGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKG1vZGVsX2lkJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5mbG9hdDE2KSUwQXBpcGUudmFlLmVuYWJsZV90aWxpbmcoKSUwQXBpcGUudG8oJTIyY3VkYSUyMiklMEElMEFvdXRwdXQlMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUzRCUyMkElMjBjYXQlMjB3YWxrcyUyMG9uJTIwdGhlJTIwZ3Jhc3MlMkMlMjByZWFsaXN0aWMlMjIlMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8ob3V0cHV0JTJDJTIwJTIyb3V0cHV0Lm1wNCUyMiUyQyUyMGZwcyUzRDE1KQ==",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">import</span> torch
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> HunyuanVideo15Pipeline
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video
<span class="hljs-meta">&gt;&gt;&gt; </span>model_id = <span class="hljs-string">&quot;hunyuanvideo-community/HunyuanVideo-1.5-480p_t2v&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe = HunyuanVideo15Pipeline.from_pretrained(model_id, torch_dtype=torch.float16)
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.vae.enable_tiling()
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>output = pipe(
<span class="hljs-meta">... </span> prompt=<span class="hljs-string">&quot;A cat walks on the grass, realistic&quot;</span>,
<span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>,
<span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>]
<span class="hljs-meta">&gt;&gt;&gt; </span>export_to_video(output, <span class="hljs-string">&quot;output.mp4&quot;</span>, fps=<span class="hljs-number">15</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),t(m);var u=e(m,2),E=o(u);n(E,{name:"encode_prompt",anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L334",parameters:[{name:"prompt",val:": str | list[str]"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"},{name:"batch_size",val:": int = 1"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
prompt to be encoded`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.device",description:`<strong>device</strong> &#x2014; (<code>torch.device</code>):
torch device`,name:"device"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.batch_size",description:`<strong>batch_size</strong> (<code>int</code>) &#x2014;
batch size of prompts, defaults to 1`,name:"batch_size"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>) &#x2014;
number of images that should be generated per prompt`,name:"num_images_per_prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings. If not provided, text embeddings will be generated from <code>prompt</code> input
argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text mask. If not provided, text mask will be generated from <code>prompt</code> input argument.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated glyph text embeddings from ByT5. If not provided, will be generated from <code>prompt</code> input
argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated glyph text mask from ByT5. If not provided, will be generated from <code>prompt</code> input
argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_mask_2"}]}),t(u);var P=e(u,2),R=o(P);n(R,{name:"prepare_cond_latents_and_mask",anchor:"diffusers.HunyuanVideo15Pipeline.prepare_cond_latents_and_mask",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L508",parameters:[{name:"latents",val:""},{name:"dtype",val:": typing.Optional[torch.dtype]"},{name:"device",val:": typing.Optional[torch.device]"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.prepare_cond_latents_and_mask.latents",description:"<strong>latents</strong> &#x2014; Main latents tensor (B, C, F, H, W)",name:"latents"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>(cond_latents_concat, mask_concat) - both are zero tensors for t2v</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p>tuple</p>
`}),l(2),t(P),t(c);var Z=e(c,2);i(Z,{title:"HunyuanVideo15ImageToVideoPipeline",local:"diffusers.HunyuanVideo15ImageToVideoPipeline",headingTag:"h2"});var _=e(Z,2),U=o(_);n(U,{name:"class diffusers.HunyuanVideo15ImageToVideoPipeline",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L193",parameters:[{name:"text_encoder",val:": Qwen2_5_VLTextModel"},{name:"tokenizer",val:": Qwen2Tokenizer"},{name:"transformer",val:": HunyuanVideo15Transformer3DModel"},{name:"vae",val:": AutoencoderKLHunyuanVideo15"},{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"},{name:"text_encoder_2",val:": T5EncoderModel"},{name:"tokenizer_2",val:": ByT5Tokenizer"},{name:"guider",val:": ClassifierFreeGuidance"},{name:"image_encoder",val:": SiglipVisionModel"},{name:"feature_extractor",val:": SiglipImageProcessorPil"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/hunyuan_video15_transformer_3d#diffusers.HunyuanVideo15Transformer3DModel">HunyuanVideo15Transformer3DModel</a>) &#x2014;
Conditional Transformer (MMDiT) architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14230/en/api/schedulers/flow_match_euler_discrete#diffusers.FlowMatchEulerDiscreteScheduler">FlowMatchEulerDiscreteScheduler</a>) &#x2014;
A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents.`,name:"scheduler"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/autoencoder_kl_hunyuan_video15#diffusers.AutoencoderKLHunyuanVideo15">AutoencoderKLHunyuanVideo15</a>) &#x2014;
Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>Qwen2.5-VL-7B-Instruct</code>) &#x2014;
<a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a>, specifically the
<a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a> variant.`,name:"text_encoder"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.tokenizer",description:"<strong>tokenizer</strong> (<code>Qwen2Tokenizer</code>) &#x2014; Tokenizer of class [Qwen2Tokenizer].",name:"tokenizer"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.text_encoder_2",description:`<strong>text_encoder_2</strong> (<code>T5EncoderModel</code>) &#x2014;
<a href="https://huggingface.co/docs/transformers/en/model_doc/t5#transformers.T5EncoderModel" rel="nofollow">T5EncoderModel</a>
variant.`,name:"text_encoder_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.tokenizer_2",description:"<strong>tokenizer_2</strong> (<code>ByT5Tokenizer</code>) &#x2014; Tokenizer of class [ByT5Tokenizer]",name:"tokenizer_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14230/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>) &#x2014;
[ClassifierFreeGuidance]for classifier free guidance.`,name:"guider"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.image_encoder",description:`<strong>image_encoder</strong> (<code>SiglipVisionModel</code>) &#x2014;
<a href="https://huggingface.co/docs/transformers/en/model_doc/siglip#transformers.SiglipVisionModel" rel="nofollow">SiglipVisionModel</a>
variant.`,name:"image_encoder"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.feature_extractor",description:`<strong>feature_extractor</strong> (<code>SiglipImageProcessor</code>) &#x2014;
<a href="https://huggingface.co/docs/transformers/en/model_doc/siglip#transformers.SiglipImageProcessor" rel="nofollow">SiglipImageProcessor</a>
variant.`,name:"feature_extractor"}]});var g=e(U,6),G=o(g);n(G,{name:"__call__",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L650",parameters:[{name:"image",val:": Image"},{name:"prompt",val:": str | list[str] = None"},{name:"negative_prompt",val:": str | list[str] = None"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"sigmas",val:": list = None"},{name:"num_videos_per_prompt",val:": int | None = 1"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": str | None = 'np'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": dict[str, typing.Any] | None = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.image",description:`<strong>image</strong> (<code>PIL.Image.Image</code>) &#x2014;
The input image to condition video generation on.`,name:"image"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts to guide the video generation. If not defined, one has to pass <code>prompt_embeds</code>
instead.`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts not to guide the video generation. If not defined, one has to pass
<code>negative_prompt_embeds</code> instead.`,name:"negative_prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) &#x2014;
The number of frames in the generated video.`,name:"num_frames"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, defaults to <code>50</code>) &#x2014;
The number of denoising steps. More denoising steps usually lead to a higher quality video at the
expense of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.sigmas",description:`<strong>sigmas</strong> (<code>list[float]</code>, <em>optional</em>) &#x2014;
Custom sigmas to use for the denoising process with schedulers which support a <code>sigmas</code> argument in
their <code>set_timesteps</code> method. If not defined, the default behavior when <code>num_inference_steps</code> is passed
will be used.`,name:"sigmas"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) &#x2014;
The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>list[torch.Generator]</code>, <em>optional</em>) &#x2014;
A <a href="https://pytorch.org/docs/stable/generated/torch.Generator.html" rel="nofollow"><code>torch.Generator</code></a> to make
generation deterministic.`,name:"generator"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated noisy latents sampled from a Gaussian distribution, to be used as inputs for video
generation. Can be used to tweak the same generation with different prompts. If not provided, a latents
tensor is generated by sampling using the supplied random <code>generator</code>.`,name:"latents"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings. Can be used to easily tweak text inputs (prompt weighting). If not
provided, text embeddings are generated from the <code>prompt</code> input argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for prompt embeddings.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt
weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input
argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds_mask",description:`<strong>negative_prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for negative prompt embeddings.`,name:"negative_prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings from the second text encoder. Can be used to easily tweak text inputs.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for prompt embeddings from the second text encoder.`,name:"prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds_2",description:`<strong>negative_prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings from the second text encoder.`,name:"negative_prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds_mask_2",description:`<strong>negative_prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated mask for negative prompt embeddings from the second text encoder.`,name:"negative_prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>&quot;np&quot;</code>) &#x2014;
The output format of the generated video. Choose between &#x201C;np&#x201D;, &#x201C;pt&#x201D;, or &#x201C;latent&#x201D;.`,name:"output_type"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) &#x2014;
Whether or not to return a <code>HunyuanVideo15PipelineOutput</code> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) &#x2014;
A kwargs dictionary that if specified is passed along to the <code>AttentionProcessor</code> as defined under
<code>self.processor</code> in
<a href="https://github.com/huggingface/diffusers/blob/main/src/diffusers/models/attention_processor.py" rel="nofollow">diffusers.models.attention_processor</a>.`,name:"attention_kwargs"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>If <code>return_dict</code> is <code>True</code>, <code>HunyuanVideo15PipelineOutput</code> is returned, otherwise a <code>tuple</code> is
returned where the first element is a list with the generated videos.</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><code>~HunyuanVideo15PipelineOutput</code> or <code>tuple</code></p>
`});var S=e(G,4);z(S,{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.example",children:(s,d)=>{var r=N(),y=e(b(r),2);a(y,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSHVueXVhblZpZGVvMTVJbWFnZVRvVmlkZW9QaXBlbGluZSUwQWZyb20lMjBkaWZmdXNlcnMudXRpbHMlMjBpbXBvcnQlMjBleHBvcnRfdG9fdmlkZW8lMEElMEFtb2RlbF9pZCUyMCUzRCUyMCUyMmh1bnl1YW52aWRlby1jb21tdW5pdHklMkZIdW55dWFuVmlkZW8tMS41LTQ4MHBfaTJ2JTIyJTBBcGlwZSUyMCUzRCUyMEh1bnl1YW5WaWRlbzE1SW1hZ2VUb1ZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKG1vZGVsX2lkJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5mbG9hdDE2KSUwQXBpcGUudmFlLmVuYWJsZV90aWxpbmcoKSUwQXBpcGUudG8oJTIyY3VkYSUyMiklMEElMEFpbWFnZSUyMCUzRCUyMGxvYWRfaW1hZ2UoJTIyaHR0cHMlM0ElMkYlMkZodWdnaW5nZmFjZS5jbyUyRmRhdGFzZXRzJTJGWWlZaVh1JTJGdGVzdGluZy1pbWFnZXMlMkZyZXNvbHZlJTJGbWFpbiUyRndhbl9pMnZfaW5wdXQuSlBHJTIyKSUwQSUwQW91dHB1dCUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwcHJvbXB0JTNEJTIyU3VtbWVyJTIwYmVhY2glMjB2YWNhdGlvbiUyMHN0eWxlJTJDJTIwYSUyMHdoaXRlJTIwY2F0JTIwd2VhcmluZyUyMHN1bmdsYXNzZXMlMjBzaXRzJTIwb24lMjBhJTIwc3VyZmJvYXJkLiUyMFRoZSUyMGZsdWZmeS1mdXJyZWQlMjBmZWxpbmUlMjBnYXplcyUyMGRpcmVjdGx5JTIwYXQlMjB0aGUlMjBjYW1lcmElMjB3aXRoJTIwYSUyMHJlbGF4ZWQlMjBleHByZXNzaW9uLiUyMEJsdXJyZWQlMjBiZWFjaCUyMHNjZW5lcnklMjBmb3JtcyUyMHRoZSUyMGJhY2tncm91bmQlMjBmZWF0dXJpbmclMjBjcnlzdGFsLWNsZWFyJTIwd2F0ZXJzJTJDJTIwZGlzdGFudCUyMGdyZWVuJTIwaGlsbHMlMkMlMjBhbmQlMjBhJTIwYmx1ZSUyMHNreSUyMGRvdHRlZCUyMHdpdGglMjB3aGl0ZSUyMGNsb3Vkcy4lMjBUaGUlMjBjYXQlMjBhc3N1bWVzJTIwYSUyMG5hdHVyYWxseSUyMHJlbGF4ZWQlMjBwb3N0dXJlJTJDJTIwYXMlMjBpZiUyMHNhdm9yaW5nJTIwdGhlJTIwc2VhJTIwYnJlZXplJTIwYW5kJTIwd2FybSUyMHN1bmxpZ2h0LiUyMEElMjBjbG9zZS11cCUyMHNob3QlMjBoaWdobGlnaHRzJTIwdGhlJTIwZmVsaW5lJ3MlMjBpbnRyaWNhdGUlMjBkZXRhaWxzJTIwYW5kJTIwdGhlJTIwcmVmcmVzaGluZyUyMGF0bW9zcGhlcmUlMjBvZiUyMHRoZSUyMHNlYXNpZGUuJTIyJTJDJTBBJTIwJTIwJTIwJTIwaW1hZ2UlM0RpbWFnZSUyQyUwQSUyMCUyMCUyMCUyMG51bV9pbmZlcmVuY2Vfc3RlcHMlM0Q1MCUyQyUwQSkuZnJhbWVzJTVCMCU1RCUwQWV4cG9ydF90b192aWRlbyhvdXRwdXQlMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">import</span> torch
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> HunyuanVideo15ImageToVideoPipeline
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video
<span class="hljs-meta">&gt;&gt;&gt; </span>model_id = <span class="hljs-string">&quot;hunyuanvideo-community/HunyuanVideo-1.5-480p_i2v&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe = HunyuanVideo15ImageToVideoPipeline.from_pretrained(model_id, torch_dtype=torch.float16)
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.vae.enable_tiling()
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>image = load_image(<span class="hljs-string">&quot;https://huggingface.co/datasets/YiYiXu/testing-images/resolve/main/wan_i2v_input.JPG&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>output = pipe(
<span class="hljs-meta">... </span> prompt=<span class="hljs-string">&quot;Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline&#x27;s intricate details and the refreshing atmosphere of the seaside.&quot;</span>,
<span class="hljs-meta">... </span> image=image,
<span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>,
<span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>]
<span class="hljs-meta">&gt;&gt;&gt; </span>export_to_video(output, <span class="hljs-string">&quot;output.mp4&quot;</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),t(g);var h=e(g,2),F=o(h);n(F,{name:"encode_prompt",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L422",parameters:[{name:"prompt",val:": str | list[str]"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"},{name:"batch_size",val:": int = 1"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) &#x2014;
prompt to be encoded`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.device",description:`<strong>device</strong> &#x2014; (<code>torch.device</code>):
torch device`,name:"device"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.batch_size",description:`<strong>batch_size</strong> (<code>int</code>) &#x2014;
batch size of prompts, defaults to 1`,name:"batch_size"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>) &#x2014;
number of images that should be generated per prompt`,name:"num_images_per_prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings. If not provided, text embeddings will be generated from <code>prompt</code> input
argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text mask. If not provided, text mask will be generated from <code>prompt</code> input argument.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated glyph text embeddings from ByT5. If not provided, will be generated from <code>prompt</code> input
argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated glyph text mask from ByT5. If not provided, will be generated from <code>prompt</code> input
argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_mask_2"}]}),t(h);var W=e(h,2),Y=o(W);n(Y,{name:"prepare_cond_latents_and_mask",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.prepare_cond_latents_and_mask",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L594",parameters:[{name:"latents",val:": Tensor"},{name:"image",val:": Image"},{name:"batch_size",val:": int"},{name:"height",val:": int"},{name:"width",val:": int"},{name:"dtype",val:": dtype"},{name:"device",val:": device"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.prepare_cond_latents_and_mask.latents",description:"<strong>latents</strong> &#x2014; Main latents tensor (B, C, F, H, W)",name:"latents"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>(cond_latents_concat, mask_concat) - both are zero tensors for t2v</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p>tuple</p>
`}),l(2),t(W),t(_);var B=e(_,2);i(B,{title:"HunyuanVideo15PipelineOutput",local:"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput",headingTag:"h2"});var f=e(B,2),q=o(f);n(q,{name:"class diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput",anchor:"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_output.py#L9",parameters:[{name:"frames",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput.frames",description:`<strong>frames</strong> (<code>torch.Tensor</code>, <code>np.ndarray</code>, or list[list[PIL.Image.Image]]) &#x2014;
List of video outputs - It can be a nested list of length <code>batch_size,</code> with each sub-list containing
denoised PIL image sequences of length <code>num_frames.</code> It can also be a NumPy array or Torch tensor of shape
<code>(batch_size, num_frames, channels, height, width)</code>.`,name:"frames"}]}),l(2),t(f);var D=e(f,2);A(D,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/hunyuan_video15.md"}),l(2),p(C,T),ne()}export{pe as component};

Xet Storage Details

Size:
50.9 kB
·
Xet hash:
77023c77471b920eb291babed98fa53e5cea774ee525403c550f437d110c55f3

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.