Buckets:
| import"../chunks/DsnmJJEf.js";import{i as L,h as O,H as i,a,D as n,E as A,s as K}from"../chunks/BtE7mKSK.js";import{p as $,o as ee,s as e,f as b,a as p,b as ne,c as o,d as v,r as t,n as l}from"../chunks/jDjavuwI.js";import{E as z}from"../chunks/SrSJA0zO.js";const oe='{"title":"HunyuanVideo-1.5","local":"hunyuanvideo-15","sections":[{"title":"Notes","local":"notes","sections":[],"depth":2},{"title":"HunyuanVideo15Pipeline","local":"diffusers.HunyuanVideo15Pipeline","sections":[],"depth":2},{"title":"HunyuanVideo15ImageToVideoPipeline","local":"diffusers.HunyuanVideo15ImageToVideoPipeline","sections":[],"depth":2},{"title":"HunyuanVideo15PipelineOutput","local":"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput","sections":[],"depth":2}],"depth":1}';var te=v('<meta name="hf:doc:metadata"/>'),N=v("<p>Examples:</p> <!>",1),se=v(`<p></p> <!> <p>HunyuanVideo-1.5 is a lightweight yet powerful video generation model that achieves state-of-the-art visual quality and motion coherence with only 8.3 billion parameters, enabling efficient inference on consumer-grade GPUs. This achievement is built upon several key components, including meticulous data curation, an advanced DiT architecture with selective and sliding tile attention (SSTA), enhanced bilingual understanding through glyph-aware text encoding, progressive pre-training and post-training, and an efficient video super-resolution network. Leveraging these designs, we developed a unified framework capable of high-quality text-to-video and image-to-video generation across multiple durations and resolutions. Extensive experiments demonstrate that this compact and proficient model establishes a new state-of-the-art among open-source models.</p> <p>You can find all the original HunyuanVideo checkpoints under the <a href="https://huggingface.co/tencent" rel="nofollow">Tencent</a> organization.</p> <blockquote class="tip"><p>Click on the HunyuanVideo models in the right sidebar for more examples of video generation tasks.</p> <p>The examples below use a checkpoint from <a href="https://huggingface.co/hunyuanvideo-community" rel="nofollow">hunyuanvideo-community</a> because the weights are stored in a layout compatible with Diffusers.</p></blockquote> <p>The example below demonstrates how to generate a video optimized for memory or inference speed.</p> <hfoptions id="usage"> | |
| <hfoption id="memory"> <p>Refer to the <a href="../../optimization/memory">Reduce memory usage</a> guide for more details about the various memory saving techniques.</p> <!> <!> <ul><li><p>HunyuanVideo1.5 use attention masks with variable-length sequences. For best performance, we recommend using an attention backend that handles padding efficiently.</p> <ul><li><strong>H100/H800:</strong> <code>_flash_3_hub</code> or <code>_flash_3_varlen_hub</code></li> <li><strong>A100/A800/RTX 4090:</strong> <code>flash_hub</code> or <code>flash_varlen_hub</code></li> <li><strong>Other GPUs:</strong> <code>sage_hub</code></li></ul></li></ul> <p>Refer to the <a href="../../optimization/attention_backends">Attention backends</a> guide for more details about using a different backend.</p> <!> <ul><li><a href="/docs/diffusers/pr_14230/en/api/pipelines/hunyuan_video15#diffusers.HunyuanVideo15Pipeline">HunyuanVideo15Pipeline</a> use guider and does not take <code>guidance_scale</code> parameter at runtime.</li></ul> <p>You can check the default guider configuration using <code>pipe.guider</code>:</p> <!> <p>To update guider configuration, you can run <code>pipe.guider = pipe.guider.new(...)</code></p> <!> <p>Read more on Guider <a href="../../using-diffusers/guiders">here</a>.</p> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for text-to-video generation using HunyuanVideo1.5.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14230/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods | |
| implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Prepare conditional latents and mask for t2v generation.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for image-to-video generation using HunyuanVideo1.5.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14230/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods | |
| implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Prepare conditional latents and mask for t2v generation.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for HunyuanVideo1.5 pipelines.</p></div> <!> <p></p>`,1);function pe(C,X){$(X,!1),ee(()=>{new URLSearchParams(window.location.search).get("fw")}),L();var T=se();O("s12q98",s=>{var d=te();K(d,"content",oe),p(s,d)});var V=e(b(T),2);i(V,{title:"HunyuanVideo-1.5",local:"hunyuanvideo-15",headingTag:"h1"});var w=e(V,12);a(w,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwQXV0b01vZGVsJTJDJTIwSHVueXVhblZpZGVvMTVQaXBlbGluZSUwQWZyb20lMjBkaWZmdXNlcnMudXRpbHMlMjBpbXBvcnQlMjBleHBvcnRfdG9fdmlkZW8lMEElMEElMEFwaXBlbGluZSUyMCUzRCUyMEh1bnl1YW5WaWRlbzE1UGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMkh1bnl1YW5WaWRlby0xLjUtRGlmZnVzZXJzLTQ4MHBfdDJ2JTIyJTJDJTBBJTIwJTIwJTIwJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiUyQyUwQSklMEElMEElMjMlMjBtb2RlbC1vZmZsb2FkaW5nJTIwYW5kJTIwdGlsaW5nJTBBcGlwZWxpbmUuZW5hYmxlX21vZGVsX2NwdV9vZmZsb2FkKCklMEFwaXBlbGluZS52YWUuZW5hYmxlX3RpbGluZygpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMGZsdWZmeSUyMHRlZGR5JTIwYmVhciUyMHNpdHMlMjBvbiUyMGElMjBiZWQlMjBvZiUyMHNvZnQlMjBwaWxsb3dzJTIwc3Vycm91bmRlZCUyMGJ5JTIwY2hpbGRyZW4ncyUyMHRveXMuJTIyJTBBdmlkZW8lMjAlM0QlMjBwaXBlbGluZShwcm9tcHQlM0Rwcm9tcHQlMkMlMjBudW1fZnJhbWVzJTNENjElMkMlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNEMzApLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMTUp",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> AutoModel, HunyuanVideo15Pipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| pipeline = HunyuanVideo15Pipeline.from_pretrained( | |
| <span class="hljs-string">"HunyuanVideo-1.5-Diffusers-480p_t2v"</span>, | |
| torch_dtype=torch.bfloat16, | |
| ) | |
| <span class="hljs-comment"># model-offloading and tiling</span> | |
| pipeline.enable_model_cpu_offload() | |
| pipeline.vae.enable_tiling() | |
| prompt = <span class="hljs-string">"A fluffy teddy bear sits on a bed of soft pillows surrounded by children's toys."</span> | |
| video = pipeline(prompt=prompt, num_frames=<span class="hljs-number">61</span>, num_inference_steps=<span class="hljs-number">30</span>).frames[<span class="hljs-number">0</span>] | |
| export_to_video(video, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">15</span>)`,lang:"py",wrap:!1});var M=e(w,2);i(M,{title:"Notes",local:"notes",headingTag:"h2"});var x=e(M,6);a(x,{code:"cGlwZS50cmFuc2Zvcm1lci5zZXRfYXR0ZW50aW9uX2JhY2tlbmQoJTIyZmxhc2hfaHViJTIyKSUyMCUyMCUyMyUyMG9yJTIweW91ciUyMHByZWZlcnJlZCUyMGJhY2tlbmQ=",highlighted:'pipe.transformer.set_attention_backend(<span class="hljs-string">"flash_hub"</span>) <span class="hljs-comment"># or your preferred backend</span>',lang:"py",wrap:!1});var k=e(x,6);a(k,{code:"cGlwZS5ndWlkZXIlMjAlMEE=",highlighted:`<span class="hljs-meta">>>> </span>pipe.guider | |
| ClassifierFreeGuidance { | |
| <span class="hljs-string">"_class_name"</span>: <span class="hljs-string">"ClassifierFreeGuidance"</span>, | |
| <span class="hljs-string">"_diffusers_version"</span>: <span class="hljs-string">"0.36.0.dev0"</span>, | |
| <span class="hljs-string">"enabled"</span>: true, | |
| <span class="hljs-string">"guidance_rescale"</span>: <span class="hljs-number">0.0</span>, | |
| <span class="hljs-string">"guidance_scale"</span>: <span class="hljs-number">6.0</span>, | |
| <span class="hljs-string">"start"</span>: <span class="hljs-number">0.0</span>, | |
| <span class="hljs-string">"stop"</span>: <span class="hljs-number">1.0</span>, | |
| <span class="hljs-string">"use_original_formulation"</span>: false | |
| } | |
| State: | |
| step: <span class="hljs-literal">None</span> | |
| num_inference_steps: <span class="hljs-literal">None</span> | |
| timestep: <span class="hljs-literal">None</span> | |
| count_prepared: <span class="hljs-number">0</span> | |
| enabled: <span class="hljs-literal">True</span> | |
| num_conditions: <span class="hljs-number">2</span>`,lang:"py",wrap:!1});var H=e(k,4);a(H,{code:"cGlwZS5ndWlkZXIlMjAlM0QlMjBwaXBlLmd1aWRlci5uZXcoZ3VpZGFuY2Vfc2NhbGUlM0Q1LjAp",highlighted:'pipe.guider = pipe.guider.new(guidance_scale=<span class="hljs-number">5.0</span>)',lang:"py",wrap:!1});var I=e(H,4);i(I,{title:"HunyuanVideo15Pipeline",local:"diffusers.HunyuanVideo15Pipeline",headingTag:"h2"});var c=e(I,2),J=o(c);n(J,{name:"class diffusers.HunyuanVideo15Pipeline",anchor:"diffusers.HunyuanVideo15Pipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L166",parameters:[{name:"text_encoder",val:": Qwen2_5_VLTextModel"},{name:"tokenizer",val:": Qwen2Tokenizer"},{name:"transformer",val:": HunyuanVideo15Transformer3DModel"},{name:"vae",val:": AutoencoderKLHunyuanVideo15"},{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"},{name:"text_encoder_2",val:": T5EncoderModel"},{name:"tokenizer_2",val:": ByT5Tokenizer"},{name:"guider",val:": ClassifierFreeGuidance"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/hunyuan_video15_transformer_3d#diffusers.HunyuanVideo15Transformer3DModel">HunyuanVideo15Transformer3DModel</a>) — | |
| Conditional Transformer (MMDiT) architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.HunyuanVideo15Pipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14230/en/api/schedulers/flow_match_euler_discrete#diffusers.FlowMatchEulerDiscreteScheduler">FlowMatchEulerDiscreteScheduler</a>) — | |
| A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents.`,name:"scheduler"},{anchor:"diffusers.HunyuanVideo15Pipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/autoencoder_kl_hunyuan_video15#diffusers.AutoencoderKLHunyuanVideo15">AutoencoderKLHunyuanVideo15</a>) — | |
| Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.HunyuanVideo15Pipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>Qwen2.5-VL-7B-Instruct</code>) — | |
| <a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a>, specifically the | |
| <a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a> variant.`,name:"text_encoder"},{anchor:"diffusers.HunyuanVideo15Pipeline.tokenizer",description:"<strong>tokenizer</strong> (<code>Qwen2Tokenizer</code>) — Tokenizer of class [Qwen2Tokenizer].",name:"tokenizer"},{anchor:"diffusers.HunyuanVideo15Pipeline.text_encoder_2",description:`<strong>text_encoder_2</strong> (<code>T5EncoderModel</code>) — | |
| <a href="https://huggingface.co/docs/transformers/en/model_doc/t5#transformers.T5EncoderModel" rel="nofollow">T5EncoderModel</a> | |
| variant.`,name:"text_encoder_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.tokenizer_2",description:"<strong>tokenizer_2</strong> (<code>ByT5Tokenizer</code>) — Tokenizer of class [ByT5Tokenizer]",name:"tokenizer_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14230/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>) — | |
| [ClassifierFreeGuidance]for classifier free guidance.`,name:"guider"}]});var m=e(J,6),j=o(m);n(j,{name:"__call__",anchor:"diffusers.HunyuanVideo15Pipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L542",parameters:[{name:"prompt",val:": str | list[str] = None"},{name:"negative_prompt",val:": str | list[str] = None"},{name:"height",val:": int | None = None"},{name:"width",val:": int | None = None"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"sigmas",val:": list = None"},{name:"num_videos_per_prompt",val:": int | None = 1"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": str | None = 'np'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": dict[str, typing.Any] | None = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| The prompt or prompts to guide the image generation. If not defined, one has to pass <code>prompt_embeds</code> | |
| instead.`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| The prompt or prompts not to guide the image generation. If not defined, one has to pass | |
| <code>negative_prompt_embeds</code> instead.`,name:"negative_prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, <em>optional</em>) — | |
| The height in pixels of the generated video.`,name:"height"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, <em>optional</em>) — | |
| The width in pixels of the generated video.`,name:"width"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) — | |
| The number of frames in the generated video.`,name:"num_frames"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, defaults to <code>50</code>) — | |
| The number of denoising steps. More denoising steps usually lead to a higher quality video at the | |
| expense of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.sigmas",description:`<strong>sigmas</strong> (<code>list[float]</code>, <em>optional</em>) — | |
| Custom sigmas to use for the denoising process with schedulers which support a <code>sigmas</code> argument in | |
| their <code>set_timesteps</code> method. If not defined, the default behavior when <code>num_inference_steps</code> is passed | |
| will be used.`,name:"sigmas"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>list[torch.Generator]</code>, <em>optional</em>) — | |
| A <a href="https://pytorch.org/docs/stable/generated/torch.Generator.html" rel="nofollow"><code>torch.Generator</code></a> to make | |
| generation deterministic.`,name:"generator"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated noisy latents sampled from a Gaussian distribution, to be used as inputs for video | |
| generation. Can be used to tweak the same generation with different prompts. If not provided, a latents | |
| tensor is generated by sampling using the supplied random <code>generator</code>.`,name:"latents"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings. Can be used to easily tweak text inputs (prompt weighting). If not | |
| provided, text embeddings are generated from the <code>prompt</code> input argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for prompt embeddings.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt | |
| weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input | |
| argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds_mask",description:`<strong>negative_prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for negative prompt embeddings.`,name:"negative_prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings from the second text encoder. Can be used to easily tweak text inputs.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for prompt embeddings from the second text encoder.`,name:"prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds_2",description:`<strong>negative_prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings from the second text encoder.`,name:"negative_prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.negative_prompt_embeds_mask_2",description:`<strong>negative_prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for negative prompt embeddings from the second text encoder.`,name:"negative_prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"np"</code>) — | |
| The output format of the generated video. Choose between “np”, “pt”, or “latent”.`,name:"output_type"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to return a <code>HunyuanVideo15PipelineOutput</code> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) — | |
| A kwargs dictionary that if specified is passed along to the <code>AttentionProcessor</code> as defined under | |
| <code>self.processor</code> in | |
| <a href="https://github.com/huggingface/diffusers/blob/main/src/diffusers/models/attention_processor.py" rel="nofollow">diffusers.models.attention_processor</a>.`,name:"attention_kwargs"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If <code>return_dict</code> is <code>True</code>, <code>HunyuanVideo15PipelineOutput</code> is returned, otherwise a <code>tuple</code> is | |
| returned where the first element is a list with the generated videos.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>~HunyuanVideo15PipelineOutput</code> or <code>tuple</code></p> | |
| `});var Q=e(j,4);z(Q,{anchor:"diffusers.HunyuanVideo15Pipeline.__call__.example",children:(s,d)=>{var r=N(),y=e(b(r),2);a(y,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSHVueXVhblZpZGVvMTVQaXBlbGluZSUwQWZyb20lMjBkaWZmdXNlcnMudXRpbHMlMjBpbXBvcnQlMjBleHBvcnRfdG9fdmlkZW8lMEElMEFtb2RlbF9pZCUyMCUzRCUyMCUyMmh1bnl1YW52aWRlby1jb21tdW5pdHklMkZIdW55dWFuVmlkZW8tMS41LTQ4MHBfdDJ2JTIyJTBBcGlwZSUyMCUzRCUyMEh1bnl1YW5WaWRlbzE1UGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKG1vZGVsX2lkJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5mbG9hdDE2KSUwQXBpcGUudmFlLmVuYWJsZV90aWxpbmcoKSUwQXBpcGUudG8oJTIyY3VkYSUyMiklMEElMEFvdXRwdXQlMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUzRCUyMkElMjBjYXQlMjB3YWxrcyUyMG9uJTIwdGhlJTIwZ3Jhc3MlMkMlMjByZWFsaXN0aWMlMjIlMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8ob3V0cHV0JTJDJTIwJTIyb3V0cHV0Lm1wNCUyMiUyQyUyMGZwcyUzRDE1KQ==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> HunyuanVideo15Pipeline | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| <span class="hljs-meta">>>> </span>model_id = <span class="hljs-string">"hunyuanvideo-community/HunyuanVideo-1.5-480p_t2v"</span> | |
| <span class="hljs-meta">>>> </span>pipe = HunyuanVideo15Pipeline.from_pretrained(model_id, torch_dtype=torch.float16) | |
| <span class="hljs-meta">>>> </span>pipe.vae.enable_tiling() | |
| <span class="hljs-meta">>>> </span>pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span>output = pipe( | |
| <span class="hljs-meta">... </span> prompt=<span class="hljs-string">"A cat walks on the grass, realistic"</span>, | |
| <span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>, | |
| <span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>] | |
| <span class="hljs-meta">>>> </span>export_to_video(output, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">15</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),t(m);var u=e(m,2),E=o(u);n(E,{name:"encode_prompt",anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L334",parameters:[{name:"prompt",val:": str | list[str]"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"},{name:"batch_size",val:": int = 1"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| prompt to be encoded`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.device",description:`<strong>device</strong> — (<code>torch.device</code>): | |
| torch device`,name:"device"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.batch_size",description:`<strong>batch_size</strong> (<code>int</code>) — | |
| batch size of prompts, defaults to 1`,name:"batch_size"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>) — | |
| number of images that should be generated per prompt`,name:"num_images_per_prompt"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings. If not provided, text embeddings will be generated from <code>prompt</code> input | |
| argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text mask. If not provided, text mask will be generated from <code>prompt</code> input argument.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated glyph text embeddings from ByT5. If not provided, will be generated from <code>prompt</code> input | |
| argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15Pipeline.encode_prompt.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated glyph text mask from ByT5. If not provided, will be generated from <code>prompt</code> input | |
| argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_mask_2"}]}),t(u);var P=e(u,2),R=o(P);n(R,{name:"prepare_cond_latents_and_mask",anchor:"diffusers.HunyuanVideo15Pipeline.prepare_cond_latents_and_mask",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5.py#L508",parameters:[{name:"latents",val:""},{name:"dtype",val:": typing.Optional[torch.dtype]"},{name:"device",val:": typing.Optional[torch.device]"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15Pipeline.prepare_cond_latents_and_mask.latents",description:"<strong>latents</strong> — Main latents tensor (B, C, F, H, W)",name:"latents"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>(cond_latents_concat, mask_concat) - both are zero tensors for t2v</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>tuple</p> | |
| `}),l(2),t(P),t(c);var Z=e(c,2);i(Z,{title:"HunyuanVideo15ImageToVideoPipeline",local:"diffusers.HunyuanVideo15ImageToVideoPipeline",headingTag:"h2"});var _=e(Z,2),U=o(_);n(U,{name:"class diffusers.HunyuanVideo15ImageToVideoPipeline",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L193",parameters:[{name:"text_encoder",val:": Qwen2_5_VLTextModel"},{name:"tokenizer",val:": Qwen2Tokenizer"},{name:"transformer",val:": HunyuanVideo15Transformer3DModel"},{name:"vae",val:": AutoencoderKLHunyuanVideo15"},{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"},{name:"text_encoder_2",val:": T5EncoderModel"},{name:"tokenizer_2",val:": ByT5Tokenizer"},{name:"guider",val:": ClassifierFreeGuidance"},{name:"image_encoder",val:": SiglipVisionModel"},{name:"feature_extractor",val:": SiglipImageProcessorPil"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/hunyuan_video15_transformer_3d#diffusers.HunyuanVideo15Transformer3DModel">HunyuanVideo15Transformer3DModel</a>) — | |
| Conditional Transformer (MMDiT) architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14230/en/api/schedulers/flow_match_euler_discrete#diffusers.FlowMatchEulerDiscreteScheduler">FlowMatchEulerDiscreteScheduler</a>) — | |
| A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents.`,name:"scheduler"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14230/en/api/models/autoencoder_kl_hunyuan_video15#diffusers.AutoencoderKLHunyuanVideo15">AutoencoderKLHunyuanVideo15</a>) — | |
| Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>Qwen2.5-VL-7B-Instruct</code>) — | |
| <a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a>, specifically the | |
| <a href="https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct" rel="nofollow">Qwen2.5-VL-7B-Instruct</a> variant.`,name:"text_encoder"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.tokenizer",description:"<strong>tokenizer</strong> (<code>Qwen2Tokenizer</code>) — Tokenizer of class [Qwen2Tokenizer].",name:"tokenizer"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.text_encoder_2",description:`<strong>text_encoder_2</strong> (<code>T5EncoderModel</code>) — | |
| <a href="https://huggingface.co/docs/transformers/en/model_doc/t5#transformers.T5EncoderModel" rel="nofollow">T5EncoderModel</a> | |
| variant.`,name:"text_encoder_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.tokenizer_2",description:"<strong>tokenizer_2</strong> (<code>ByT5Tokenizer</code>) — Tokenizer of class [ByT5Tokenizer]",name:"tokenizer_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14230/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>) — | |
| [ClassifierFreeGuidance]for classifier free guidance.`,name:"guider"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.image_encoder",description:`<strong>image_encoder</strong> (<code>SiglipVisionModel</code>) — | |
| <a href="https://huggingface.co/docs/transformers/en/model_doc/siglip#transformers.SiglipVisionModel" rel="nofollow">SiglipVisionModel</a> | |
| variant.`,name:"image_encoder"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.feature_extractor",description:`<strong>feature_extractor</strong> (<code>SiglipImageProcessor</code>) — | |
| <a href="https://huggingface.co/docs/transformers/en/model_doc/siglip#transformers.SiglipImageProcessor" rel="nofollow">SiglipImageProcessor</a> | |
| variant.`,name:"feature_extractor"}]});var g=e(U,6),G=o(g);n(G,{name:"__call__",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L650",parameters:[{name:"image",val:": Image"},{name:"prompt",val:": str | list[str] = None"},{name:"negative_prompt",val:": str | list[str] = None"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"sigmas",val:": list = None"},{name:"num_videos_per_prompt",val:": int | None = 1"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": str | None = 'np'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": dict[str, typing.Any] | None = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.image",description:`<strong>image</strong> (<code>PIL.Image.Image</code>) — | |
| The input image to condition video generation on.`,name:"image"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| The prompt or prompts to guide the video generation. If not defined, one has to pass <code>prompt_embeds</code> | |
| instead.`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| The prompt or prompts not to guide the video generation. If not defined, one has to pass | |
| <code>negative_prompt_embeds</code> instead.`,name:"negative_prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) — | |
| The number of frames in the generated video.`,name:"num_frames"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, defaults to <code>50</code>) — | |
| The number of denoising steps. More denoising steps usually lead to a higher quality video at the | |
| expense of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.sigmas",description:`<strong>sigmas</strong> (<code>list[float]</code>, <em>optional</em>) — | |
| Custom sigmas to use for the denoising process with schedulers which support a <code>sigmas</code> argument in | |
| their <code>set_timesteps</code> method. If not defined, the default behavior when <code>num_inference_steps</code> is passed | |
| will be used.`,name:"sigmas"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>list[torch.Generator]</code>, <em>optional</em>) — | |
| A <a href="https://pytorch.org/docs/stable/generated/torch.Generator.html" rel="nofollow"><code>torch.Generator</code></a> to make | |
| generation deterministic.`,name:"generator"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated noisy latents sampled from a Gaussian distribution, to be used as inputs for video | |
| generation. Can be used to tweak the same generation with different prompts. If not provided, a latents | |
| tensor is generated by sampling using the supplied random <code>generator</code>.`,name:"latents"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings. Can be used to easily tweak text inputs (prompt weighting). If not | |
| provided, text embeddings are generated from the <code>prompt</code> input argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for prompt embeddings.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt | |
| weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input | |
| argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds_mask",description:`<strong>negative_prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for negative prompt embeddings.`,name:"negative_prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings from the second text encoder. Can be used to easily tweak text inputs.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for prompt embeddings from the second text encoder.`,name:"prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds_2",description:`<strong>negative_prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings from the second text encoder.`,name:"negative_prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.negative_prompt_embeds_mask_2",description:`<strong>negative_prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated mask for negative prompt embeddings from the second text encoder.`,name:"negative_prompt_embeds_mask_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"np"</code>) — | |
| The output format of the generated video. Choose between “np”, “pt”, or “latent”.`,name:"output_type"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to return a <code>HunyuanVideo15PipelineOutput</code> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) — | |
| A kwargs dictionary that if specified is passed along to the <code>AttentionProcessor</code> as defined under | |
| <code>self.processor</code> in | |
| <a href="https://github.com/huggingface/diffusers/blob/main/src/diffusers/models/attention_processor.py" rel="nofollow">diffusers.models.attention_processor</a>.`,name:"attention_kwargs"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If <code>return_dict</code> is <code>True</code>, <code>HunyuanVideo15PipelineOutput</code> is returned, otherwise a <code>tuple</code> is | |
| returned where the first element is a list with the generated videos.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>~HunyuanVideo15PipelineOutput</code> or <code>tuple</code></p> | |
| `});var S=e(G,4);z(S,{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.__call__.example",children:(s,d)=>{var r=N(),y=e(b(r),2);a(y,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwSHVueXVhblZpZGVvMTVJbWFnZVRvVmlkZW9QaXBlbGluZSUwQWZyb20lMjBkaWZmdXNlcnMudXRpbHMlMjBpbXBvcnQlMjBleHBvcnRfdG9fdmlkZW8lMEElMEFtb2RlbF9pZCUyMCUzRCUyMCUyMmh1bnl1YW52aWRlby1jb21tdW5pdHklMkZIdW55dWFuVmlkZW8tMS41LTQ4MHBfaTJ2JTIyJTBBcGlwZSUyMCUzRCUyMEh1bnl1YW5WaWRlbzE1SW1hZ2VUb1ZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKG1vZGVsX2lkJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5mbG9hdDE2KSUwQXBpcGUudmFlLmVuYWJsZV90aWxpbmcoKSUwQXBpcGUudG8oJTIyY3VkYSUyMiklMEElMEFpbWFnZSUyMCUzRCUyMGxvYWRfaW1hZ2UoJTIyaHR0cHMlM0ElMkYlMkZodWdnaW5nZmFjZS5jbyUyRmRhdGFzZXRzJTJGWWlZaVh1JTJGdGVzdGluZy1pbWFnZXMlMkZyZXNvbHZlJTJGbWFpbiUyRndhbl9pMnZfaW5wdXQuSlBHJTIyKSUwQSUwQW91dHB1dCUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwcHJvbXB0JTNEJTIyU3VtbWVyJTIwYmVhY2glMjB2YWNhdGlvbiUyMHN0eWxlJTJDJTIwYSUyMHdoaXRlJTIwY2F0JTIwd2VhcmluZyUyMHN1bmdsYXNzZXMlMjBzaXRzJTIwb24lMjBhJTIwc3VyZmJvYXJkLiUyMFRoZSUyMGZsdWZmeS1mdXJyZWQlMjBmZWxpbmUlMjBnYXplcyUyMGRpcmVjdGx5JTIwYXQlMjB0aGUlMjBjYW1lcmElMjB3aXRoJTIwYSUyMHJlbGF4ZWQlMjBleHByZXNzaW9uLiUyMEJsdXJyZWQlMjBiZWFjaCUyMHNjZW5lcnklMjBmb3JtcyUyMHRoZSUyMGJhY2tncm91bmQlMjBmZWF0dXJpbmclMjBjcnlzdGFsLWNsZWFyJTIwd2F0ZXJzJTJDJTIwZGlzdGFudCUyMGdyZWVuJTIwaGlsbHMlMkMlMjBhbmQlMjBhJTIwYmx1ZSUyMHNreSUyMGRvdHRlZCUyMHdpdGglMjB3aGl0ZSUyMGNsb3Vkcy4lMjBUaGUlMjBjYXQlMjBhc3N1bWVzJTIwYSUyMG5hdHVyYWxseSUyMHJlbGF4ZWQlMjBwb3N0dXJlJTJDJTIwYXMlMjBpZiUyMHNhdm9yaW5nJTIwdGhlJTIwc2VhJTIwYnJlZXplJTIwYW5kJTIwd2FybSUyMHN1bmxpZ2h0LiUyMEElMjBjbG9zZS11cCUyMHNob3QlMjBoaWdobGlnaHRzJTIwdGhlJTIwZmVsaW5lJ3MlMjBpbnRyaWNhdGUlMjBkZXRhaWxzJTIwYW5kJTIwdGhlJTIwcmVmcmVzaGluZyUyMGF0bW9zcGhlcmUlMjBvZiUyMHRoZSUyMHNlYXNpZGUuJTIyJTJDJTBBJTIwJTIwJTIwJTIwaW1hZ2UlM0RpbWFnZSUyQyUwQSUyMCUyMCUyMCUyMG51bV9pbmZlcmVuY2Vfc3RlcHMlM0Q1MCUyQyUwQSkuZnJhbWVzJTVCMCU1RCUwQWV4cG9ydF90b192aWRlbyhvdXRwdXQlMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> HunyuanVideo15ImageToVideoPipeline | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| <span class="hljs-meta">>>> </span>model_id = <span class="hljs-string">"hunyuanvideo-community/HunyuanVideo-1.5-480p_i2v"</span> | |
| <span class="hljs-meta">>>> </span>pipe = HunyuanVideo15ImageToVideoPipeline.from_pretrained(model_id, torch_dtype=torch.float16) | |
| <span class="hljs-meta">>>> </span>pipe.vae.enable_tiling() | |
| <span class="hljs-meta">>>> </span>pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span>image = load_image(<span class="hljs-string">"https://huggingface.co/datasets/YiYiXu/testing-images/resolve/main/wan_i2v_input.JPG"</span>) | |
| <span class="hljs-meta">>>> </span>output = pipe( | |
| <span class="hljs-meta">... </span> prompt=<span class="hljs-string">"Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline's intricate details and the refreshing atmosphere of the seaside."</span>, | |
| <span class="hljs-meta">... </span> image=image, | |
| <span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>, | |
| <span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>] | |
| <span class="hljs-meta">>>> </span>export_to_video(output, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),t(g);var h=e(g,2),F=o(h);n(F,{name:"encode_prompt",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L422",parameters:[{name:"prompt",val:": str | list[str]"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"},{name:"batch_size",val:": int = 1"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_2",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds_mask_2",val:": typing.Optional[torch.Tensor] = None"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| prompt to be encoded`,name:"prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.device",description:`<strong>device</strong> — (<code>torch.device</code>): | |
| torch device`,name:"device"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.batch_size",description:`<strong>batch_size</strong> (<code>int</code>) — | |
| batch size of prompts, defaults to 1`,name:"batch_size"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>) — | |
| number of images that should be generated per prompt`,name:"num_images_per_prompt"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings. If not provided, text embeddings will be generated from <code>prompt</code> input | |
| argument.`,name:"prompt_embeds"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds_mask",description:`<strong>prompt_embeds_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text mask. If not provided, text mask will be generated from <code>prompt</code> input argument.`,name:"prompt_embeds_mask"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds_2",description:`<strong>prompt_embeds_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated glyph text embeddings from ByT5. If not provided, will be generated from <code>prompt</code> input | |
| argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_2"},{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.encode_prompt.prompt_embeds_mask_2",description:`<strong>prompt_embeds_mask_2</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated glyph text mask from ByT5. If not provided, will be generated from <code>prompt</code> input | |
| argument using self.tokenizer_2 and self.text_encoder_2.`,name:"prompt_embeds_mask_2"}]}),t(h);var W=e(h,2),Y=o(W);n(Y,{name:"prepare_cond_latents_and_mask",anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.prepare_cond_latents_and_mask",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_hunyuan_video1_5_image2video.py#L594",parameters:[{name:"latents",val:": Tensor"},{name:"image",val:": Image"},{name:"batch_size",val:": int"},{name:"height",val:": int"},{name:"width",val:": int"},{name:"dtype",val:": dtype"},{name:"device",val:": device"}],parametersDescription:[{anchor:"diffusers.HunyuanVideo15ImageToVideoPipeline.prepare_cond_latents_and_mask.latents",description:"<strong>latents</strong> — Main latents tensor (B, C, F, H, W)",name:"latents"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>(cond_latents_concat, mask_concat) - both are zero tensors for t2v</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>tuple</p> | |
| `}),l(2),t(W),t(_);var B=e(_,2);i(B,{title:"HunyuanVideo15PipelineOutput",local:"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput",headingTag:"h2"});var f=e(B,2),q=o(f);n(q,{name:"class diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput",anchor:"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14230/src/diffusers/pipelines/hunyuan_video1_5/pipeline_output.py#L9",parameters:[{name:"frames",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.pipelines.hunyuan_video1_5.pipeline_output.HunyuanVideo15PipelineOutput.frames",description:`<strong>frames</strong> (<code>torch.Tensor</code>, <code>np.ndarray</code>, or list[list[PIL.Image.Image]]) — | |
| List of video outputs - It can be a nested list of length <code>batch_size,</code> with each sub-list containing | |
| denoised PIL image sequences of length <code>num_frames.</code> It can also be a NumPy array or Torch tensor of shape | |
| <code>(batch_size, num_frames, channels, height, width)</code>.`,name:"frames"}]}),l(2),t(f);var D=e(f,2);A(D,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/hunyuan_video15.md"}),l(2),p(C,T),ne()}export{pe as component}; | |
Xet Storage Details
- Size:
- 50.9 kB
- Xet hash:
- 77023c77471b920eb291babed98fa53e5cea774ee525403c550f437d110c55f3
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.