Buckets:

download
raw
54.3 kB
import"../chunks/DsnmJJEf.js";import{i as L,h as D,C as Y,H as o,a,D as t,E as O,s as K}from"../chunks/BtE7mKSK.js";import{p as $,o as ee,s as e,f as h,a as p,b as oe,c as n,d as M,r as i,n as l}from"../chunks/jDjavuwI.js";import{E as Q}from"../chunks/SrSJA0zO.js";const te='{"title":"Motif-Video","local":"motif-video","sections":[{"title":"Text-to-Video Generation","local":"text-to-video-generation","sections":[],"depth":2},{"title":"Image-to-Video Generation","local":"image-to-video-generation","sections":[{"title":"Memory-efficient Inference","local":"memory-efficient-inference","sections":[],"depth":3}],"depth":2},{"title":"MotifVideoPipeline","local":"diffusers.MotifVideoPipeline","sections":[],"depth":2},{"title":"MotifVideoImage2VideoPipeline","local":"diffusers.MotifVideoImage2VideoPipeline","sections":[],"depth":2},{"title":"MotifVideoPipelineOutput","local":"diffusers.MotifVideoPipelineOutput","sections":[],"depth":2}],"depth":1}';var ne=M('<meta name="hf:doc:metadata"/>'),R=M("<p>Examples:</p> <!>",1),ie=M(`<p></p> <!> <!> <p><a href="https://arxiv.org/abs/2604.16503" rel="nofollow">Technical Report</a></p> <p>Motif-Video is a 2B parameter diffusion transformer designed for text-to-video and image-to-video generation. It features a three-stage architecture with 12 dual-stream + 16 single-stream + 8 DDT decoder layers, Shared Cross-Attention for stable text-video alignment under long video sequences, T5Gemma2 text encoder, and rectified flow matching for velocity prediction.</p> <p align="center"><img src="https://huggingface.co/Motif-Technologies/Motif-Video-2B/resolve/main/assets/architecture.png" width="90%" alt="Motif-Video architecture"/></p> <!> <p>Use <code>MotifVideoPipeline</code> for text-to-video generation:</p> <!> <!> <p>Use <code>MotifVideoImage2VideoPipeline</code> for image-to-video generation:</p> <!> <!> <p>For GPUs with less than 30GB VRAM (e.g., RTX 4090), use model CPU offloading:</p> <!> <!> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for text-to-video generation using Motif-Video.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14192/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods
implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for text-to-video generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encodes the prompt into text encoder hidden states.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for image-to-video generation using Motif-Video with first frame conditioning.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14192/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods
implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for image-to-video generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encodes the prompt into text encoder hidden states.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for Motif-Video pipelines.</p></div> <!> <p></p>`,1);function pe(X,E){$(E,!1),ee(()=>{new URLSearchParams(window.location.search).get("fw")}),L();var y=ie();D("mz3w07",s=>{var d=ne();K(d,"content",te),p(s,d)});var v=e(h(y),2);Y(v,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var b=e(v,2);o(b,{title:"Motif-Video",local:"motif-video",headingTag:"h1"});var T=e(b,8);o(T,{title:"Text-to-Video Generation",local:"text-to-video-generation",headingTag:"h2"});var U=e(T,4);a(U,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb1BpcGVsaW5lJTBBZnJvbSUyMGRpZmZ1c2Vycy51dGlscyUyMGltcG9ydCUyMGV4cG9ydF90b192aWRlbyUwQSUwQSUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMk1vdGlmLVRlY2hub2xvZ2llcyUyRk1vdGlmLVZpZGVvLTJCJTIyJTJDJTBBJTIwJTIwJTIwJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiUyQyUwQSklMEFwaXBlLnRvKCUyMmN1ZGElMjIpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMHdvbWFuJTIwd2l0aCUyMGxvbmclMjBicm93biUyMGhhaXIlMjBhbmQlMjBsaWdodCUyMHNraW4lMjBzbWlsZXMlMjBhdCUyMGFub3RoZXIlMjB3b21hbiUyMHdpdGglMjBsb25nJTIwYmxvbmRlJTIwaGFpci4lMjIlMEFuZWdhdGl2ZV9wcm9tcHQlMjAlM0QlMjAlMjJ3b3JzdCUyMHF1YWxpdHklMkMlMjBpbmNvbnNpc3RlbnQlMjBtb3Rpb24lMkMlMjBibHVycnklMkMlMjBqaXR0ZXJ5JTJDJTIwZGlzdG9ydGVkJTIyJTBBJTBBdmlkZW8lMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMG5lZ2F0aXZlX3Byb21wdCUzRG5lZ2F0aXZlX3Byb21wdCUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEMTI4MCUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRDczNiUyQyUwQSUyMCUyMCUyMCUyMG51bV9mcmFtZXMlM0QxMjElMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoPipeline
<span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video
pipe = MotifVideoPipeline.from_pretrained(
<span class="hljs-string">&quot;Motif-Technologies/Motif-Video-2B&quot;</span>,
torch_dtype=torch.bfloat16,
)
pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>)
prompt = <span class="hljs-string">&quot;A woman with long brown hair and light skin smiles at another woman with long blonde hair.&quot;</span>
negative_prompt = <span class="hljs-string">&quot;worst quality, inconsistent motion, blurry, jittery, distorted&quot;</span>
video = pipe(
prompt=prompt,
negative_prompt=negative_prompt,
width=<span class="hljs-number">1280</span>,
height=<span class="hljs-number">736</span>,
num_frames=<span class="hljs-number">121</span>,
num_inference_steps=<span class="hljs-number">50</span>,
).frames[<span class="hljs-number">0</span>]
export_to_video(video, <span class="hljs-string">&quot;output.mp4&quot;</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1});var w=e(U,2);o(w,{title:"Image-to-Video Generation",local:"image-to-video-generation",headingTag:"h2"});var J=e(w,4);a(J,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb0ltYWdlMlZpZGVvUGlwZWxpbmUlMEFmcm9tJTIwZGlmZnVzZXJzLnV0aWxzJTIwaW1wb3J0JTIwZXhwb3J0X3RvX3ZpZGVvJTJDJTIwbG9hZF9pbWFnZSUwQSUwQSUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvSW1hZ2UyVmlkZW9QaXBlbGluZS5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwJTIyTW90aWYtVGVjaG5vbG9naWVzJTJGTW90aWYtVmlkZW8tMkIlMjIlMkMlMEElMjAlMjAlMjAlMjB0b3JjaF9kdHlwZSUzRHRvcmNoLmJmbG9hdDE2JTJDJTBBKSUwQXBpcGUudG8oJTIyY3VkYSUyMiklMEElMEFpbWFnZSUyMCUzRCUyMGxvYWRfaW1hZ2UoJTIyaW5wdXRfaW1hZ2UucG5nJTIyKSUwQXByb21wdCUyMCUzRCUyMCUyMkElMjBjaW5lbWF0aWMlMjBzY2VuZSUyMHdpdGglMjB2aXZpZCUyMGNvbG9ycy4lMjIlMEFuZWdhdGl2ZV9wcm9tcHQlMjAlM0QlMjAlMjJ3b3JzdCUyMHF1YWxpdHklMkMlMjBibHVycnklMkMlMjBqaXR0ZXJ5JTJDJTIwZGlzdG9ydGVkJTIyJTBBJTBBdmlkZW8lMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMGltYWdlJTNEaW1hZ2UlMkMlMEElMjAlMjAlMjAlMjBwcm9tcHQlM0Rwcm9tcHQlMkMlMEElMjAlMjAlMjAlMjBuZWdhdGl2ZV9wcm9tcHQlM0RuZWdhdGl2ZV9wcm9tcHQlMkMlMEElMjAlMjAlMjAlMjB3aWR0aCUzRDEyODAlMkMlMEElMjAlMjAlMjAlMjBoZWlnaHQlM0Q3MzYlMkMlMEElMjAlMjAlMjAlMjBudW1fZnJhbWVzJTNEMTIxJTJDJTBBJTIwJTIwJTIwJTIwbnVtX2luZmVyZW5jZV9zdGVwcyUzRDUwJTJDJTBBKS5mcmFtZXMlNUIwJTVEJTBBZXhwb3J0X3RvX3ZpZGVvKHZpZGVvJTJDJTIwJTIyaTJ2X291dHB1dC5tcDQlMjIlMkMlMjBmcHMlM0QyNCk=",highlighted:`<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoImage2VideoPipeline
<span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_image
pipe = MotifVideoImage2VideoPipeline.from_pretrained(
<span class="hljs-string">&quot;Motif-Technologies/Motif-Video-2B&quot;</span>,
torch_dtype=torch.bfloat16,
)
pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>)
image = load_image(<span class="hljs-string">&quot;input_image.png&quot;</span>)
prompt = <span class="hljs-string">&quot;A cinematic scene with vivid colors.&quot;</span>
negative_prompt = <span class="hljs-string">&quot;worst quality, blurry, jittery, distorted&quot;</span>
video = pipe(
image=image,
prompt=prompt,
negative_prompt=negative_prompt,
width=<span class="hljs-number">1280</span>,
height=<span class="hljs-number">736</span>,
num_frames=<span class="hljs-number">121</span>,
num_inference_steps=<span class="hljs-number">50</span>,
).frames[<span class="hljs-number">0</span>]
export_to_video(video, <span class="hljs-string">&quot;i2v_output.mp4&quot;</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1});var V=e(J,2);o(V,{title:"Memory-efficient Inference",local:"memory-efficient-inference",headingTag:"h3"});var j=e(V,4);a(j,{code:"ZXhwb3J0JTIwUFlUT1JDSF9DVURBX0FMTE9DX0NPTkYlM0RleHBhbmRhYmxlX3NlZ21lbnRzJTNBVHJ1ZQ==",highlighted:'<span class="hljs-built_in">export</span> PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True',lang:"bash",wrap:!1});var I=e(j,2);a(I,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb1BpcGVsaW5lJTBBZnJvbSUyMGRpZmZ1c2Vycy51dGlscyUyMGltcG9ydCUyMGV4cG9ydF90b192aWRlbyUwQSUwQSUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMk1vdGlmLVRlY2hub2xvZ2llcyUyRk1vdGlmLVZpZGVvLTJCJTIyJTJDJTBBJTIwJTIwJTIwJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiUyQyUwQSklMEFwaXBlLmVuYWJsZV9tb2RlbF9jcHVfb2ZmbG9hZCgpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMHdvbWFuJTIwd2l0aCUyMGxvbmclMjBicm93biUyMGhhaXIlMjBhbmQlMjBsaWdodCUyMHNraW4lMjBzbWlsZXMlMjBhdCUyMGFub3RoZXIlMjB3b21hbiUyMHdpdGglMjBsb25nJTIwYmxvbmRlJTIwaGFpci4lMjIlMEFuZWdhdGl2ZV9wcm9tcHQlMjAlM0QlMjAlMjJ3b3JzdCUyMHF1YWxpdHklMkMlMjBpbmNvbnNpc3RlbnQlMjBtb3Rpb24lMkMlMjBibHVycnklMkMlMjBqaXR0ZXJ5JTJDJTIwZGlzdG9ydGVkJTIyJTBBJTBBdmlkZW8lMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMG5lZ2F0aXZlX3Byb21wdCUzRG5lZ2F0aXZlX3Byb21wdCUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEMTI4MCUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRDczNiUyQyUwQSUyMCUyMCUyMCUyMG51bV9mcmFtZXMlM0QxMjElMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoPipeline
<span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video
pipe = MotifVideoPipeline.from_pretrained(
<span class="hljs-string">&quot;Motif-Technologies/Motif-Video-2B&quot;</span>,
torch_dtype=torch.bfloat16,
)
pipe.enable_model_cpu_offload()
prompt = <span class="hljs-string">&quot;A woman with long brown hair and light skin smiles at another woman with long blonde hair.&quot;</span>
negative_prompt = <span class="hljs-string">&quot;worst quality, inconsistent motion, blurry, jittery, distorted&quot;</span>
video = pipe(
prompt=prompt,
negative_prompt=negative_prompt,
width=<span class="hljs-number">1280</span>,
height=<span class="hljs-number">736</span>,
num_frames=<span class="hljs-number">121</span>,
num_inference_steps=<span class="hljs-number">50</span>,
).frames[<span class="hljs-number">0</span>]
export_to_video(video, <span class="hljs-string">&quot;output.mp4&quot;</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1});var Z=e(I,2);o(Z,{title:"MotifVideoPipeline",local:"diffusers.MotifVideoPipeline",headingTag:"h2"});var c=e(Z,2),x=n(c);t(x,{name:"class diffusers.MotifVideoPipeline",anchor:"diffusers.MotifVideoPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video.py#L148",parameters:[{name:"scheduler",val:": SchedulerMixin"},{name:"vae",val:": AutoencoderKLWan"},{name:"text_encoder",val:": T5Gemma2Encoder"},{name:"tokenizer",val:": PreTrainedTokenizerBase"},{name:"transformer",val:": MotifVideoTransformer3DModel"},{name:"guider",val:": BaseGuidance"},{name:"feature_extractor",val:": typing.Optional[transformers.models.siglip.image_processing_pil_siglip.SiglipImageProcessorPil] = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/motif_video_transformer_3d#diffusers.MotifVideoTransformer3DModel">MotifVideoTransformer3DModel</a>) &#x2014;
Conditional Transformer architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.MotifVideoPipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14192/en/api/schedulers/overview#diffusers.SchedulerMixin">SchedulerMixin</a>) &#x2014;
A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents. Should be an
instance of a class inheriting from <code>SchedulerMixin</code>, such as <a href="/docs/diffusers/pr_14192/en/api/schedulers/multistep_dpm_solver#diffusers.DPMSolverMultistepScheduler">DPMSolverMultistepScheduler</a>. If not
provided, uses the scheduler attached to the pretrained model.`,name:"scheduler"},{anchor:"diffusers.MotifVideoPipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/autoencoder_kl_wan#diffusers.AutoencoderKLWan">AutoencoderKLWan</a>) &#x2014;
Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.MotifVideoPipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>T5Gemma2Encoder</code>) &#x2014;
Primary text encoder for encoding text prompts into embeddings.`,name:"text_encoder"},{anchor:"diffusers.MotifVideoPipeline.tokenizer",description:`<strong>tokenizer</strong> (<code>PreTrainedTokenizerBase</code>) &#x2014;
Tokenizer corresponding to the primary text encoder.`,name:"tokenizer"},{anchor:"diffusers.MotifVideoPipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.BaseGuidance">BaseGuidance</a>) &#x2014;
The guidance method to use. Should be an instance of a class inheriting from <code>BaseGuidance</code>, such as
<a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>, <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.AdaptiveProjectedGuidance">AdaptiveProjectedGuidance</a>, or <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.SkipLayerGuidance">SkipLayerGuidance</a>. If not provided,
defaults to <code>ClassifierFreeGuidance</code>.`,name:"guider"}]});var m=e(x,6),G=n(m);t(G,{name:"__call__",anchor:"diffusers.MotifVideoPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video.py#L492",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"height",val:": int = 736"},{name:"width",val:": int = 1280"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"timesteps",val:": typing.Optional[typing.List[int]] = None"},{name:"num_videos_per_prompt",val:": typing.Optional[int] = 1"},{name:"generator",val:": typing.Union[torch.Generator, typing.List[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": typing.Optional[str] = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": typing.Optional[typing.Dict[str, typing.Any]] = None"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int, typing.Dict], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": typing.List[str] = ['latents']"},{name:"max_sequence_length",val:": int = 512"},{name:"vae_batch_size",val:": int | None = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts to guide the video generation. If not defined, one has to pass <code>prompt_embeds</code>.`,name:"prompt"},{anchor:"diffusers.MotifVideoPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts not to guide the video generation. If not defined, one has to pass
<code>negative_prompt_embeds</code> instead. Ignored when not using guidance.`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoPipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, defaults to <code>736</code>) &#x2014;
The height in pixels of the generated video.`,name:"height"},{anchor:"diffusers.MotifVideoPipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, defaults to <code>1280</code>) &#x2014;
The width in pixels of the generated video.`,name:"width"},{anchor:"diffusers.MotifVideoPipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) &#x2014;
The number of video frames to generate.`,name:"num_frames"},{anchor:"diffusers.MotifVideoPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 50) &#x2014;
The number of denoising steps. More denoising steps usually lead to a higher quality video at the
expense of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.MotifVideoPipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<code>List[int]</code>, <em>optional</em>) &#x2014;
Custom timesteps to use for the denoising process.`,name:"timesteps"},{anchor:"diffusers.MotifVideoPipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) &#x2014;
The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoPipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>List[torch.Generator]</code>, <em>optional</em>) &#x2014;
PyTorch Generator object(s) for deterministic generation.`,name:"generator"},{anchor:"diffusers.MotifVideoPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated noisy latents.`,name:"latents"},{anchor:"diffusers.MotifVideoPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.__call__.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.__call__.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>&quot;pil&quot;</code>) &#x2014;
The output format of the generated video. Choose between <code>&quot;pil&quot;</code>, <code>&quot;np&quot;</code>, or <code>&quot;latent&quot;</code>.`,name:"output_type"},{anchor:"diffusers.MotifVideoPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) &#x2014;
Whether or not to return a <a href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput">~MotifVideoPipelineOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.MotifVideoPipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) &#x2014;
Arguments passed to the attention processor.`,name:"attention_kwargs"},{anchor:"diffusers.MotifVideoPipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <em>optional</em>) &#x2014;
A function or subclass of <code>PipelineCallback</code> or <code>MultiPipelineCallbacks</code> called at the end of each
denoising step.`,name:"callback_on_step_end"},{anchor:"diffusers.MotifVideoPipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>List</code>, <em>optional</em>) &#x2014;
The list of tensor inputs for the <code>callback_on_step_end</code> function.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.MotifVideoPipeline.__call__.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to <code>512</code>) &#x2014;
Maximum sequence length for the tokenizer.`,name:"max_sequence_length"},{anchor:"diffusers.MotifVideoPipeline.__call__.vae_batch_size",description:`<strong>vae_batch_size</strong> (<code>int</code>, <em>optional</em>) &#x2014;
Batch size for VAE decoding. If provided and latents batch size is larger, VAE decoding will be done in
chunks.`,name:"vae_batch_size"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>If <code>return_dict</code> is <code>True</code>, <a
href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput"
>~MotifVideoPipelineOutput</a> is returned, otherwise a <code>tuple</code> is returned
where the first element is a list of generated video frames.</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><a
href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput"
>~MotifVideoPipelineOutput</a> or <code>tuple</code></p>
`});var S=e(G,4);Q(S,{anchor:"diffusers.MotifVideoPipeline.__call__.example",children:(s,d)=>{var r=R(),_=e(h(r),2);a(_,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb1BpcGVsaW5lJTBBZnJvbSUyMGRpZmZ1c2Vycy51dGlscyUyMGltcG9ydCUyMGV4cG9ydF90b192aWRlbyUwQSUwQSUyMyUyMExvYWQlMjB0aGUlMjBNb3RpZi1WaWRlbyUyMHBpcGVsaW5lJTBBbW90aWZfdmlkZW9fbW9kZWxfaWQlMjAlM0QlMjAlMjJNb3RpZi1UZWNobm9sb2dpZXMlMkZNb3RpZi1WaWRlby0yQiUyMiUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKG1vdGlmX3ZpZGVvX21vZGVsX2lkJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiklMEFwaXBlLnRvKCUyMmN1ZGElMjIpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMHdvbWFuJTIwd2l0aCUyMGxvbmclMjBicm93biUyMGhhaXIlMjBhbmQlMjBsaWdodCUyMHNraW4lMjBzbWlsZXMlMjBhdCUyMGFub3RoZXIlMjB3b21hbiUyMHdpdGglMjBsb25nJTIwYmxvbmRlJTIwaGFpci4lMjBUaGUlMjB3b21hbiUyMHdpdGglMjBicm93biUyMGhhaXIlMjB3ZWFycyUyMGElMjBibGFjayUyMGphY2tldCUyMGFuZCUyMGhhcyUyMGElMjBzbWFsbCUyQyUyMGJhcmVseSUyMG5vdGljZWFibGUlMjBtb2xlJTIwb24lMjBoZXIlMjByaWdodCUyMGNoZWVrLiUyMFRoZSUyMGNhbWVyYSUyMGFuZ2xlJTIwaXMlMjBhJTIwY2xvc2UtdXAlMkMlMjBmb2N1c2VkJTIwb24lMjB0aGUlMjB3b21hbiUyMHdpdGglMjBicm93biUyMGhhaXIncyUyMGZhY2UuJTIwVGhlJTIwbGlnaHRpbmclMjBpcyUyMHdhcm0lMjBhbmQlMjBuYXR1cmFsJTJDJTIwbGlrZWx5JTIwZnJvbSUyMHRoZSUyMHNldHRpbmclMjBzdW4lMkMlMjBjYXN0aW5nJTIwYSUyMHNvZnQlMjBnbG93JTIwb24lMjB0aGUlMjBzY2VuZS4lMjBUaGUlMjBzY2VuZSUyMGFwcGVhcnMlMjB0byUyMGJlJTIwcmVhbC1saWZlJTIwZm9vdGFnZSUyMiUwQW5lZ2F0aXZlX3Byb21wdCUyMCUzRCUyMCUyMndvcnN0JTIwcXVhbGl0eSUyQyUyMGluY29uc2lzdGVudCUyMG1vdGlvbiUyQyUyMGJsdXJyeSUyQyUyMGppdHRlcnklMkMlMjBkaXN0b3J0ZWQlMjIlMEElMEF2aWRlbyUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwcHJvbXB0JTNEcHJvbXB0JTJDJTBBJTIwJTIwJTIwJTIwbmVnYXRpdmVfcHJvbXB0JTNEbmVnYXRpdmVfcHJvbXB0JTJDJTBBJTIwJTIwJTIwJTIwd2lkdGglM0QxMjgwJTJDJTBBJTIwJTIwJTIwJTIwaGVpZ2h0JTNENzM2JTJDJTBBJTIwJTIwJTIwJTIwbnVtX2ZyYW1lcyUzRDEyMSUyQyUwQSUyMCUyMCUyMCUyMG51bV9pbmZlcmVuY2Vfc3RlcHMlM0Q1MCUyQyUwQSkuZnJhbWVzJTVCMCU1RCUwQWV4cG9ydF90b192aWRlbyh2aWRlbyUyQyUyMCUyMm91dHB1dC5tcDQlMjIlMkMlMjBmcHMlM0QyNCk=",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">import</span> torch
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoPipeline
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># Load the Motif-Video pipeline</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>motif_video_model_id = <span class="hljs-string">&quot;Motif-Technologies/Motif-Video-2B&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe = MotifVideoPipeline.from_pretrained(motif_video_model_id, torch_dtype=torch.bfloat16)
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>prompt = <span class="hljs-string">&quot;A woman with long brown hair and light skin smiles at another woman with long blonde hair. The woman with brown hair wears a black jacket and has a small, barely noticeable mole on her right cheek. The camera angle is a close-up, focused on the woman with brown hair&#x27;s face. The lighting is warm and natural, likely from the setting sun, casting a soft glow on the scene. The scene appears to be real-life footage&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>negative_prompt = <span class="hljs-string">&quot;worst quality, inconsistent motion, blurry, jittery, distorted&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>video = pipe(
<span class="hljs-meta">... </span> prompt=prompt,
<span class="hljs-meta">... </span> negative_prompt=negative_prompt,
<span class="hljs-meta">... </span> width=<span class="hljs-number">1280</span>,
<span class="hljs-meta">... </span> height=<span class="hljs-number">736</span>,
<span class="hljs-meta">... </span> num_frames=<span class="hljs-number">121</span>,
<span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>,
<span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>]
<span class="hljs-meta">&gt;&gt;&gt; </span>export_to_video(video, <span class="hljs-string">&quot;output.mp4&quot;</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),i(m);var B=e(m,2),z=n(B);t(z,{name:"encode_prompt",anchor:"diffusers.MotifVideoPipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video.py#L247",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"max_sequence_length",val:": int = 512"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts to be encoded.`,name:"prompt"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts not to guide the image generation. If not defined, one has to pass
<code>negative_prompt_embeds</code> instead. Ignored when not using guidance (i.e., ignored if <code>guidance_scale</code> is
less than <code>1</code>).`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) &#x2014;
Number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt
weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input
argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to 512) &#x2014;
Maximum sequence length for the tokenizer.`,name:"max_sequence_length"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.device",description:`<strong>device</strong> (<code>torch.device</code>, <em>optional</em>) &#x2014;
Device to place tensors on.`,name:"device"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.dtype",description:`<strong>dtype</strong> (<code>torch.dtype</code>, <em>optional</em>) &#x2014;
Data type for tensors.`,name:"dtype"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>A tuple containing:</p>
<ul>
<li><code>prompt_embeds</code>: The text embeddings for the positive prompt</li>
<li><code>negative_prompt_embeds</code>: The text embeddings for the negative prompt (None if not using guidance)</li>
<li><code>prompt_attention_mask</code>: The attention mask for the positive prompt</li>
<li><code>negative_prompt_attention_mask</code>: The attention mask for the negative prompt (None if not using
guidance)</li>
</ul>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><code>tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]</code></p>
`}),l(2),i(B),i(c);var k=e(c,2);o(k,{title:"MotifVideoImage2VideoPipeline",local:"diffusers.MotifVideoImage2VideoPipeline",headingTag:"h2"});var f=e(k,2),C=n(f);t(C,{name:"class diffusers.MotifVideoImage2VideoPipeline",anchor:"diffusers.MotifVideoImage2VideoPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video_image2video.py#L157",parameters:[{name:"scheduler",val:": SchedulerMixin"},{name:"vae",val:": AutoencoderKLWan"},{name:"text_encoder",val:": T5Gemma2Encoder"},{name:"tokenizer",val:": PreTrainedTokenizerBase"},{name:"transformer",val:": MotifVideoTransformer3DModel"},{name:"guider",val:": BaseGuidance"},{name:"feature_extractor",val:": SiglipImageProcessorPil"}],parametersDescription:[{anchor:"diffusers.MotifVideoImage2VideoPipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/motif_video_transformer_3d#diffusers.MotifVideoTransformer3DModel">MotifVideoTransformer3DModel</a>) &#x2014;
Conditional Transformer architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14192/en/api/schedulers/overview#diffusers.SchedulerMixin">SchedulerMixin</a>) &#x2014;
A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents. Should be an
instance of a class inheriting from <code>SchedulerMixin</code>, such as <a href="/docs/diffusers/pr_14192/en/api/schedulers/multistep_dpm_solver#diffusers.DPMSolverMultistepScheduler">DPMSolverMultistepScheduler</a>. If not
provided, uses the scheduler attached to the pretrained model.`,name:"scheduler"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/autoencoder_kl_wan#diffusers.AutoencoderKLWan">AutoencoderKLWan</a>) &#x2014;
Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>T5Gemma2Encoder</code>) &#x2014;
Primary text encoder for encoding text prompts into embeddings.`,name:"text_encoder"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.tokenizer",description:`<strong>tokenizer</strong> (<code>PreTrainedTokenizerBase</code>) &#x2014;
Tokenizer corresponding to the primary text encoder.`,name:"tokenizer"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.feature_extractor",description:`<strong>feature_extractor</strong> (<code>SiglipImageProcessor</code>) &#x2014;
Image processor for the SigLIP vision encoder.`,name:"feature_extractor"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.BaseGuidance">BaseGuidance</a>) &#x2014;
The guidance method to use. Should be an instance of a class inheriting from <code>BaseGuidance</code>, such as
<a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>, <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.AdaptiveProjectedGuidance">AdaptiveProjectedGuidance</a>, or <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.SkipLayerGuidance">SkipLayerGuidance</a>. If not provided,
defaults to <code>ClassifierFreeGuidance</code>.`,name:"guider"}]});var g=e(C,6),P=n(g);t(P,{name:"__call__",anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video_image2video.py#L620",parameters:[{name:"image",val:": typing.Union[PIL.Image.Image, numpy.ndarray, torch.Tensor, list[PIL.Image.Image], list[numpy.ndarray], list[torch.Tensor]]"},{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"height",val:": int = 736"},{name:"width",val:": int = 1280"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"timesteps",val:": typing.Optional[typing.List[int]] = None"},{name:"num_videos_per_prompt",val:": typing.Optional[int] = 1"},{name:"generator",val:": typing.Union[torch.Generator, typing.List[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": typing.Optional[str] = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": typing.Optional[typing.Dict[str, typing.Any]] = None"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int, typing.Dict], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": typing.List[str] = ['latents']"},{name:"max_sequence_length",val:": int = 512"}],parametersDescription:[{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.image",description:`<strong>image</strong> (<code>PipelineImageInput</code>) &#x2014;
The input image to use as the first frame for video generation.`,name:"image"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>) &#x2014;
The prompt or prompts to guide the video generation.`,name:"prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts not to guide the video generation.`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, defaults to <code>736</code>) &#x2014;
The height in pixels of the generated video.`,name:"height"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, defaults to <code>1280</code>) &#x2014;
The width in pixels of the generated video.`,name:"width"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) &#x2014;
The number of video frames to generate.`,name:"num_frames"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 50) &#x2014;
The number of denoising steps.`,name:"num_inference_steps"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<code>List[int]</code>, <em>optional</em>) &#x2014;
Custom timesteps to use for the denoising process.`,name:"timesteps"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) &#x2014;
The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>List[torch.Generator]</code>, <em>optional</em>) &#x2014;
PyTorch Generator object(s) for deterministic generation.`,name:"generator"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated noisy latents.`,name:"latents"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>&quot;pil&quot;</code>) &#x2014;
The output format of the generated video.`,name:"output_type"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) &#x2014;
Whether or not to return a <a href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput">~MotifVideoPipelineOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) &#x2014;
Arguments passed to the attention processor.`,name:"attention_kwargs"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <em>optional</em>) &#x2014;
A function or subclass of <code>PipelineCallback</code> called at the end of each denoising step.`,name:"callback_on_step_end"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>List</code>, <em>optional</em>) &#x2014;
The list of tensor inputs for the <code>callback_on_step_end</code> function.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to <code>512</code>) &#x2014;
Maximum sequence length for the tokenizer.`,name:"max_sequence_length"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>If <code>return_dict</code> is <code>True</code>, <a
href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput"
>~MotifVideoPipelineOutput</a> is returned, otherwise a <code>tuple</code> is returned
where the first element is a list of generated video frames.</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><a
href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput"
>~MotifVideoPipelineOutput</a> or <code>tuple</code></p>
`});var F=e(P,4);Q(F,{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.example",children:(s,d)=>{var r=R(),_=e(h(r),2);a(_,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwUElMJTIwaW1wb3J0JTIwSW1hZ2UlMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb0ltYWdlMlZpZGVvUGlwZWxpbmUlMEFmcm9tJTIwZGlmZnVzZXJzLnV0aWxzJTIwaW1wb3J0JTIwZXhwb3J0X3RvX3ZpZGVvJTJDJTIwbG9hZF9pbWFnZSUwQSUwQSUyMyUyMExvYWQlMjB0aGUlMjBNb3RpZi1WaWRlbyUyMGltYWdlLXRvLXZpZGVvJTIwcGlwZWxpbmUlMEFtb3RpZl92aWRlb19tb2RlbF9pZCUyMCUzRCUyMCUyMk1vdGlmLVRlY2hub2xvZ2llcyUyRk1vdGlmLVZpZGVvLTJCJTIyJTBBcGlwZSUyMCUzRCUyME1vdGlmVmlkZW9JbWFnZTJWaWRlb1BpcGVsaW5lLmZyb21fcHJldHJhaW5lZChtb3RpZl92aWRlb19tb2RlbF9pZCUyQyUyMHRvcmNoX2R0eXBlJTNEdG9yY2guYmZsb2F0MTYpJTBBcGlwZS50byglMjJjdWRhJTIyKSUwQSUwQSUyMyUyMExvYWQlMjBhbiUyMGltYWdlJTBBaW1hZ2UlMjAlM0QlMjBsb2FkX2ltYWdlKCUwQSUyMCUyMCUyMCUyMCUyMmh0dHBzJTNBJTJGJTJGaHVnZ2luZ2ZhY2UuY28lMkZkYXRhc2V0cyUyRmh1Z2dpbmdmYWNlJTJGZG9jdW1lbnRhdGlvbi1pbWFnZXMlMkZyZXNvbHZlJTJGbWFpbiUyRmRpZmZ1c2VycyUyRmFzdHJvbmF1dC5wbmclMjIlMEEpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQW4lMjBhc3Ryb25hdXQlMjBpcyUyMHdhbGtpbmclMjBvbiUyMHRoZSUyMG1vb24lMjBzdXJmYWNlJTJDJTIwa2lja2luZyUyMHVwJTIwZHVzdCUyMHdpdGglMjBlYWNoJTIwc3RlcCUyMiUwQW5lZ2F0aXZlX3Byb21wdCUyMCUzRCUyMCUyMndvcnN0JTIwcXVhbGl0eSUyQyUyMGluY29uc2lzdGVudCUyMG1vdGlvbiUyQyUyMGJsdXJyeSUyQyUyMGppdHRlcnklMkMlMjBkaXN0b3J0ZWQlMjIlMEElMEF2aWRlbyUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwaW1hZ2UlM0RpbWFnZSUyQyUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMG5lZ2F0aXZlX3Byb21wdCUzRG5lZ2F0aXZlX3Byb21wdCUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEMTI4MCUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRDczNiUyQyUwQSUyMCUyMCUyMCUyMG51bV9mcmFtZXMlM0QxMjElMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">import</span> torch
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoImage2VideoPipeline
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_image
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># Load the Motif-Video image-to-video pipeline</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>motif_video_model_id = <span class="hljs-string">&quot;Motif-Technologies/Motif-Video-2B&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe = MotifVideoImage2VideoPipeline.from_pretrained(motif_video_model_id, torch_dtype=torch.bfloat16)
<span class="hljs-meta">&gt;&gt;&gt; </span>pipe.to(<span class="hljs-string">&quot;cuda&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># Load an image</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>image = load_image(
<span class="hljs-meta">... </span> <span class="hljs-string">&quot;https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.png&quot;</span>
<span class="hljs-meta">... </span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>prompt = <span class="hljs-string">&quot;An astronaut is walking on the moon surface, kicking up dust with each step&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>negative_prompt = <span class="hljs-string">&quot;worst quality, inconsistent motion, blurry, jittery, distorted&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>video = pipe(
<span class="hljs-meta">... </span> image=image,
<span class="hljs-meta">... </span> prompt=prompt,
<span class="hljs-meta">... </span> negative_prompt=negative_prompt,
<span class="hljs-meta">... </span> width=<span class="hljs-number">1280</span>,
<span class="hljs-meta">... </span> height=<span class="hljs-number">736</span>,
<span class="hljs-meta">... </span> num_frames=<span class="hljs-number">121</span>,
<span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>,
<span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>]
<span class="hljs-meta">&gt;&gt;&gt; </span>export_to_video(video, <span class="hljs-string">&quot;output.mp4&quot;</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),i(g);var W=e(g,2),q=n(W);t(q,{name:"encode_prompt",anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video_image2video.py#L259",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"max_sequence_length",val:": int = 512"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts to be encoded.`,name:"prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) &#x2014;
The prompt or prompts not to guide the image generation. If not defined, one has to pass
<code>negative_prompt_embeds</code> instead. Ignored when not using guidance (i.e., ignored if <code>guidance_scale</code> is
less than <code>1</code>).`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) &#x2014;
Number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt
weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input
argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) &#x2014;
Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to 512) &#x2014;
Maximum sequence length for the tokenizer.`,name:"max_sequence_length"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.device",description:`<strong>device</strong> (<code>torch.device</code>, <em>optional</em>) &#x2014;
Device to place tensors on.`,name:"device"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.dtype",description:`<strong>dtype</strong> (<code>torch.dtype</code>, <em>optional</em>) &#x2014;
Data type for tensors.`,name:"dtype"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>A tuple containing:</p>
<ul>
<li><code>prompt_embeds</code>: The text embeddings for the positive prompt</li>
<li><code>negative_prompt_embeds</code>: The text embeddings for the negative prompt (None if not using guidance)</li>
<li><code>prompt_attention_mask</code>: The attention mask for the positive prompt</li>
<li><code>negative_prompt_attention_mask</code>: The attention mask for the negative prompt (None if not using
guidance)</li>
</ul>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><code>tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]</code></p>
`}),l(2),i(W),i(f);var N=e(f,2);o(N,{title:"MotifVideoPipelineOutput",local:"diffusers.MotifVideoPipelineOutput",headingTag:"h2"});var u=e(N,2),A=n(u);t(A,{name:"class diffusers.MotifVideoPipelineOutput",anchor:"diffusers.MotifVideoPipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_output.py#L9",parameters:[{name:"frames",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipelineOutput.frames",description:`<strong>frames</strong> (<code>torch.Tensor</code>, <code>np.ndarray</code>, or List[List[PIL.Image.Image]]) &#x2014;
List of video outputs - It can be a nested list of length <code>batch_size,</code> with each sub-list containing
denoised PIL image sequences of length <code>num_frames.</code> It can also be a NumPy array or Torch tensor of shape
<code>(batch_size, num_frames, channels, height, width)</code>.`,name:"frames"}]}),l(2),i(u);var H=e(u,2);O(H,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/motif_video.md"}),l(2),p(X,y),oe()}export{pe as component};

Xet Storage Details

Size:
54.3 kB
·
Xet hash:
01d91a3222f4f1b74810956d299ad0e282f2a03035115fa122f5adcdf0cccd88

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.