Buckets:
| import"../chunks/DsnmJJEf.js";import{i as L,h as D,C as Y,H as o,a,D as t,E as O,s as K}from"../chunks/BtE7mKSK.js";import{p as $,o as ee,s as e,f as h,a as p,b as oe,c as n,d as M,r as i,n as l}from"../chunks/jDjavuwI.js";import{E as Q}from"../chunks/SrSJA0zO.js";const te='{"title":"Motif-Video","local":"motif-video","sections":[{"title":"Text-to-Video Generation","local":"text-to-video-generation","sections":[],"depth":2},{"title":"Image-to-Video Generation","local":"image-to-video-generation","sections":[{"title":"Memory-efficient Inference","local":"memory-efficient-inference","sections":[],"depth":3}],"depth":2},{"title":"MotifVideoPipeline","local":"diffusers.MotifVideoPipeline","sections":[],"depth":2},{"title":"MotifVideoImage2VideoPipeline","local":"diffusers.MotifVideoImage2VideoPipeline","sections":[],"depth":2},{"title":"MotifVideoPipelineOutput","local":"diffusers.MotifVideoPipelineOutput","sections":[],"depth":2}],"depth":1}';var ne=M('<meta name="hf:doc:metadata"/>'),R=M("<p>Examples:</p> <!>",1),ie=M(`<p></p> <!> <!> <p><a href="https://arxiv.org/abs/2604.16503" rel="nofollow">Technical Report</a></p> <p>Motif-Video is a 2B parameter diffusion transformer designed for text-to-video and image-to-video generation. It features a three-stage architecture with 12 dual-stream + 16 single-stream + 8 DDT decoder layers, Shared Cross-Attention for stable text-video alignment under long video sequences, T5Gemma2 text encoder, and rectified flow matching for velocity prediction.</p> <p align="center"><img src="https://huggingface.co/Motif-Technologies/Motif-Video-2B/resolve/main/assets/architecture.png" width="90%" alt="Motif-Video architecture"/></p> <!> <p>Use <code>MotifVideoPipeline</code> for text-to-video generation:</p> <!> <!> <p>Use <code>MotifVideoImage2VideoPipeline</code> for image-to-video generation:</p> <!> <!> <p>For GPUs with less than 30GB VRAM (e.g., RTX 4090), use model CPU offloading:</p> <!> <!> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for text-to-video generation using Motif-Video.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14192/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods | |
| implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for text-to-video generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encodes the prompt into text encoder hidden states.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for image-to-video generation using Motif-Video with first frame conditioning.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14192/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods | |
| implemented for all pipelines (downloading, saving, running on a particular device, etc.).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The call function to the pipeline for image-to-video generation.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encodes the prompt into text encoder hidden states.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for Motif-Video pipelines.</p></div> <!> <p></p>`,1);function pe(X,E){$(E,!1),ee(()=>{new URLSearchParams(window.location.search).get("fw")}),L();var y=ie();D("mz3w07",s=>{var d=ne();K(d,"content",te),p(s,d)});var v=e(h(y),2);Y(v,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var b=e(v,2);o(b,{title:"Motif-Video",local:"motif-video",headingTag:"h1"});var T=e(b,8);o(T,{title:"Text-to-Video Generation",local:"text-to-video-generation",headingTag:"h2"});var U=e(T,4);a(U,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb1BpcGVsaW5lJTBBZnJvbSUyMGRpZmZ1c2Vycy51dGlscyUyMGltcG9ydCUyMGV4cG9ydF90b192aWRlbyUwQSUwQSUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMk1vdGlmLVRlY2hub2xvZ2llcyUyRk1vdGlmLVZpZGVvLTJCJTIyJTJDJTBBJTIwJTIwJTIwJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiUyQyUwQSklMEFwaXBlLnRvKCUyMmN1ZGElMjIpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMHdvbWFuJTIwd2l0aCUyMGxvbmclMjBicm93biUyMGhhaXIlMjBhbmQlMjBsaWdodCUyMHNraW4lMjBzbWlsZXMlMjBhdCUyMGFub3RoZXIlMjB3b21hbiUyMHdpdGglMjBsb25nJTIwYmxvbmRlJTIwaGFpci4lMjIlMEFuZWdhdGl2ZV9wcm9tcHQlMjAlM0QlMjAlMjJ3b3JzdCUyMHF1YWxpdHklMkMlMjBpbmNvbnNpc3RlbnQlMjBtb3Rpb24lMkMlMjBibHVycnklMkMlMjBqaXR0ZXJ5JTJDJTIwZGlzdG9ydGVkJTIyJTBBJTBBdmlkZW8lMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMG5lZ2F0aXZlX3Byb21wdCUzRG5lZ2F0aXZlX3Byb21wdCUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEMTI4MCUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRDczNiUyQyUwQSUyMCUyMCUyMCUyMG51bV9mcmFtZXMlM0QxMjElMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoPipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| pipe = MotifVideoPipeline.from_pretrained( | |
| <span class="hljs-string">"Motif-Technologies/Motif-Video-2B"</span>, | |
| torch_dtype=torch.bfloat16, | |
| ) | |
| pipe.to(<span class="hljs-string">"cuda"</span>) | |
| prompt = <span class="hljs-string">"A woman with long brown hair and light skin smiles at another woman with long blonde hair."</span> | |
| negative_prompt = <span class="hljs-string">"worst quality, inconsistent motion, blurry, jittery, distorted"</span> | |
| video = pipe( | |
| prompt=prompt, | |
| negative_prompt=negative_prompt, | |
| width=<span class="hljs-number">1280</span>, | |
| height=<span class="hljs-number">736</span>, | |
| num_frames=<span class="hljs-number">121</span>, | |
| num_inference_steps=<span class="hljs-number">50</span>, | |
| ).frames[<span class="hljs-number">0</span>] | |
| export_to_video(video, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1});var w=e(U,2);o(w,{title:"Image-to-Video Generation",local:"image-to-video-generation",headingTag:"h2"});var J=e(w,4);a(J,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb0ltYWdlMlZpZGVvUGlwZWxpbmUlMEFmcm9tJTIwZGlmZnVzZXJzLnV0aWxzJTIwaW1wb3J0JTIwZXhwb3J0X3RvX3ZpZGVvJTJDJTIwbG9hZF9pbWFnZSUwQSUwQSUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvSW1hZ2UyVmlkZW9QaXBlbGluZS5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwJTIyTW90aWYtVGVjaG5vbG9naWVzJTJGTW90aWYtVmlkZW8tMkIlMjIlMkMlMEElMjAlMjAlMjAlMjB0b3JjaF9kdHlwZSUzRHRvcmNoLmJmbG9hdDE2JTJDJTBBKSUwQXBpcGUudG8oJTIyY3VkYSUyMiklMEElMEFpbWFnZSUyMCUzRCUyMGxvYWRfaW1hZ2UoJTIyaW5wdXRfaW1hZ2UucG5nJTIyKSUwQXByb21wdCUyMCUzRCUyMCUyMkElMjBjaW5lbWF0aWMlMjBzY2VuZSUyMHdpdGglMjB2aXZpZCUyMGNvbG9ycy4lMjIlMEFuZWdhdGl2ZV9wcm9tcHQlMjAlM0QlMjAlMjJ3b3JzdCUyMHF1YWxpdHklMkMlMjBibHVycnklMkMlMjBqaXR0ZXJ5JTJDJTIwZGlzdG9ydGVkJTIyJTBBJTBBdmlkZW8lMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMGltYWdlJTNEaW1hZ2UlMkMlMEElMjAlMjAlMjAlMjBwcm9tcHQlM0Rwcm9tcHQlMkMlMEElMjAlMjAlMjAlMjBuZWdhdGl2ZV9wcm9tcHQlM0RuZWdhdGl2ZV9wcm9tcHQlMkMlMEElMjAlMjAlMjAlMjB3aWR0aCUzRDEyODAlMkMlMEElMjAlMjAlMjAlMjBoZWlnaHQlM0Q3MzYlMkMlMEElMjAlMjAlMjAlMjBudW1fZnJhbWVzJTNEMTIxJTJDJTBBJTIwJTIwJTIwJTIwbnVtX2luZmVyZW5jZV9zdGVwcyUzRDUwJTJDJTBBKS5mcmFtZXMlNUIwJTVEJTBBZXhwb3J0X3RvX3ZpZGVvKHZpZGVvJTJDJTIwJTIyaTJ2X291dHB1dC5tcDQlMjIlMkMlMjBmcHMlM0QyNCk=",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoImage2VideoPipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_image | |
| pipe = MotifVideoImage2VideoPipeline.from_pretrained( | |
| <span class="hljs-string">"Motif-Technologies/Motif-Video-2B"</span>, | |
| torch_dtype=torch.bfloat16, | |
| ) | |
| pipe.to(<span class="hljs-string">"cuda"</span>) | |
| image = load_image(<span class="hljs-string">"input_image.png"</span>) | |
| prompt = <span class="hljs-string">"A cinematic scene with vivid colors."</span> | |
| negative_prompt = <span class="hljs-string">"worst quality, blurry, jittery, distorted"</span> | |
| video = pipe( | |
| image=image, | |
| prompt=prompt, | |
| negative_prompt=negative_prompt, | |
| width=<span class="hljs-number">1280</span>, | |
| height=<span class="hljs-number">736</span>, | |
| num_frames=<span class="hljs-number">121</span>, | |
| num_inference_steps=<span class="hljs-number">50</span>, | |
| ).frames[<span class="hljs-number">0</span>] | |
| export_to_video(video, <span class="hljs-string">"i2v_output.mp4"</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1});var V=e(J,2);o(V,{title:"Memory-efficient Inference",local:"memory-efficient-inference",headingTag:"h3"});var j=e(V,4);a(j,{code:"ZXhwb3J0JTIwUFlUT1JDSF9DVURBX0FMTE9DX0NPTkYlM0RleHBhbmRhYmxlX3NlZ21lbnRzJTNBVHJ1ZQ==",highlighted:'<span class="hljs-built_in">export</span> PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True',lang:"bash",wrap:!1});var I=e(j,2);a(I,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb1BpcGVsaW5lJTBBZnJvbSUyMGRpZmZ1c2Vycy51dGlscyUyMGltcG9ydCUyMGV4cG9ydF90b192aWRlbyUwQSUwQSUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMk1vdGlmLVRlY2hub2xvZ2llcyUyRk1vdGlmLVZpZGVvLTJCJTIyJTJDJTBBJTIwJTIwJTIwJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiUyQyUwQSklMEFwaXBlLmVuYWJsZV9tb2RlbF9jcHVfb2ZmbG9hZCgpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMHdvbWFuJTIwd2l0aCUyMGxvbmclMjBicm93biUyMGhhaXIlMjBhbmQlMjBsaWdodCUyMHNraW4lMjBzbWlsZXMlMjBhdCUyMGFub3RoZXIlMjB3b21hbiUyMHdpdGglMjBsb25nJTIwYmxvbmRlJTIwaGFpci4lMjIlMEFuZWdhdGl2ZV9wcm9tcHQlMjAlM0QlMjAlMjJ3b3JzdCUyMHF1YWxpdHklMkMlMjBpbmNvbnNpc3RlbnQlMjBtb3Rpb24lMkMlMjBibHVycnklMkMlMjBqaXR0ZXJ5JTJDJTIwZGlzdG9ydGVkJTIyJTBBJTBBdmlkZW8lMjAlM0QlMjBwaXBlKCUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMG5lZ2F0aXZlX3Byb21wdCUzRG5lZ2F0aXZlX3Byb21wdCUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEMTI4MCUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRDczNiUyQyUwQSUyMCUyMCUyMCUyMG51bV9mcmFtZXMlM0QxMjElMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoPipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| pipe = MotifVideoPipeline.from_pretrained( | |
| <span class="hljs-string">"Motif-Technologies/Motif-Video-2B"</span>, | |
| torch_dtype=torch.bfloat16, | |
| ) | |
| pipe.enable_model_cpu_offload() | |
| prompt = <span class="hljs-string">"A woman with long brown hair and light skin smiles at another woman with long blonde hair."</span> | |
| negative_prompt = <span class="hljs-string">"worst quality, inconsistent motion, blurry, jittery, distorted"</span> | |
| video = pipe( | |
| prompt=prompt, | |
| negative_prompt=negative_prompt, | |
| width=<span class="hljs-number">1280</span>, | |
| height=<span class="hljs-number">736</span>, | |
| num_frames=<span class="hljs-number">121</span>, | |
| num_inference_steps=<span class="hljs-number">50</span>, | |
| ).frames[<span class="hljs-number">0</span>] | |
| export_to_video(video, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1});var Z=e(I,2);o(Z,{title:"MotifVideoPipeline",local:"diffusers.MotifVideoPipeline",headingTag:"h2"});var c=e(Z,2),x=n(c);t(x,{name:"class diffusers.MotifVideoPipeline",anchor:"diffusers.MotifVideoPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video.py#L148",parameters:[{name:"scheduler",val:": SchedulerMixin"},{name:"vae",val:": AutoencoderKLWan"},{name:"text_encoder",val:": T5Gemma2Encoder"},{name:"tokenizer",val:": PreTrainedTokenizerBase"},{name:"transformer",val:": MotifVideoTransformer3DModel"},{name:"guider",val:": BaseGuidance"},{name:"feature_extractor",val:": typing.Optional[transformers.models.siglip.image_processing_pil_siglip.SiglipImageProcessorPil] = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/motif_video_transformer_3d#diffusers.MotifVideoTransformer3DModel">MotifVideoTransformer3DModel</a>) — | |
| Conditional Transformer architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.MotifVideoPipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14192/en/api/schedulers/overview#diffusers.SchedulerMixin">SchedulerMixin</a>) — | |
| A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents. Should be an | |
| instance of a class inheriting from <code>SchedulerMixin</code>, such as <a href="/docs/diffusers/pr_14192/en/api/schedulers/multistep_dpm_solver#diffusers.DPMSolverMultistepScheduler">DPMSolverMultistepScheduler</a>. If not | |
| provided, uses the scheduler attached to the pretrained model.`,name:"scheduler"},{anchor:"diffusers.MotifVideoPipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/autoencoder_kl_wan#diffusers.AutoencoderKLWan">AutoencoderKLWan</a>) — | |
| Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.MotifVideoPipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>T5Gemma2Encoder</code>) — | |
| Primary text encoder for encoding text prompts into embeddings.`,name:"text_encoder"},{anchor:"diffusers.MotifVideoPipeline.tokenizer",description:`<strong>tokenizer</strong> (<code>PreTrainedTokenizerBase</code>) — | |
| Tokenizer corresponding to the primary text encoder.`,name:"tokenizer"},{anchor:"diffusers.MotifVideoPipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.BaseGuidance">BaseGuidance</a>) — | |
| The guidance method to use. Should be an instance of a class inheriting from <code>BaseGuidance</code>, such as | |
| <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>, <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.AdaptiveProjectedGuidance">AdaptiveProjectedGuidance</a>, or <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.SkipLayerGuidance">SkipLayerGuidance</a>. If not provided, | |
| defaults to <code>ClassifierFreeGuidance</code>.`,name:"guider"}]});var m=e(x,6),G=n(m);t(G,{name:"__call__",anchor:"diffusers.MotifVideoPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video.py#L492",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"height",val:": int = 736"},{name:"width",val:": int = 1280"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"timesteps",val:": typing.Optional[typing.List[int]] = None"},{name:"num_videos_per_prompt",val:": typing.Optional[int] = 1"},{name:"generator",val:": typing.Union[torch.Generator, typing.List[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": typing.Optional[str] = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": typing.Optional[typing.Dict[str, typing.Any]] = None"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int, typing.Dict], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": typing.List[str] = ['latents']"},{name:"max_sequence_length",val:": int = 512"},{name:"vae_batch_size",val:": int | None = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| The prompt or prompts to guide the video generation. If not defined, one has to pass <code>prompt_embeds</code>.`,name:"prompt"},{anchor:"diffusers.MotifVideoPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| The prompt or prompts not to guide the video generation. If not defined, one has to pass | |
| <code>negative_prompt_embeds</code> instead. Ignored when not using guidance.`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoPipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, defaults to <code>736</code>) — | |
| The height in pixels of the generated video.`,name:"height"},{anchor:"diffusers.MotifVideoPipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, defaults to <code>1280</code>) — | |
| The width in pixels of the generated video.`,name:"width"},{anchor:"diffusers.MotifVideoPipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) — | |
| The number of video frames to generate.`,name:"num_frames"},{anchor:"diffusers.MotifVideoPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 50) — | |
| The number of denoising steps. More denoising steps usually lead to a higher quality video at the | |
| expense of slower inference.`,name:"num_inference_steps"},{anchor:"diffusers.MotifVideoPipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| Custom timesteps to use for the denoising process.`,name:"timesteps"},{anchor:"diffusers.MotifVideoPipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoPipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>List[torch.Generator]</code>, <em>optional</em>) — | |
| PyTorch Generator object(s) for deterministic generation.`,name:"generator"},{anchor:"diffusers.MotifVideoPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated noisy latents.`,name:"latents"},{anchor:"diffusers.MotifVideoPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.__call__.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.__call__.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"pil"</code>) — | |
| The output format of the generated video. Choose between <code>"pil"</code>, <code>"np"</code>, or <code>"latent"</code>.`,name:"output_type"},{anchor:"diffusers.MotifVideoPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to return a <a href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput">~MotifVideoPipelineOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.MotifVideoPipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) — | |
| Arguments passed to the attention processor.`,name:"attention_kwargs"},{anchor:"diffusers.MotifVideoPipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <em>optional</em>) — | |
| A function or subclass of <code>PipelineCallback</code> or <code>MultiPipelineCallbacks</code> called at the end of each | |
| denoising step.`,name:"callback_on_step_end"},{anchor:"diffusers.MotifVideoPipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>List</code>, <em>optional</em>) — | |
| The list of tensor inputs for the <code>callback_on_step_end</code> function.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.MotifVideoPipeline.__call__.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to <code>512</code>) — | |
| Maximum sequence length for the tokenizer.`,name:"max_sequence_length"},{anchor:"diffusers.MotifVideoPipeline.__call__.vae_batch_size",description:`<strong>vae_batch_size</strong> (<code>int</code>, <em>optional</em>) — | |
| Batch size for VAE decoding. If provided and latents batch size is larger, VAE decoding will be done in | |
| chunks.`,name:"vae_batch_size"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If <code>return_dict</code> is <code>True</code>, <a | |
| href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput" | |
| >~MotifVideoPipelineOutput</a> is returned, otherwise a <code>tuple</code> is returned | |
| where the first element is a list of generated video frames.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput" | |
| >~MotifVideoPipelineOutput</a> or <code>tuple</code></p> | |
| `});var S=e(G,4);Q(S,{anchor:"diffusers.MotifVideoPipeline.__call__.example",children:(s,d)=>{var r=R(),_=e(h(r),2);a(_,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb1BpcGVsaW5lJTBBZnJvbSUyMGRpZmZ1c2Vycy51dGlscyUyMGltcG9ydCUyMGV4cG9ydF90b192aWRlbyUwQSUwQSUyMyUyMExvYWQlMjB0aGUlMjBNb3RpZi1WaWRlbyUyMHBpcGVsaW5lJTBBbW90aWZfdmlkZW9fbW9kZWxfaWQlMjAlM0QlMjAlMjJNb3RpZi1UZWNobm9sb2dpZXMlMkZNb3RpZi1WaWRlby0yQiUyMiUwQXBpcGUlMjAlM0QlMjBNb3RpZlZpZGVvUGlwZWxpbmUuZnJvbV9wcmV0cmFpbmVkKG1vdGlmX3ZpZGVvX21vZGVsX2lkJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiklMEFwaXBlLnRvKCUyMmN1ZGElMjIpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMHdvbWFuJTIwd2l0aCUyMGxvbmclMjBicm93biUyMGhhaXIlMjBhbmQlMjBsaWdodCUyMHNraW4lMjBzbWlsZXMlMjBhdCUyMGFub3RoZXIlMjB3b21hbiUyMHdpdGglMjBsb25nJTIwYmxvbmRlJTIwaGFpci4lMjBUaGUlMjB3b21hbiUyMHdpdGglMjBicm93biUyMGhhaXIlMjB3ZWFycyUyMGElMjBibGFjayUyMGphY2tldCUyMGFuZCUyMGhhcyUyMGElMjBzbWFsbCUyQyUyMGJhcmVseSUyMG5vdGljZWFibGUlMjBtb2xlJTIwb24lMjBoZXIlMjByaWdodCUyMGNoZWVrLiUyMFRoZSUyMGNhbWVyYSUyMGFuZ2xlJTIwaXMlMjBhJTIwY2xvc2UtdXAlMkMlMjBmb2N1c2VkJTIwb24lMjB0aGUlMjB3b21hbiUyMHdpdGglMjBicm93biUyMGhhaXIncyUyMGZhY2UuJTIwVGhlJTIwbGlnaHRpbmclMjBpcyUyMHdhcm0lMjBhbmQlMjBuYXR1cmFsJTJDJTIwbGlrZWx5JTIwZnJvbSUyMHRoZSUyMHNldHRpbmclMjBzdW4lMkMlMjBjYXN0aW5nJTIwYSUyMHNvZnQlMjBnbG93JTIwb24lMjB0aGUlMjBzY2VuZS4lMjBUaGUlMjBzY2VuZSUyMGFwcGVhcnMlMjB0byUyMGJlJTIwcmVhbC1saWZlJTIwZm9vdGFnZSUyMiUwQW5lZ2F0aXZlX3Byb21wdCUyMCUzRCUyMCUyMndvcnN0JTIwcXVhbGl0eSUyQyUyMGluY29uc2lzdGVudCUyMG1vdGlvbiUyQyUyMGJsdXJyeSUyQyUyMGppdHRlcnklMkMlMjBkaXN0b3J0ZWQlMjIlMEElMEF2aWRlbyUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwcHJvbXB0JTNEcHJvbXB0JTJDJTBBJTIwJTIwJTIwJTIwbmVnYXRpdmVfcHJvbXB0JTNEbmVnYXRpdmVfcHJvbXB0JTJDJTBBJTIwJTIwJTIwJTIwd2lkdGglM0QxMjgwJTJDJTBBJTIwJTIwJTIwJTIwaGVpZ2h0JTNENzM2JTJDJTBBJTIwJTIwJTIwJTIwbnVtX2ZyYW1lcyUzRDEyMSUyQyUwQSUyMCUyMCUyMCUyMG51bV9pbmZlcmVuY2Vfc3RlcHMlM0Q1MCUyQyUwQSkuZnJhbWVzJTVCMCU1RCUwQWV4cG9ydF90b192aWRlbyh2aWRlbyUyQyUyMCUyMm91dHB1dC5tcDQlMjIlMkMlMjBmcHMlM0QyNCk=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoPipeline | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Load the Motif-Video pipeline</span> | |
| <span class="hljs-meta">>>> </span>motif_video_model_id = <span class="hljs-string">"Motif-Technologies/Motif-Video-2B"</span> | |
| <span class="hljs-meta">>>> </span>pipe = MotifVideoPipeline.from_pretrained(motif_video_model_id, torch_dtype=torch.bfloat16) | |
| <span class="hljs-meta">>>> </span>pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span>prompt = <span class="hljs-string">"A woman with long brown hair and light skin smiles at another woman with long blonde hair. The woman with brown hair wears a black jacket and has a small, barely noticeable mole on her right cheek. The camera angle is a close-up, focused on the woman with brown hair's face. The lighting is warm and natural, likely from the setting sun, casting a soft glow on the scene. The scene appears to be real-life footage"</span> | |
| <span class="hljs-meta">>>> </span>negative_prompt = <span class="hljs-string">"worst quality, inconsistent motion, blurry, jittery, distorted"</span> | |
| <span class="hljs-meta">>>> </span>video = pipe( | |
| <span class="hljs-meta">... </span> prompt=prompt, | |
| <span class="hljs-meta">... </span> negative_prompt=negative_prompt, | |
| <span class="hljs-meta">... </span> width=<span class="hljs-number">1280</span>, | |
| <span class="hljs-meta">... </span> height=<span class="hljs-number">736</span>, | |
| <span class="hljs-meta">... </span> num_frames=<span class="hljs-number">121</span>, | |
| <span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>, | |
| <span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>] | |
| <span class="hljs-meta">>>> </span>export_to_video(video, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),i(m);var B=e(m,2),z=n(B);t(z,{name:"encode_prompt",anchor:"diffusers.MotifVideoPipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video.py#L247",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"max_sequence_length",val:": int = 512"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| The prompt or prompts to be encoded.`,name:"prompt"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| The prompt or prompts not to guide the image generation. If not defined, one has to pass | |
| <code>negative_prompt_embeds</code> instead. Ignored when not using guidance (i.e., ignored if <code>guidance_scale</code> is | |
| less than <code>1</code>).`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt | |
| weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input | |
| argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to 512) — | |
| Maximum sequence length for the tokenizer.`,name:"max_sequence_length"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.device",description:`<strong>device</strong> (<code>torch.device</code>, <em>optional</em>) — | |
| Device to place tensors on.`,name:"device"},{anchor:"diffusers.MotifVideoPipeline.encode_prompt.dtype",description:`<strong>dtype</strong> (<code>torch.dtype</code>, <em>optional</em>) — | |
| Data type for tensors.`,name:"dtype"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A tuple containing:</p> | |
| <ul> | |
| <li><code>prompt_embeds</code>: The text embeddings for the positive prompt</li> | |
| <li><code>negative_prompt_embeds</code>: The text embeddings for the negative prompt (None if not using guidance)</li> | |
| <li><code>prompt_attention_mask</code>: The attention mask for the positive prompt</li> | |
| <li><code>negative_prompt_attention_mask</code>: The attention mask for the negative prompt (None if not using | |
| guidance)</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]</code></p> | |
| `}),l(2),i(B),i(c);var k=e(c,2);o(k,{title:"MotifVideoImage2VideoPipeline",local:"diffusers.MotifVideoImage2VideoPipeline",headingTag:"h2"});var f=e(k,2),C=n(f);t(C,{name:"class diffusers.MotifVideoImage2VideoPipeline",anchor:"diffusers.MotifVideoImage2VideoPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video_image2video.py#L157",parameters:[{name:"scheduler",val:": SchedulerMixin"},{name:"vae",val:": AutoencoderKLWan"},{name:"text_encoder",val:": T5Gemma2Encoder"},{name:"tokenizer",val:": PreTrainedTokenizerBase"},{name:"transformer",val:": MotifVideoTransformer3DModel"},{name:"guider",val:": BaseGuidance"},{name:"feature_extractor",val:": SiglipImageProcessorPil"}],parametersDescription:[{anchor:"diffusers.MotifVideoImage2VideoPipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/motif_video_transformer_3d#diffusers.MotifVideoTransformer3DModel">MotifVideoTransformer3DModel</a>) — | |
| Conditional Transformer architecture to denoise the encoded video latents.`,name:"transformer"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14192/en/api/schedulers/overview#diffusers.SchedulerMixin">SchedulerMixin</a>) — | |
| A scheduler to be used in combination with <code>transformer</code> to denoise the encoded video latents. Should be an | |
| instance of a class inheriting from <code>SchedulerMixin</code>, such as <a href="/docs/diffusers/pr_14192/en/api/schedulers/multistep_dpm_solver#diffusers.DPMSolverMultistepScheduler">DPMSolverMultistepScheduler</a>. If not | |
| provided, uses the scheduler attached to the pretrained model.`,name:"scheduler"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14192/en/api/models/autoencoder_kl_wan#diffusers.AutoencoderKLWan">AutoencoderKLWan</a>) — | |
| Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.`,name:"vae"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>T5Gemma2Encoder</code>) — | |
| Primary text encoder for encoding text prompts into embeddings.`,name:"text_encoder"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.tokenizer",description:`<strong>tokenizer</strong> (<code>PreTrainedTokenizerBase</code>) — | |
| Tokenizer corresponding to the primary text encoder.`,name:"tokenizer"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.feature_extractor",description:`<strong>feature_extractor</strong> (<code>SiglipImageProcessor</code>) — | |
| Image processor for the SigLIP vision encoder.`,name:"feature_extractor"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.guider",description:`<strong>guider</strong> (<a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.BaseGuidance">BaseGuidance</a>) — | |
| The guidance method to use. Should be an instance of a class inheriting from <code>BaseGuidance</code>, such as | |
| <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.ClassifierFreeGuidance">ClassifierFreeGuidance</a>, <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.AdaptiveProjectedGuidance">AdaptiveProjectedGuidance</a>, or <a href="/docs/diffusers/pr_14192/en/api/modular_diffusers/guiders#diffusers.SkipLayerGuidance">SkipLayerGuidance</a>. If not provided, | |
| defaults to <code>ClassifierFreeGuidance</code>.`,name:"guider"}]});var g=e(C,6),P=n(g);t(P,{name:"__call__",anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video_image2video.py#L620",parameters:[{name:"image",val:": typing.Union[PIL.Image.Image, numpy.ndarray, torch.Tensor, list[PIL.Image.Image], list[numpy.ndarray], list[torch.Tensor]]"},{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"height",val:": int = 736"},{name:"width",val:": int = 1280"},{name:"num_frames",val:": int = 121"},{name:"num_inference_steps",val:": int = 50"},{name:"timesteps",val:": typing.Optional[typing.List[int]] = None"},{name:"num_videos_per_prompt",val:": typing.Optional[int] = 1"},{name:"generator",val:": typing.Union[torch.Generator, typing.List[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": typing.Optional[str] = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"attention_kwargs",val:": typing.Optional[typing.Dict[str, typing.Any]] = None"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int, typing.Dict], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": typing.List[str] = ['latents']"},{name:"max_sequence_length",val:": int = 512"}],parametersDescription:[{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.image",description:`<strong>image</strong> (<code>PipelineImageInput</code>) — | |
| The input image to use as the first frame for video generation.`,name:"image"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>) — | |
| The prompt or prompts to guide the video generation.`,name:"prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| The prompt or prompts not to guide the video generation.`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, defaults to <code>736</code>) — | |
| The height in pixels of the generated video.`,name:"height"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, defaults to <code>1280</code>) — | |
| The width in pixels of the generated video.`,name:"width"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, defaults to <code>121</code>) — | |
| The number of video frames to generate.`,name:"num_frames"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 50) — | |
| The number of denoising steps.`,name:"num_inference_steps"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<code>List[int]</code>, <em>optional</em>) — | |
| Custom timesteps to use for the denoising process.`,name:"timesteps"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| The number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>List[torch.Generator]</code>, <em>optional</em>) — | |
| PyTorch Generator object(s) for deterministic generation.`,name:"generator"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated noisy latents.`,name:"latents"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"pil"</code>) — | |
| The output format of the generated video.`,name:"output_type"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to return a <a href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput">~MotifVideoPipelineOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) — | |
| Arguments passed to the attention processor.`,name:"attention_kwargs"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <em>optional</em>) — | |
| A function or subclass of <code>PipelineCallback</code> called at the end of each denoising step.`,name:"callback_on_step_end"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>List</code>, <em>optional</em>) — | |
| The list of tensor inputs for the <code>callback_on_step_end</code> function.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to <code>512</code>) — | |
| Maximum sequence length for the tokenizer.`,name:"max_sequence_length"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If <code>return_dict</code> is <code>True</code>, <a | |
| href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput" | |
| >~MotifVideoPipelineOutput</a> is returned, otherwise a <code>tuple</code> is returned | |
| where the first element is a list of generated video frames.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/diffusers/pr_14192/en/api/pipelines/motif_video#diffusers.MotifVideoPipelineOutput" | |
| >~MotifVideoPipelineOutput</a> or <code>tuple</code></p> | |
| `});var F=e(P,4);Q(F,{anchor:"diffusers.MotifVideoImage2VideoPipeline.__call__.example",children:(s,d)=>{var r=R(),_=e(h(r),2);a(_,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwUElMJTIwaW1wb3J0JTIwSW1hZ2UlMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW90aWZWaWRlb0ltYWdlMlZpZGVvUGlwZWxpbmUlMEFmcm9tJTIwZGlmZnVzZXJzLnV0aWxzJTIwaW1wb3J0JTIwZXhwb3J0X3RvX3ZpZGVvJTJDJTIwbG9hZF9pbWFnZSUwQSUwQSUyMyUyMExvYWQlMjB0aGUlMjBNb3RpZi1WaWRlbyUyMGltYWdlLXRvLXZpZGVvJTIwcGlwZWxpbmUlMEFtb3RpZl92aWRlb19tb2RlbF9pZCUyMCUzRCUyMCUyMk1vdGlmLVRlY2hub2xvZ2llcyUyRk1vdGlmLVZpZGVvLTJCJTIyJTBBcGlwZSUyMCUzRCUyME1vdGlmVmlkZW9JbWFnZTJWaWRlb1BpcGVsaW5lLmZyb21fcHJldHJhaW5lZChtb3RpZl92aWRlb19tb2RlbF9pZCUyQyUyMHRvcmNoX2R0eXBlJTNEdG9yY2guYmZsb2F0MTYpJTBBcGlwZS50byglMjJjdWRhJTIyKSUwQSUwQSUyMyUyMExvYWQlMjBhbiUyMGltYWdlJTBBaW1hZ2UlMjAlM0QlMjBsb2FkX2ltYWdlKCUwQSUyMCUyMCUyMCUyMCUyMmh0dHBzJTNBJTJGJTJGaHVnZ2luZ2ZhY2UuY28lMkZkYXRhc2V0cyUyRmh1Z2dpbmdmYWNlJTJGZG9jdW1lbnRhdGlvbi1pbWFnZXMlMkZyZXNvbHZlJTJGbWFpbiUyRmRpZmZ1c2VycyUyRmFzdHJvbmF1dC5wbmclMjIlMEEpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQW4lMjBhc3Ryb25hdXQlMjBpcyUyMHdhbGtpbmclMjBvbiUyMHRoZSUyMG1vb24lMjBzdXJmYWNlJTJDJTIwa2lja2luZyUyMHVwJTIwZHVzdCUyMHdpdGglMjBlYWNoJTIwc3RlcCUyMiUwQW5lZ2F0aXZlX3Byb21wdCUyMCUzRCUyMCUyMndvcnN0JTIwcXVhbGl0eSUyQyUyMGluY29uc2lzdGVudCUyMG1vdGlvbiUyQyUyMGJsdXJyeSUyQyUyMGppdHRlcnklMkMlMjBkaXN0b3J0ZWQlMjIlMEElMEF2aWRlbyUyMCUzRCUyMHBpcGUoJTBBJTIwJTIwJTIwJTIwaW1hZ2UlM0RpbWFnZSUyQyUwQSUyMCUyMCUyMCUyMHByb21wdCUzRHByb21wdCUyQyUwQSUyMCUyMCUyMCUyMG5lZ2F0aXZlX3Byb21wdCUzRG5lZ2F0aXZlX3Byb21wdCUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEMTI4MCUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRDczNiUyQyUwQSUyMCUyMCUyMCUyMG51bV9mcmFtZXMlM0QxMjElMkMlMEElMjAlMjAlMjAlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNENTAlMkMlMEEpLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJvdXRwdXQubXA0JTIyJTJDJTIwZnBzJTNEMjQp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MotifVideoImage2VideoPipeline | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video, load_image | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Load the Motif-Video image-to-video pipeline</span> | |
| <span class="hljs-meta">>>> </span>motif_video_model_id = <span class="hljs-string">"Motif-Technologies/Motif-Video-2B"</span> | |
| <span class="hljs-meta">>>> </span>pipe = MotifVideoImage2VideoPipeline.from_pretrained(motif_video_model_id, torch_dtype=torch.bfloat16) | |
| <span class="hljs-meta">>>> </span>pipe.to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Load an image</span> | |
| <span class="hljs-meta">>>> </span>image = load_image( | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.png"</span> | |
| <span class="hljs-meta">... </span>) | |
| <span class="hljs-meta">>>> </span>prompt = <span class="hljs-string">"An astronaut is walking on the moon surface, kicking up dust with each step"</span> | |
| <span class="hljs-meta">>>> </span>negative_prompt = <span class="hljs-string">"worst quality, inconsistent motion, blurry, jittery, distorted"</span> | |
| <span class="hljs-meta">>>> </span>video = pipe( | |
| <span class="hljs-meta">... </span> image=image, | |
| <span class="hljs-meta">... </span> prompt=prompt, | |
| <span class="hljs-meta">... </span> negative_prompt=negative_prompt, | |
| <span class="hljs-meta">... </span> width=<span class="hljs-number">1280</span>, | |
| <span class="hljs-meta">... </span> height=<span class="hljs-number">736</span>, | |
| <span class="hljs-meta">... </span> num_frames=<span class="hljs-number">121</span>, | |
| <span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>, | |
| <span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>] | |
| <span class="hljs-meta">>>> </span>export_to_video(video, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">24</span>)`,lang:"python",wrap:!1}),p(s,r)},$$slots:{default:!0}}),i(g);var W=e(g,2),q=n(W);t(q,{name:"encode_prompt",anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_motif_video_image2video.py#L259",parameters:[{name:"prompt",val:": typing.Union[str, typing.List[str]]"},{name:"negative_prompt",val:": typing.Union[str, typing.List[str], NoneType] = None"},{name:"num_videos_per_prompt",val:": int = 1"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"max_sequence_length",val:": int = 512"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"}],parametersDescription:[{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| The prompt or prompts to be encoded.`,name:"prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| The prompt or prompts not to guide the image generation. If not defined, one has to pass | |
| <code>negative_prompt_embeds</code> instead. Ignored when not using guidance (i.e., ignored if <code>guidance_scale</code> is | |
| less than <code>1</code>).`,name:"negative_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.num_videos_per_prompt",description:`<strong>num_videos_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Number of videos to generate per prompt.`,name:"num_videos_per_prompt"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings.`,name:"prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt | |
| weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input | |
| argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for text embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated attention mask for negative text embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.max_sequence_length",description:`<strong>max_sequence_length</strong> (<code>int</code>, defaults to 512) — | |
| Maximum sequence length for the tokenizer.`,name:"max_sequence_length"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.device",description:`<strong>device</strong> (<code>torch.device</code>, <em>optional</em>) — | |
| Device to place tensors on.`,name:"device"},{anchor:"diffusers.MotifVideoImage2VideoPipeline.encode_prompt.dtype",description:`<strong>dtype</strong> (<code>torch.dtype</code>, <em>optional</em>) — | |
| Data type for tensors.`,name:"dtype"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A tuple containing:</p> | |
| <ul> | |
| <li><code>prompt_embeds</code>: The text embeddings for the positive prompt</li> | |
| <li><code>negative_prompt_embeds</code>: The text embeddings for the negative prompt (None if not using guidance)</li> | |
| <li><code>prompt_attention_mask</code>: The attention mask for the positive prompt</li> | |
| <li><code>negative_prompt_attention_mask</code>: The attention mask for the negative prompt (None if not using | |
| guidance)</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]</code></p> | |
| `}),l(2),i(W),i(f);var N=e(f,2);o(N,{title:"MotifVideoPipelineOutput",local:"diffusers.MotifVideoPipelineOutput",headingTag:"h2"});var u=e(N,2),A=n(u);t(A,{name:"class diffusers.MotifVideoPipelineOutput",anchor:"diffusers.MotifVideoPipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14192/src/diffusers/pipelines/motif_video/pipeline_output.py#L9",parameters:[{name:"frames",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.MotifVideoPipelineOutput.frames",description:`<strong>frames</strong> (<code>torch.Tensor</code>, <code>np.ndarray</code>, or List[List[PIL.Image.Image]]) — | |
| List of video outputs - It can be a nested list of length <code>batch_size,</code> with each sub-list containing | |
| denoised PIL image sequences of length <code>num_frames.</code> It can also be a NumPy array or Torch tensor of shape | |
| <code>(batch_size, num_frames, channels, height, width)</code>.`,name:"frames"}]}),l(2),i(u);var H=e(u,2);O(H,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/motif_video.md"}),l(2),p(X,y),oe()}export{pe as component}; | |
Xet Storage Details
- Size:
- 54.3 kB
- Xet hash:
- 01d91a3222f4f1b74810956d299ad0e282f2a03035115fa122f5adcdf0cccd88
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.