Buckets:
| import"../chunks/DsnmJJEf.js";import{i as k,h as x,H as n,a as v,D as a,E as W,s as G}from"../chunks/CmJXCtRL.js";import{p as z,o as V,s as e,f as T,a as d,b as P,c as s,d as c,r as o,n as m}from"../chunks/DK803DsY.js";import{E as N}from"../chunks/Bu2vAape.js";const X='{"title":"EasyAnimate","local":"easyanimate","sections":[{"title":"Quantization","local":"quantization","sections":[],"depth":2},{"title":"EasyAnimatePipeline","local":"diffusers.EasyAnimatePipeline","sections":[],"depth":2},{"title":"EasyAnimatePipelineOutput","local":"diffusers.pipelines.easyanimate.pipeline_output.EasyAnimatePipelineOutput","sections":[],"depth":2}],"depth":1}';var C=c('<meta name="hf:doc:metadata"/>'),Y=c("<p>Examples:</p> <!>",1),Q=c(`<p></p> <!> <p><a href="https://github.com/aigc-apps/EasyAnimate" rel="nofollow">EasyAnimate</a> by Alibaba PAI.</p> <p>The description from it’s GitHub page: <em>EasyAnimate is a pipeline based on the transformer architecture, designed for generating AI images and videos, and for training baseline models and Lora models for Diffusion Transformer. We support direct prediction from pre-trained EasyAnimate models, allowing for the generation of videos with various resolutions, approximately 6 seconds in length, at 8fps (EasyAnimateV5.1, 1 to 49 frames). Additionally, users can train their own baseline and Lora models for specific style transformations.</em></p> <p>This pipeline was contributed by <a href="https://github.com/bubbliiiing" rel="nofollow">bubbliiiing</a>. The original codebase can be found <a href="https://huggingface.co/alibaba-pai" rel="nofollow">here</a>. The original weights can be found under <a href="https://huggingface.co/alibaba-pai" rel="nofollow">hf.co/alibaba-pai</a>.</p> <p>There are two official EasyAnimate checkpoints for text-to-video and video-to-video.</p> <table><thead><tr><th align="center">checkpoints</th><th align="center">recommended inference dtype</th></tr></thead><tbody><tr><td align="center"><a href="https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh" rel="nofollow"><code>alibaba-pai/EasyAnimateV5.1-12b-zh</code></a></td><td align="center">torch.float16</td></tr><tr><td align="center"><a href="https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP" rel="nofollow"><code>alibaba-pai/EasyAnimateV5.1-12b-zh-InP</code></a></td><td align="center">torch.float16</td></tr></tbody></table> <p>There is one official EasyAnimate checkpoints available for image-to-video and video-to-video.</p> <table><thead><tr><th align="center">checkpoints</th><th align="center">recommended inference dtype</th></tr></thead><tbody><tr><td align="center"><a href="https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP" rel="nofollow"><code>alibaba-pai/EasyAnimateV5.1-12b-zh-InP</code></a></td><td align="center">torch.float16</td></tr></tbody></table> <p>There are two official EasyAnimate checkpoints available for control-to-video.</p> <table><thead><tr><th align="center">checkpoints</th><th align="center">recommended inference dtype</th></tr></thead><tbody><tr><td align="center"><a href="https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control" rel="nofollow"><code>alibaba-pai/EasyAnimateV5.1-12b-zh-Control</code></a></td><td align="center">torch.float16</td></tr><tr><td align="center"><a href="https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control-Camera" rel="nofollow"><code>alibaba-pai/EasyAnimateV5.1-12b-zh-Control-Camera</code></a></td><td align="center">torch.float16</td></tr></tbody></table> <p>For the EasyAnimateV5.1 series:</p> <ul><li>Text-to-video (T2V) and Image-to-video (I2V) works for multiple resolutions. The width and height can vary from 256 to 1024.</li> <li>Both T2V and I2V models support generation with 1~49 frames and work best at this value. Exporting videos at 8 FPS is recommended.</li></ul> <!> <p>Quantization helps reduce the memory requirements of very large models by storing model weights in a lower precision data type. However, quantization may have varying impact on video quality depending on the video model.</p> <p>Refer to the <a href="../../quantization/overview">Quantization</a> overview to learn more about supported quantization backends and selecting a quantization backend that supports your use case. The example below demonstrates how to load a quantized <a href="/docs/diffusers/pr_14204/en/api/pipelines/easyanimate#diffusers.EasyAnimatePipeline">EasyAnimatePipeline</a> for inference with bitsandbytes.</p> <!> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Pipeline for text-to-video generation using EasyAnimate.</p> <p>This model inherits from <a href="/docs/diffusers/pr_14204/en/api/pipelines/overview#diffusers.DiffusionPipeline">DiffusionPipeline</a>. Check the superclass documentation for the generic methods the | |
| library implements for all the pipelines (such as downloading or saving, running on a particular device, etc.)</p> <p>EasyAnimate uses one text encoder <a href="https://huggingface.co/Qwen/Qwen2-VL-7B-Instruct" rel="nofollow">qwen2 vl</a> in V5.1.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Generates images or video using the EasyAnimate pipeline based on the provided prompts.</p> <!></div> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Encodes the prompt into text encoder hidden states.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Output class for EasyAnimate pipelines.</p></div> <!> <p></p>`,1);function S(J,E){z(E,!1),V(()=>{new URLSearchParams(window.location.search).get("fw")}),k();var h=Q();x("1ol2z8d",t=>{var p=C();G(p,"content",X),d(t,p)});var u=e(T(h),2);n(u,{title:"EasyAnimate",local:"easyanimate",headingTag:"h1"});var g=e(u,24);n(g,{title:"Quantization",local:"quantization",headingTag:"h2"});var f=e(g,6);v(f,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwQml0c0FuZEJ5dGVzQ29uZmlnJTIwYXMlMjBEaWZmdXNlcnNCaXRzQW5kQnl0ZXNDb25maWclMkMlMjBFYXN5QW5pbWF0ZVRyYW5zZm9ybWVyM0RNb2RlbCUyQyUyMEVhc3lBbmltYXRlUGlwZWxpbmUlMEFmcm9tJTIwZGlmZnVzZXJzLnV0aWxzJTIwaW1wb3J0JTIwZXhwb3J0X3RvX3ZpZGVvJTBBJTBBcXVhbnRfY29uZmlnJTIwJTNEJTIwRGlmZnVzZXJzQml0c0FuZEJ5dGVzQ29uZmlnKGxvYWRfaW5fOGJpdCUzRFRydWUpJTBBdHJhbnNmb3JtZXJfOGJpdCUyMCUzRCUyMEVhc3lBbmltYXRlVHJhbnNmb3JtZXIzRE1vZGVsLmZyb21fcHJldHJhaW5lZCglMEElMjAlMjAlMjAlMjAlMjJhbGliYWJhLXBhaSUyRkVhc3lBbmltYXRlVjUuMS0xMmItemglMjIlMkMlMEElMjAlMjAlMjAlMjBzdWJmb2xkZXIlM0QlMjJ0cmFuc2Zvcm1lciUyMiUyQyUwQSUyMCUyMCUyMCUyMHF1YW50aXphdGlvbl9jb25maWclM0RxdWFudF9jb25maWclMkMlMEElMjAlMjAlMjAlMjB0b3JjaF9kdHlwZSUzRHRvcmNoLmZsb2F0MTYlMkMlMEEpJTBBJTBBcGlwZWxpbmUlMjAlM0QlMjBFYXN5QW5pbWF0ZVBpcGVsaW5lLmZyb21fcHJldHJhaW5lZCglMEElMjAlMjAlMjAlMjAlMjJhbGliYWJhLXBhaSUyRkVhc3lBbmltYXRlVjUuMS0xMmItemglMjIlMkMlMEElMjAlMjAlMjAlMjB0cmFuc2Zvcm1lciUzRHRyYW5zZm9ybWVyXzhiaXQlMkMlMEElMjAlMjAlMjAlMjB0b3JjaF9kdHlwZSUzRHRvcmNoLmZsb2F0MTYlMkMlMEElMjAlMjAlMjAlMjBkZXZpY2VfbWFwJTNEJTIyYmFsYW5jZWQlMjIlMkMlMEEpJTBBJTBBcHJvbXB0JTIwJTNEJTIwJTIyQSUyMGNhdCUyMHdhbGtzJTIwb24lMjB0aGUlMjBncmFzcyUyQyUyMHJlYWxpc3RpYyUyMHN0eWxlLiUyMiUwQW5lZ2F0aXZlX3Byb21wdCUyMCUzRCUyMCUyMmJhZCUyMGRldGFpbGVkJTIyJTBBdmlkZW8lMjAlM0QlMjBwaXBlbGluZShwcm9tcHQlM0Rwcm9tcHQlMkMlMjBuZWdhdGl2ZV9wcm9tcHQlM0RuZWdhdGl2ZV9wcm9tcHQlMkMlMjBudW1fZnJhbWVzJTNENDklMkMlMjBudW1faW5mZXJlbmNlX3N0ZXBzJTNEMzApLmZyYW1lcyU1QjAlNUQlMEFleHBvcnRfdG9fdmlkZW8odmlkZW8lMkMlMjAlMjJjYXQubXA0JTIyJTJDJTIwZnBzJTNEOCk=",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> BitsAndBytesConfig <span class="hljs-keyword">as</span> DiffusersBitsAndBytesConfig, EasyAnimateTransformer3DModel, EasyAnimatePipeline | |
| <span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| quant_config = DiffusersBitsAndBytesConfig(load_in_8bit=<span class="hljs-literal">True</span>) | |
| transformer_8bit = EasyAnimateTransformer3DModel.from_pretrained( | |
| <span class="hljs-string">"alibaba-pai/EasyAnimateV5.1-12b-zh"</span>, | |
| subfolder=<span class="hljs-string">"transformer"</span>, | |
| quantization_config=quant_config, | |
| torch_dtype=torch.float16, | |
| ) | |
| pipeline = EasyAnimatePipeline.from_pretrained( | |
| <span class="hljs-string">"alibaba-pai/EasyAnimateV5.1-12b-zh"</span>, | |
| transformer=transformer_8bit, | |
| torch_dtype=torch.float16, | |
| device_map=<span class="hljs-string">"balanced"</span>, | |
| ) | |
| prompt = <span class="hljs-string">"A cat walks on the grass, realistic style."</span> | |
| negative_prompt = <span class="hljs-string">"bad detailed"</span> | |
| video = pipeline(prompt=prompt, negative_prompt=negative_prompt, num_frames=<span class="hljs-number">49</span>, num_inference_steps=<span class="hljs-number">30</span>).frames[<span class="hljs-number">0</span>] | |
| export_to_video(video, <span class="hljs-string">"cat.mp4"</span>, fps=<span class="hljs-number">8</span>)`,lang:"py",wrap:!1});var _=e(f,2);n(_,{title:"EasyAnimatePipeline",local:"diffusers.EasyAnimatePipeline",headingTag:"h2"});var i=e(_,2),y=s(i);a(y,{name:"class diffusers.EasyAnimatePipeline",anchor:"diffusers.EasyAnimatePipeline",source:"https://github.com/huggingface/diffusers/blob/vr_14204/src/diffusers/pipelines/easyanimate/pipeline_easyanimate.py#L186",parameters:[{name:"vae",val:": AutoencoderKLMagvit"},{name:"text_encoder",val:": transformers.models.qwen2_vl.modeling_qwen2_vl.Qwen2VLForConditionalGeneration | transformers.models.bert.modeling_bert.BertModel"},{name:"tokenizer",val:": transformers.models.qwen2.tokenization_qwen2.Qwen2Tokenizer | transformers.models.bert.tokenization_bert.BertTokenizer"},{name:"transformer",val:": EasyAnimateTransformer3DModel"},{name:"scheduler",val:": FlowMatchEulerDiscreteScheduler"}],parametersDescription:[{anchor:"diffusers.EasyAnimatePipeline.vae",description:`<strong>vae</strong> (<a href="/docs/diffusers/pr_14204/en/api/models/autoencoderkl_magvit#diffusers.AutoencoderKLMagvit">AutoencoderKLMagvit</a>) — | |
| Variational Auto-Encoder (VAE) Model to encode and decode video to and from latent representations.`,name:"vae"},{anchor:"diffusers.EasyAnimatePipeline.text_encoder",description:`<strong>text_encoder</strong> (<code>~transformers.Qwen2VLForConditionalGeneration</code>, <code>~transformers.BertModel</code> | None) — | |
| EasyAnimate uses <a href="https://huggingface.co/Qwen/Qwen2-VL-7B-Instruct" rel="nofollow">qwen2 vl</a> in V5.1.`,name:"text_encoder"},{anchor:"diffusers.EasyAnimatePipeline.tokenizer",description:`<strong>tokenizer</strong> (<code>~transformers.Qwen2Tokenizer</code>, <code>~transformers.BertTokenizer</code> | None) — | |
| A <code>Qwen2Tokenizer</code> or <code>BertTokenizer</code> to tokenize text.`,name:"tokenizer"},{anchor:"diffusers.EasyAnimatePipeline.transformer",description:`<strong>transformer</strong> (<a href="/docs/diffusers/pr_14204/en/api/models/easyanimate_transformer3d#diffusers.EasyAnimateTransformer3DModel">EasyAnimateTransformer3DModel</a>) — | |
| The EasyAnimate model designed by EasyAnimate Team.`,name:"transformer"},{anchor:"diffusers.EasyAnimatePipeline.scheduler",description:`<strong>scheduler</strong> (<a href="/docs/diffusers/pr_14204/en/api/schedulers/flow_match_euler_discrete#diffusers.FlowMatchEulerDiscreteScheduler">FlowMatchEulerDiscreteScheduler</a>) — | |
| A scheduler to be used in combination with EasyAnimate to denoise the encoded image latents.`,name:"scheduler"}]});var r=e(y,8),b=s(r);a(b,{name:"__call__",anchor:"diffusers.EasyAnimatePipeline.__call__",source:"https://github.com/huggingface/diffusers/blob/vr_14204/src/diffusers/pipelines/easyanimate/pipeline_easyanimate.py#L524",parameters:[{name:"prompt",val:": str | list[str] = None"},{name:"num_frames",val:": int | None = 49"},{name:"height",val:": int | None = 512"},{name:"width",val:": int | None = 512"},{name:"num_inference_steps",val:": int | None = 50"},{name:"guidance_scale",val:": float | None = 5.0"},{name:"negative_prompt",val:": str | list[str] | None = None"},{name:"num_images_per_prompt",val:": int | None = 1"},{name:"eta",val:": float | None = 0.0"},{name:"generator",val:": typing.Union[torch.Generator, list[torch.Generator], NoneType] = None"},{name:"latents",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"timesteps",val:": list[int] | None = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"output_type",val:": str | None = 'pil'"},{name:"return_dict",val:": bool = True"},{name:"callback_on_step_end",val:": typing.Union[typing.Callable[[int, int], NoneType], diffusers.callbacks.PipelineCallback, diffusers.callbacks.MultiPipelineCallbacks, NoneType] = None"},{name:"callback_on_step_end_tensor_inputs",val:": list = ['latents']"},{name:"guidance_rescale",val:": float = 0.0"}],parametersDescription:[{anchor:"diffusers.EasyAnimatePipeline.__call__.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| Text prompts to guide the image or video generation. If not provided, use <code>prompt_embeds</code> instead.`,name:"prompt"},{anchor:"diffusers.EasyAnimatePipeline.__call__.num_frames",description:`<strong>num_frames</strong> (<code>int</code>, <em>optional</em>) — | |
| Length of the generated video (in frames).`,name:"num_frames"},{anchor:"diffusers.EasyAnimatePipeline.__call__.height",description:`<strong>height</strong> (<code>int</code>, <em>optional</em>) — | |
| Height of the generated image in pixels.`,name:"height"},{anchor:"diffusers.EasyAnimatePipeline.__call__.width",description:`<strong>width</strong> (<code>int</code>, <em>optional</em>) — | |
| Width of the generated image in pixels.`,name:"width"},{anchor:"diffusers.EasyAnimatePipeline.__call__.num_inference_steps",description:`<strong>num_inference_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 50) — | |
| Number of denoising steps during generation. More steps generally yield higher quality images but slow | |
| down inference.`,name:"num_inference_steps"},{anchor:"diffusers.EasyAnimatePipeline.__call__.guidance_scale",description:`<strong>guidance_scale</strong> (<code>float</code>, <em>optional</em>, defaults to 5.0) — | |
| Encourages the model to align outputs with prompts. A higher value may decrease image quality.`,name:"guidance_scale"},{anchor:"diffusers.EasyAnimatePipeline.__call__.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| Prompts indicating what to exclude in generation. If not specified, use <code>negative_prompt_embeds</code>.`,name:"negative_prompt"},{anchor:"diffusers.EasyAnimatePipeline.__call__.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Number of images to generate for each prompt.`,name:"num_images_per_prompt"},{anchor:"diffusers.EasyAnimatePipeline.__call__.eta",description:`<strong>eta</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| Applies to DDIM scheduling. Controlled by the eta parameter from the related literature.`,name:"eta"},{anchor:"diffusers.EasyAnimatePipeline.__call__.generator",description:`<strong>generator</strong> (<code>torch.Generator</code> or <code>list[torch.Generator]</code>, <em>optional</em>) — | |
| A generator to ensure reproducibility in image generation.`,name:"generator"},{anchor:"diffusers.EasyAnimatePipeline.__call__.latents",description:`<strong>latents</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Predefined latent tensors to condition generation.`,name:"latents"},{anchor:"diffusers.EasyAnimatePipeline.__call__.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Text embeddings for the prompts. Overrides prompt string inputs for more flexibility.`,name:"prompt_embeds"},{anchor:"diffusers.EasyAnimatePipeline.__call__.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Embeddings for negative prompts. Overrides string inputs if defined.`,name:"negative_prompt_embeds"},{anchor:"diffusers.EasyAnimatePipeline.__call__.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Attention mask for the primary prompt embeddings.`,name:"prompt_attention_mask"},{anchor:"diffusers.EasyAnimatePipeline.__call__.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Attention mask for negative prompt embeddings.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.EasyAnimatePipeline.__call__.output_type",description:`<strong>output_type</strong> (<code>str</code>, <em>optional</em>, defaults to “latent”) — | |
| Format of the generated output, either as a PIL image or as a NumPy array.`,name:"output_type"},{anchor:"diffusers.EasyAnimatePipeline.__call__.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| If <code>True</code>, returns a structured output. Otherwise returns a simple tuple.`,name:"return_dict"},{anchor:"diffusers.EasyAnimatePipeline.__call__.callback_on_step_end",description:`<strong>callback_on_step_end</strong> (<code>Callable</code>, <em>optional</em>) — | |
| Functions called at the end of each denoising step.`,name:"callback_on_step_end"},{anchor:"diffusers.EasyAnimatePipeline.__call__.callback_on_step_end_tensor_inputs",description:`<strong>callback_on_step_end_tensor_inputs</strong> (<code>list[str]</code>, <em>optional</em>) — | |
| Tensor names to be included in callback function calls.`,name:"callback_on_step_end_tensor_inputs"},{anchor:"diffusers.EasyAnimatePipeline.__call__.guidance_rescale",description:`<strong>guidance_rescale</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| Adjusts noise levels based on guidance scale.`,name:"guidance_rescale"},{anchor:"diffusers.EasyAnimatePipeline.__call__.timesteps",description:`<strong>timesteps</strong> (<code>list[int]</code>, <em>optional</em>) — | |
| Custom timesteps to use for the denoising process. If not defined, the scheduler’s default schedule for | |
| <code>num_inference_steps</code> is used.`,name:"timesteps"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>If <code>return_dict</code> is <code>True</code>, <a | |
| href="/docs/diffusers/pr_14204/en/api/pipelines/stable_diffusion/depth2img#diffusers.pipelines.stable_diffusion.StableDiffusionPipelineOutput" | |
| >StableDiffusionPipelineOutput</a> is returned, | |
| otherwise a <code>tuple</code> is returned where the first element is a list with the generated images and the | |
| second element is a list of <code>bool</code>s indicating whether the corresponding generated image contains | |
| “not-safe-for-work” (nsfw) content.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/diffusers/pr_14204/en/api/pipelines/stable_diffusion/depth2img#diffusers.pipelines.stable_diffusion.StableDiffusionPipelineOutput" | |
| >StableDiffusionPipelineOutput</a> or <code>tuple</code></p> | |
| `});var A=e(b,4);N(A,{anchor:"diffusers.EasyAnimatePipeline.__call__.example",children:(t,p)=>{var w=Y(),U=e(T(w),2);v(U,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwRWFzeUFuaW1hdGVQaXBlbGluZSUwQWZyb20lMjBkaWZmdXNlcnMudXRpbHMlMjBpbXBvcnQlMjBleHBvcnRfdG9fdmlkZW8lMEElMEElMjMlMjBNb2RlbHMlM0ElMjAlMjJhbGliYWJhLXBhaSUyRkVhc3lBbmltYXRlVjUuMS0xMmItemglMjIlMEFwaXBlJTIwJTNEJTIwRWFzeUFuaW1hdGVQaXBlbGluZS5mcm9tX3ByZXRyYWluZWQoJTBBJTIwJTIwJTIwJTIwJTIyYWxpYmFiYS1wYWklMkZFYXN5QW5pbWF0ZVY1LjEtN2ItemgtZGlmZnVzZXJzJTIyJTJDJTIwdG9yY2hfZHR5cGUlM0R0b3JjaC5mbG9hdDE2JTBBKS50byglMjJjdWRhJTIyKSUwQXByb21wdCUyMCUzRCUyMCglMEElMjAlMjAlMjAlMjAlMjJBJTIwcGFuZGElMkMlMjBkcmVzc2VkJTIwaW4lMjBhJTIwc21hbGwlMkMlMjByZWQlMjBqYWNrZXQlMjBhbmQlMjBhJTIwdGlueSUyMGhhdCUyQyUyMHNpdHMlMjBvbiUyMGElMjB3b29kZW4lMjBzdG9vbCUyMGluJTIwYSUyMHNlcmVuZSUyMGJhbWJvbyUyMGZvcmVzdC4lMjAlMjIlMEElMjAlMjAlMjAlMjAlMjJUaGUlMjBwYW5kYSdzJTIwZmx1ZmZ5JTIwcGF3cyUyMHN0cnVtJTIwYSUyMG1pbmlhdHVyZSUyMGFjb3VzdGljJTIwZ3VpdGFyJTJDJTIwcHJvZHVjaW5nJTIwc29mdCUyQyUyMG1lbG9kaWMlMjB0dW5lcy4lMjBOZWFyYnklMkMlMjBhJTIwZmV3JTIwb3RoZXIlMjAlMjIlMEElMjAlMjAlMjAlMjAlMjJwYW5kYXMlMjBnYXRoZXIlMkMlMjB3YXRjaGluZyUyMGN1cmlvdXNseSUyMGFuZCUyMHNvbWUlMjBjbGFwcGluZyUyMGluJTIwcmh5dGhtLiUyMFN1bmxpZ2h0JTIwZmlsdGVycyUyMHRocm91Z2glMjB0aGUlMjB0YWxsJTIwYmFtYm9vJTJDJTIwJTIyJTBBJTIwJTIwJTIwJTIwJTIyY2FzdGluZyUyMGElMjBnZW50bGUlMjBnbG93JTIwb24lMjB0aGUlMjBzY2VuZS4lMjBUaGUlMjBwYW5kYSdzJTIwZmFjZSUyMGlzJTIwZXhwcmVzc2l2ZSUyQyUyMHNob3dpbmclMjBjb25jZW50cmF0aW9uJTIwYW5kJTIwam95JTIwYXMlMjBpdCUyMHBsYXlzLiUyMCUyMiUwQSUyMCUyMCUyMCUyMCUyMlRoZSUyMGJhY2tncm91bmQlMjBpbmNsdWRlcyUyMGElMjBzbWFsbCUyQyUyMGZsb3dpbmclMjBzdHJlYW0lMjBhbmQlMjB2aWJyYW50JTIwZ3JlZW4lMjBmb2xpYWdlJTJDJTIwZW5oYW5jaW5nJTIwdGhlJTIwcGVhY2VmdWwlMjBhbmQlMjBtYWdpY2FsJTIwJTIyJTBBJTIwJTIwJTIwJTIwJTIyYXRtb3NwaGVyZSUyMG9mJTIwdGhpcyUyMHVuaXF1ZSUyMG11c2ljYWwlMjBwZXJmb3JtYW5jZS4lMjIlMEEpJTBBc2FtcGxlX3NpemUlMjAlM0QlMjAoNTEyJTJDJTIwNTEyKSUwQXZpZGVvJTIwJTNEJTIwcGlwZSglMEElMjAlMjAlMjAlMjBwcm9tcHQlM0Rwcm9tcHQlMkMlMEElMjAlMjAlMjAlMjBndWlkYW5jZV9zY2FsZSUzRDYlMkMlMEElMjAlMjAlMjAlMjBuZWdhdGl2ZV9wcm9tcHQlM0QlMjJiYWQlMjBkZXRhaWxlZCUyMiUyQyUwQSUyMCUyMCUyMCUyMGhlaWdodCUzRHNhbXBsZV9zaXplJTVCMCU1RCUyQyUwQSUyMCUyMCUyMCUyMHdpZHRoJTNEc2FtcGxlX3NpemUlNUIxJTVEJTJDJTBBJTIwJTIwJTIwJTIwbnVtX2luZmVyZW5jZV9zdGVwcyUzRDUwJTJDJTBBKS5mcmFtZXMlNUIwJTVEJTBBZXhwb3J0X3RvX3ZpZGVvKHZpZGVvJTJDJTIwJTIyb3V0cHV0Lm1wNCUyMiUyQyUyMGZwcyUzRDgp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> EasyAnimatePipeline | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> export_to_video | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Models: "alibaba-pai/EasyAnimateV5.1-12b-zh"</span> | |
| <span class="hljs-meta">>>> </span>pipe = EasyAnimatePipeline.from_pretrained( | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"alibaba-pai/EasyAnimateV5.1-7b-zh-diffusers"</span>, torch_dtype=torch.float16 | |
| <span class="hljs-meta">... </span>).to(<span class="hljs-string">"cuda"</span>) | |
| <span class="hljs-meta">>>> </span>prompt = ( | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"A panda, dressed in a small, red jacket and a tiny hat, sits on a wooden stool in a serene bamboo forest. "</span> | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"The panda's fluffy paws strum a miniature acoustic guitar, producing soft, melodic tunes. Nearby, a few other "</span> | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"pandas gather, watching curiously and some clapping in rhythm. Sunlight filters through the tall bamboo, "</span> | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"casting a gentle glow on the scene. The panda's face is expressive, showing concentration and joy as it plays. "</span> | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"The background includes a small, flowing stream and vibrant green foliage, enhancing the peaceful and magical "</span> | |
| <span class="hljs-meta">... </span> <span class="hljs-string">"atmosphere of this unique musical performance."</span> | |
| <span class="hljs-meta">... </span>) | |
| <span class="hljs-meta">>>> </span>sample_size = (<span class="hljs-number">512</span>, <span class="hljs-number">512</span>) | |
| <span class="hljs-meta">>>> </span>video = pipe( | |
| <span class="hljs-meta">... </span> prompt=prompt, | |
| <span class="hljs-meta">... </span> guidance_scale=<span class="hljs-number">6</span>, | |
| <span class="hljs-meta">... </span> negative_prompt=<span class="hljs-string">"bad detailed"</span>, | |
| <span class="hljs-meta">... </span> height=sample_size[<span class="hljs-number">0</span>], | |
| <span class="hljs-meta">... </span> width=sample_size[<span class="hljs-number">1</span>], | |
| <span class="hljs-meta">... </span> num_inference_steps=<span class="hljs-number">50</span>, | |
| <span class="hljs-meta">... </span>).frames[<span class="hljs-number">0</span>] | |
| <span class="hljs-meta">>>> </span>export_to_video(video, <span class="hljs-string">"output.mp4"</span>, fps=<span class="hljs-number">8</span>)`,lang:"python",wrap:!1}),d(t,w)},$$slots:{default:!0}}),o(r);var M=e(r,2),Z=s(M);a(Z,{name:"encode_prompt",anchor:"diffusers.EasyAnimatePipeline.encode_prompt",source:"https://github.com/huggingface/diffusers/blob/vr_14204/src/diffusers/pipelines/easyanimate/pipeline_easyanimate.py#L241",parameters:[{name:"prompt",val:": str | list[str]"},{name:"num_images_per_prompt",val:": int = 1"},{name:"do_classifier_free_guidance",val:": bool = True"},{name:"negative_prompt",val:": str | list[str] | None = None"},{name:"prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_embeds",val:": typing.Optional[torch.Tensor] = None"},{name:"prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"negative_prompt_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"device",val:": typing.Optional[torch.device] = None"},{name:"dtype",val:": typing.Optional[torch.dtype] = None"},{name:"max_sequence_length",val:": int = 256"}],parametersDescription:[{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.prompt",description:`<strong>prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| prompt to be encoded`,name:"prompt"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.device",description:`<strong>device</strong> — (<code>torch.device</code>): | |
| torch device`,name:"device"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.dtype",description:`<strong>dtype</strong> (<code>torch.dtype</code>) — | |
| torch dtype`,name:"dtype"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.num_images_per_prompt",description:`<strong>num_images_per_prompt</strong> (<code>int</code>) — | |
| number of images that should be generated per prompt`,name:"num_images_per_prompt"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.do_classifier_free_guidance",description:`<strong>do_classifier_free_guidance</strong> (<code>bool</code>) — | |
| whether to use classifier free guidance or not`,name:"do_classifier_free_guidance"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.negative_prompt",description:`<strong>negative_prompt</strong> (<code>str</code> or <code>list[str]</code>, <em>optional</em>) — | |
| The prompt or prompts not to guide the image generation. If not defined, one has to pass | |
| <code>negative_prompt_embeds</code> instead. Ignored when not using guidance (i.e., ignored if <code>guidance_scale</code> is | |
| less than <code>1</code>).`,name:"negative_prompt"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.prompt_embeds",description:`<strong>prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt weighting. If not | |
| provided, text embeddings will be generated from <code>prompt</code> input argument.`,name:"prompt_embeds"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.negative_prompt_embeds",description:`<strong>negative_prompt_embeds</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Pre-generated negative text embeddings. Can be used to easily tweak text inputs, <em>e.g.</em> prompt | |
| weighting. If not provided, negative_prompt_embeds will be generated from <code>negative_prompt</code> input | |
| argument.`,name:"negative_prompt_embeds"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.prompt_attention_mask",description:`<strong>prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Attention mask for the prompt. Required when <code>prompt_embeds</code> is passed directly.`,name:"prompt_attention_mask"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.negative_prompt_attention_mask",description:`<strong>negative_prompt_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Attention mask for the negative prompt. Required when <code>negative_prompt_embeds</code> is passed directly.`,name:"negative_prompt_attention_mask"},{anchor:"diffusers.EasyAnimatePipeline.encode_prompt.max_sequence_length",description:"<strong>max_sequence_length</strong> (<code>int</code>, <em>optional</em>) — maximum sequence length to use for the prompt.",name:"max_sequence_length"}]}),m(2),o(M),o(i);var j=e(i,2);n(j,{title:"EasyAnimatePipelineOutput",local:"diffusers.pipelines.easyanimate.pipeline_output.EasyAnimatePipelineOutput",headingTag:"h2"});var l=e(j,2),B=s(l);a(B,{name:"class diffusers.pipelines.easyanimate.pipeline_output.EasyAnimatePipelineOutput",anchor:"diffusers.pipelines.easyanimate.pipeline_output.EasyAnimatePipelineOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14204/src/diffusers/pipelines/easyanimate/pipeline_output.py#L9",parameters:[{name:"frames",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.pipelines.easyanimate.pipeline_output.EasyAnimatePipelineOutput.frames",description:`<strong>frames</strong> (<code>torch.Tensor</code>, <code>np.ndarray</code>, or list[list[PIL.Image.Image]]) — | |
| list of video outputs - It can be a nested list of length <code>batch_size,</code> with each sub-list containing | |
| denoised PIL image sequences of length <code>num_frames.</code> It can also be a NumPy array or Torch tensor of shape | |
| <code>(batch_size, num_frames, channels, height, width)</code>.`,name:"frames"}]}),m(2),o(l);var I=e(l,2);W(I,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/easyanimate.md"}),m(2),d(J,h),P()}export{S as component}; | |
Xet Storage Details
- Size:
- 30.7 kB
- Xet hash:
- 28ed40d4c595df3296f5e20a0111bcdb76baa79040f27ff62209b983cb21634a
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.