Buckets:
| import"../chunks/DsnmJJEf.js";import{i as C,h as B,H as o,a as J,D as t,E as N,s as q}from"../chunks/BtE7mKSK.js";import{p as P,o as U,s as e,f as Z,a as T,b as j,c as n,d as v,n as d,r as a}from"../chunks/jDjavuwI.js";const G='{"title":"Anima","local":"anima","sections":[{"title":"AnimaModularPipeline","local":"diffusers.AnimaModularPipeline","sections":[],"depth":2},{"title":"AnimaAutoBlocks","local":"diffusers.AnimaAutoBlocks","sections":[],"depth":2},{"title":"AnimaTextConditioner","local":"diffusers.AnimaTextConditioner","sections":[],"depth":2}],"depth":1}';var I=v('<meta name="hf:doc:metadata"/>'),Q=v(`<p></p> <!> <p>Anima is a text-to-image model that reuses the <a href="/docs/diffusers/pr_13881/en/api/models/cosmos_transformer3d#diffusers.CosmosTransformer3DModel">CosmosTransformer3DModel</a> with a Qwen3 text encoder, a T5-token text conditioner, and the <a href="/docs/diffusers/pr_13881/en/api/models/autoencoderkl_qwenimage#diffusers.AutoencoderKLQwenImage">AutoencoderKLQwenImage</a> VAE.</p> <!> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>A ModularPipeline for Anima.</p> <blockquote class="warning"><p>> This is an experimental feature and is likely to change in the future.</p></blockquote></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Auto Modular pipeline for text-to-image generation using Anima.</p> <p>Supported workflows:</p> <ul><li><code>text2image</code>: requires <code>prompt</code></li></ul> <p>Components: | |
| text_encoder (<code>Qwen3Model</code>) tokenizer (<code>Qwen2Tokenizer</code>) t5_tokenizer (<code>T5TokenizerFast</code>) text_conditioner | |
| (<code>AnimaTextConditioner</code>) guider (<code>ClassifierFreeGuidance</code>) transformer (<code>CosmosTransformer3DModel</code>) scheduler | |
| (<code>FlowMatchEulerDiscreteScheduler</code>) vae (<code>AutoencoderKLQwenImage</code>) image_processor (<code>VaeImageProcessor</code>)</p> <p>Inputs: | |
| prompt (<code>str</code>): | |
| The prompt or prompts to guide image generation. | |
| negative_prompt (<code>str</code>, <em>optional</em>): | |
| The prompt or prompts not to guide the image generation. | |
| max_sequence_length (<code>int</code>, <em>optional</em>, defaults to 512): | |
| Maximum sequence length for prompt encoding. | |
| num_images_per_prompt (<code>int</code>, <em>optional</em>, defaults to 1): | |
| The number of images to generate per prompt. | |
| height (<code>int</code>, <em>optional</em>): | |
| The height in pixels of the generated image. | |
| width (<code>int</code>, <em>optional</em>): | |
| The width in pixels of the generated image. | |
| latents (<code>Tensor</code>, <em>optional</em>): | |
| Pre-generated noisy latents for image generation. | |
| generator (<code>Generator</code>, <em>optional</em>): | |
| Torch generator for deterministic generation. | |
| num_inference_steps (<code>int</code>, <em>optional</em>, defaults to 50): | |
| The number of denoising steps. | |
| sigmas (<code>list</code>, <em>optional</em>): | |
| Custom sigmas for the denoising process. | |
| *<em>denoiser_input_fields (<code>None</code>,</em>optional<em>): | |
| The conditional model inputs for the Anima denoiser. | |
| output_type (<code>str</code>,</em>optional*, defaults to pil): | |
| Output format: ‘pil’, ‘np’, ‘pt’.</p> <p>Outputs: | |
| images (<code>list</code>): | |
| Generated images.</p></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Text conditioner used by Anima to map Qwen3 hidden states and T5 token ids to Cosmos text embeddings.</p> <p>Anima reuses the Cosmos Predict2 DiT. The only model-specific conditioning module is this LLM adapter, which | |
| cross-attends from learned T5 token embeddings to Qwen3 text encoder hidden states before the diffusion loop. <code>target_dim</code> is the conditioner output dimension and must match the transformer’s <code>text_embed_dim</code>.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div></div> <!> <p></p>`,1);function X(A,x){P(x,!1),U(()=>{new URLSearchParams(window.location.search).get("fw")}),C();var m=Q();B("8uy2e4",_=>{var b=I();q(b,"content",G),T(_,b)});var c=e(Z(m),2);o(c,{title:"Anima",local:"anima",headingTag:"h1"});var l=e(c,4);J(l,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTW9kdWxhclBpcGVsaW5lJTBBJTBBcGlwZSUyMCUzRCUyME1vZHVsYXJQaXBlbGluZS5mcm9tX3ByZXRyYWluZWQoJTIyY2lyY2xlc3RvbmUtbGFicyUyRkFuaW1hLUJhc2UtdjEuMC1EaWZmdXNlcnMlMjIpJTBBcGlwZS5sb2FkX2NvbXBvbmVudHModG9yY2hfZHR5cGUlM0R0b3JjaC5iZmxvYXQxNiklMEFwaXBlLnRvKCUyMmN1ZGElMjIpJTBBJTBBaW1hZ2UlMjAlM0QlMjBwaXBlKHByb21wdCUzRCUyMm1hc3RlcnBpZWNlJTJDJTIwYmVzdCUyMHF1YWxpdHklMkMlMjAxZ2lybCUyQyUyMHNvbG8lMkMlMjBjaXR5JTIwbGlnaHRzJTIyKS5pbWFnZXMlNUIwJTVE",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ModularPipeline | |
| pipe = ModularPipeline.from_pretrained(<span class="hljs-string">"circlestone-labs/Anima-Base-v1.0-Diffusers"</span>) | |
| pipe.load_components(torch_dtype=torch.bfloat16) | |
| pipe.to(<span class="hljs-string">"cuda"</span>) | |
| image = pipe(prompt=<span class="hljs-string">"masterpiece, best quality, 1girl, solo, city lights"</span>).images[<span class="hljs-number">0</span>]`,lang:"python",wrap:!1});var p=e(l,2);o(p,{title:"AnimaModularPipeline",local:"diffusers.AnimaModularPipeline",headingTag:"h2"});var i=e(p,2),y=n(i);t(y,{name:"class diffusers.AnimaModularPipeline",anchor:"diffusers.AnimaModularPipeline",source:"https://github.com/huggingface/diffusers/blob/vr_13881/src/diffusers/modular_pipelines/anima/modular_pipeline.py#L19",parameters:[{name:"blocks",val:": diffusers.modular_pipelines.modular_pipeline.ModularPipelineBlocks | None = None"},{name:"pretrained_model_name_or_path",val:": str | os.PathLike | None = None"},{name:"components_manager",val:": diffusers.modular_pipelines.components_manager.ComponentsManager | None = None"},{name:"collection",val:": str | None = None"},{name:"modular_config_dict",val:": dict[str, typing.Any] | None = None"},{name:"config_dict",val:": dict[str, typing.Any] | None = None"},{name:"**kwargs",val:""}]}),d(4),a(i);var u=e(i,2);o(u,{title:"AnimaAutoBlocks",local:"diffusers.AnimaAutoBlocks",headingTag:"h2"});var s=e(u,2),w=n(s);t(w,{name:"class diffusers.AnimaAutoBlocks",anchor:"diffusers.AnimaAutoBlocks",source:"https://github.com/huggingface/diffusers/blob/vr_13881/src/diffusers/modular_pipelines/anima/modular_blocks_anima.py#L126",parameters:[]}),d(12),a(s);var f=e(s,2);o(f,{title:"AnimaTextConditioner",local:"diffusers.AnimaTextConditioner",headingTag:"h2"});var r=e(f,2),g=n(r);t(g,{name:"class diffusers.AnimaTextConditioner",anchor:"diffusers.AnimaTextConditioner",source:"https://github.com/huggingface/diffusers/blob/vr_13881/src/diffusers/models/condition_embedders/condition_embedder_anima.py#L229",parameters:[{name:"source_dim",val:": int = 1024"},{name:"target_dim",val:": int = 1024"},{name:"model_dim",val:": int = 1024"},{name:"num_layers",val:": int = 6"},{name:"num_attention_heads",val:": int = 16"},{name:"mlp_ratio",val:": float = 4.0"},{name:"target_vocab_size",val:": int = 32128"},{name:"use_self_attention",val:": bool = True"},{name:"use_layer_norm",val:": bool = False"},{name:"min_sequence_length",val:": int = 512"}]});var h=e(g,6),k=n(h);t(k,{name:"forward",anchor:"diffusers.AnimaTextConditioner.forward",source:"https://github.com/huggingface/diffusers/blob/vr_13881/src/diffusers/models/condition_embedders/condition_embedder_anima.py#L285",parameters:[{name:"source_hidden_states",val:": Tensor"},{name:"target_input_ids",val:": Tensor"},{name:"target_attention_mask",val:": typing.Optional[torch.Tensor] = None"},{name:"source_attention_mask",val:": typing.Optional[torch.Tensor] = None"}],parametersDescription:[{anchor:"diffusers.AnimaTextConditioner.forward.source_hidden_states",description:`<strong>source_hidden_states</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, source_sequence_length, source_dim)</code>) — | |
| Qwen3 text encoder hidden states to condition on.`,name:"source_hidden_states"},{anchor:"diffusers.AnimaTextConditioner.forward.target_input_ids",description:`<strong>target_input_ids</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, target_sequence_length)</code>) — | |
| T5 token ids used as learned query tokens.`,name:"target_input_ids"},{anchor:"diffusers.AnimaTextConditioner.forward.target_attention_mask",description:`<strong>target_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Attention mask for the target T5 token ids.`,name:"target_attention_mask"},{anchor:"diffusers.AnimaTextConditioner.forward.source_attention_mask",description:`<strong>source_attention_mask</strong> (<code>torch.Tensor</code>, <em>optional</em>) — | |
| Attention mask for the source Qwen3 hidden states.`,name:"source_attention_mask"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Text conditioning embeddings for the Cosmos transformer.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>torch.Tensor</code></p> | |
| `}),a(h),a(r);var M=e(r,2);N(M,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/anima.md"}),d(2),T(A,m),j()}export{X as component}; | |
Xet Storage Details
- Size:
- 9.4 kB
- Xet hash:
- 1878879d885185e07bb5c5b3bea02cbfa36cf216f4afea3f15f892d995e0a256
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.