Buckets:
| import"../chunks/DsnmJJEf.js";import{i as y,h as H,C as D,H as t,a as q,D as r,E as z,s as j}from"../chunks/BtE7mKSK.js";import{p as A,o as J,s as e,f as L,a as M,b as N,c as i,d as x,r as a,n as g}from"../chunks/jDjavuwI.js";const U='{"title":"MiniMaxH3Transformer3DModel","local":"minimaxh3transformer3dmodel","sections":[{"title":"MiniMaxH3Transformer3DModel","local":"diffusers.MiniMaxH3Transformer3DModel","sections":[],"depth":2},{"title":"MiniMaxH3TransformerOutput","local":"diffusers.models.transformers.transformer_minimax_h3.MiniMaxH3TransformerOutput","sections":[],"depth":2}],"depth":1}';var B=x('<meta name="hf:doc:metadata"/>'),E=x(`<p></p> <!> <!> <p>A Diffusion Transformer model for joint video and audio generation, introduced in <a href="https://huggingface.co/MiniMaxAI" rel="nofollow">MiniMax-H3</a> by MiniMax.</p> <p>MiniMax-H3 runs a single stack of blocks over <strong>one packed 1-D sequence</strong> that holds the text conditioning, the conditioning image and video rows, the audio rows and the target video rows at once. Attention is full self-attention over that sequence, so there is no cross-attention and no per-modality block weights. Modality-specific behaviour comes only from the two input patch projections, the per-row modality tag that selects the AdaLN modulation parameters, and the two output heads.</p> <p>Building the packed layout is the caller’s job, which is why the forward signature takes the layout apart from the latents: the <code>(t, h, w)</code> position grid, the per-row modality tags, the per-row timestep indices and the three index tensors that address the video, audio and text rows. <a href="/docs/diffusers/pr_14355/en/api/pipelines/minimax_h3#diffusers.MiniMaxH3Blocks">MiniMaxH3Blocks</a> and <a href="/docs/diffusers/pr_14355/en/api/pipelines/minimax_h3#diffusers.MiniMaxH3Ref2VABlocks">MiniMaxH3Ref2VABlocks</a> build all of it.</p> <p>A layout that carries padding rows (tag <code>-1</code>) needs a masked attention backend, since those rows are kept in their own attention document by a boolean mask; a padless sequence needs no mask and keeps every backend available.</p> <p>One repository holds both released checkpoint partitions, so the subfolder is what selects the task: <code>transformer/</code> for the text and keyframe tasks, <code>transformer_ref/</code> for the omni-reference task.</p> <!> <p>The checkpoint is mixed precision: the two input patch projections, the timestep MLP and the two output heads are float32 while the block stack is bfloat16. <code>from_pretrained</code> keeps that layout through <code>_keep_in_fp32_modules</code>, so pass <code>dtype=torch.bfloat16</code> and let it place the float32 modules rather than casting the model with <code>.to(torch.bfloat16)</code> afterwards.</p> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>A Transformer model for joint video + audio generation, introduced in MiniMax-H3.</p> <p>MiniMax-H3 runs a single stack of blocks over <strong>one packed 1-D sequence</strong> that holds the text condition, the | |
| conditioning image / video rows, the audio rows and the target video rows. Attention is full self-attention over | |
| that sequence; there is no cross-attention and no per-modality block weights. Modality-specific behaviour comes | |
| only from the two input patch projections, the per-row AdaLN modality tag, and the two output heads.</p> <p>The caller is responsible for building the packed layout: patchifying the video latents, ordering the rows, and | |
| producing the <code>(t, h, w)</code> position grid, the per-row modality tags and the per-row timestep indices. Padding rows | |
| (tag <code>-1</code>) are kept in a separate attention document, matching the reference implementation, which pads to a | |
| multiple of 64 for FlashAttention with <code>cu_seqlens = [0, used, S]</code>. Prefer dropping them — a padless sequence | |
| needs no attention mask, keeping the unmasked attention backends available.</p> <p>The batch axis is a pure replication axis: the structural arguments (<code>timestep</code>, <code>timestep_indices</code>, <code>token_tags</code>, <code>position_ids</code> and the three index tensors) describe one packed layout that every batch item shares, and each item | |
| is a single attention document.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The output of <a href="/docs/diffusers/pr_14355/en/api/models/minimax_h3_transformer3d#diffusers.MiniMaxH3Transformer3DModel">MiniMaxH3Transformer3DModel</a>.</p></div> <!> <p></p>`,1);function I(T,v){A(v,!1),J(()=>{new URLSearchParams(window.location.search).get("fw")}),y();var s=E();H("17kdxl5",u=>{var _=B();j(_,"content",U),M(u,_)});var d=e(L(s),2);D(d,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var c=e(d,2);t(c,{title:"MiniMaxH3Transformer3DModel",local:"minimaxh3transformer3dmodel",headingTag:"h1"});var m=e(c,12);q(m,{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwZGlmZnVzZXJzJTIwaW1wb3J0JTIwTWluaU1heEgzVHJhbnNmb3JtZXIzRE1vZGVsJTBBJTBBdHJhbnNmb3JtZXIlMjAlM0QlMjBNaW5pTWF4SDNUcmFuc2Zvcm1lcjNETW9kZWwuZnJvbV9wcmV0cmFpbmVkKCUwQSUyMCUyMCUyMCUyMCUyMk1pbmlNYXhBSSUyRk1pbmlNYXgtSDMlMjIlMkMlMjBzdWJmb2xkZXIlM0QlMjJ0cmFuc2Zvcm1lciUyMiUyQyUyMGR0eXBlJTNEdG9yY2guYmZsb2F0MTYlMEEpLnRvKCUyMmN1ZGElMjIp",highlighted:`<span class="hljs-keyword">import</span> torch | |
| <span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MiniMaxH3Transformer3DModel | |
| transformer = MiniMaxH3Transformer3DModel.from_pretrained( | |
| <span class="hljs-string">"MiniMaxAI/MiniMax-H3"</span>, subfolder=<span class="hljs-string">"transformer"</span>, dtype=torch.bfloat16 | |
| ).to(<span class="hljs-string">"cuda"</span>)`,lang:"python",wrap:!1});var f=e(m,4);t(f,{title:"MiniMaxH3Transformer3DModel",local:"diffusers.MiniMaxH3Transformer3DModel",headingTag:"h2"});var o=e(f,2),h=i(o);r(h,{name:"class diffusers.MiniMaxH3Transformer3DModel",anchor:"diffusers.MiniMaxH3Transformer3DModel",source:"https://github.com/huggingface/diffusers/blob/vr_14355/src/diffusers/models/transformers/transformer_minimax_h3.py#L374",parameters:[{name:"num_attention_heads",val:": int = 56"},{name:"attention_head_dim",val:": int = 128"},{name:"hidden_size",val:": int = 5376"},{name:"num_layers",val:": int = 50"},{name:"num_refiner_layers",val:": int = 2"},{name:"ffn_dim",val:": int = 14336"},{name:"in_channels",val:": int = 24"},{name:"audio_in_channels",val:": int = 32"},{name:"patch_size",val:": tuple = (1, 2, 2)"},{name:"text_dim",val:": int = 5120"},{name:"freq_dim",val:": int = 256"},{name:"time_embed_hidden_dim",val:": int = 5376"},{name:"time_embed_dim",val:": int = 2688"},{name:"rope_freq_dim",val:": int = 16"},{name:"rope_theta",val:": float = 10000.0"},{name:"norm_eps",val:": float = 1e-05"},{name:"qk_norm_eps",val:": float = 1e-05"},{name:"final_norm_eps",val:": float = 1e-05"}],parametersDescription:[{anchor:"diffusers.MiniMaxH3Transformer3DModel.num_attention_heads",description:`<strong>num_attention_heads</strong> (<code>int</code>, defaults to <code>56</code>) — | |
| The number of heads to use for multi-head attention.`,name:"num_attention_heads"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.attention_head_dim",description:`<strong>attention_head_dim</strong> (<code>int</code>, defaults to <code>128</code>) — | |
| The number of channels in each attention head. Note that <code>num_attention_heads * attention_head_dim</code> is | |
| <em>larger</em> than <code>hidden_size</code> in MiniMax-H3.`,name:"attention_head_dim"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.hidden_size",description:`<strong>hidden_size</strong> (<code>int</code>, defaults to <code>5376</code>) — | |
| The number of channels of the packed sequence (the residual stream).`,name:"hidden_size"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.num_layers",description:`<strong>num_layers</strong> (<code>int</code>, defaults to <code>50</code>) — | |
| The number of transformer blocks.`,name:"num_layers"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.num_refiner_layers",description:`<strong>num_refiner_layers</strong> (<code>int</code>, defaults to <code>2</code>) — | |
| The number of token refiner blocks applied to the projected text stream.`,name:"num_refiner_layers"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.ffn_dim",description:`<strong>ffn_dim</strong> (<code>int</code>, defaults to <code>14336</code>) — | |
| The inner dimension of the SwiGLU feed-forward layers.`,name:"ffn_dim"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.in_channels",description:`<strong>in_channels</strong> (<code>int</code>, defaults to <code>24</code>) — | |
| The number of channels of the video latents.`,name:"in_channels"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.audio_in_channels",description:`<strong>audio_in_channels</strong> (<code>int</code>, defaults to <code>32</code>) — | |
| The number of channels of the audio latents.`,name:"audio_in_channels"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.patch_size",description:`<strong>patch_size</strong> (<code>tuple[int, int, int]</code>, defaults to <code>(1, 2, 2)</code>) — | |
| The <code>(t, h, w)</code> patch used to pack the video latents into rows.`,name:"patch_size"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.text_dim",description:`<strong>text_dim</strong> (<code>int</code>, defaults to <code>5120</code>) — | |
| The number of channels of the text conditioning produced by the text encoder.`,name:"text_dim"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.freq_dim",description:`<strong>freq_dim</strong> (<code>int</code>, defaults to <code>256</code>) — | |
| The dimension of the sinusoidal timestep embedding. Timesteps are consumed unscaled in <code>[0, 1]</code>.`,name:"freq_dim"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.time_embed_hidden_dim",description:`<strong>time_embed_hidden_dim</strong> (<code>int</code>, defaults to <code>5376</code>) — | |
| The inner dimension of the timestep MLP.`,name:"time_embed_hidden_dim"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.time_embed_dim",description:`<strong>time_embed_dim</strong> (<code>int</code>, defaults to <code>2688</code>) — | |
| The output dimension of the timestep MLP, i.e. the input of every AdaLN projection.`,name:"time_embed_dim"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.rope_freq_dim",description:`<strong>rope_freq_dim</strong> (<code>int</code>, defaults to <code>16</code>) — | |
| The number of rotary frequencies per axis. The <code>(t, h, w)</code> axes share one <code>inv_freq</code> buffer of this length | |
| and <code>2 * 3 * rope_freq_dim</code> of the <code>attention_head_dim</code> channels are rotated.`,name:"rope_freq_dim"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.rope_theta",description:`<strong>rope_theta</strong> (<code>float</code>, defaults to <code>10000.0</code>) — | |
| The base of the rotary frequency schedule the <code>rope.inv_freq</code> buffer is computed from.`,name:"rope_theta"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.norm_eps",description:`<strong>norm_eps</strong> (<code>float</code>, defaults to <code>1e-5</code>) — | |
| Epsilon of the pre-attention and pre-feed-forward norms.`,name:"norm_eps"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.qk_norm_eps",description:`<strong>qk_norm_eps</strong> (<code>float</code>, defaults to <code>1e-5</code>) — | |
| Epsilon of the per-head query/key norms.`,name:"qk_norm_eps"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.final_norm_eps",description:`<strong>final_norm_eps</strong> (<code>float</code>, defaults to <code>1e-5</code>) — | |
| Epsilon of the token refiner output norm and of <code>norm_out</code>.`,name:"final_norm_eps"}]});var l=e(h,10),b=i(l);r(b,{name:"forward",anchor:"diffusers.MiniMaxH3Transformer3DModel.forward",source:"https://github.com/huggingface/diffusers/blob/vr_14355/src/diffusers/models/transformers/transformer_minimax_h3.py#L529",parameters:[{name:"hidden_states",val:": Tensor"},{name:"audio_hidden_states",val:": Tensor"},{name:"encoder_hidden_states",val:": Tensor"},{name:"timestep",val:": Tensor"},{name:"timestep_indices",val:": Tensor"},{name:"token_tags",val:": Tensor"},{name:"position_ids",val:": Tensor"},{name:"video_indices",val:": Tensor"},{name:"audio_indices",val:": Tensor"},{name:"text_indices",val:": Tensor"},{name:"attention_kwargs",val:": dict[str, typing.Any] | None = None"},{name:"return_dict",val:": bool = True"}],parametersDescription:[{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.hidden_states",description:`<strong>hidden_states</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, num_video_tokens, in_channels * prod(patch_size))</code>) — | |
| Patchified video latent rows — conditioning rows and target rows — ordered as they appear in the packed | |
| sequence, i.e. matching <code>video_indices</code>.`,name:"hidden_states"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.audio_hidden_states",description:`<strong>audio_hidden_states</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, num_audio_tokens, audio_in_channels)</code>) — | |
| Audio latent rows, ordered to match <code>audio_indices</code>.`,name:"audio_hidden_states"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.encoder_hidden_states",description:`<strong>encoder_hidden_states</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, num_text_tokens, text_dim)</code>) — | |
| Text conditioning, ordered to match <code>text_indices</code>.`,name:"encoder_hidden_states"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.timestep",description:`<strong>timestep</strong> (<code>torch.Tensor</code> of shape <code>(num_timesteps,)</code>) — | |
| The <em>distinct</em> timestep values present in the packed sequence, in <code>[0, 1]</code> and unscaled. One forward | |
| serves rows at different noise levels (target video, target audio, conditioning rows).`,name:"timestep"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.timestep_indices",description:`<strong>timestep_indices</strong> (<code>torch.Tensor</code> of shape <code>(seq_len,)</code>) — | |
| For every row of the packed sequence, the index of its timestep in <code>timestep</code>.`,name:"timestep_indices"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.token_tags",description:`<strong>token_tags</strong> (<code>torch.Tensor</code> of shape <code>(seq_len,)</code>) — | |
| For every row of the packed sequence, its modality: <code>0</code> video, <code>1</code> text, <code>2</code> audio, <code>-1</code> padding. | |
| Padding rows form their own attention document and never reach the outputs.`,name:"token_tags"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.position_ids",description:`<strong>position_ids</strong> (<code>torch.Tensor</code> of shape <code>(seq_len, 3)</code>) — | |
| The <code>(t, h, w)</code> rotary coordinates of every row of the packed sequence.`,name:"position_ids"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.video_indices",description:`<strong>video_indices</strong> (<code>torch.Tensor</code> of shape <code>(num_video_tokens,)</code>) — | |
| Positions of the video rows in the packed sequence.`,name:"video_indices"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.audio_indices",description:`<strong>audio_indices</strong> (<code>torch.Tensor</code> of shape <code>(num_audio_tokens,)</code>) — | |
| Positions of the audio rows in the packed sequence.`,name:"audio_indices"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.text_indices",description:`<strong>text_indices</strong> (<code>torch.Tensor</code> of shape <code>(num_text_tokens,)</code>) — | |
| Positions of the text rows in the packed sequence.`,name:"text_indices"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.attention_kwargs",description:`<strong>attention_kwargs</strong> (<code>dict</code>, <em>optional</em>) — | |
| A kwargs dictionary that, if specified, may carry a <code>scale</code> entry which is applied to the LoRA layers.`,name:"attention_kwargs"},{anchor:"diffusers.MiniMaxH3Transformer3DModel.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, defaults to <code>True</code>) — | |
| Whether to return a <code>MiniMaxH3TransformerOutput</code> instead of a plain tuple.`,name:"return_dict"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The video velocity of shape <code>(batch_size, num_video_tokens, in_channels * prod(patch_size))</code> and the | |
| audio velocity of shape <code>(batch_size, num_audio_tokens, audio_in_channels)</code>, in the row order of | |
| <code>video_indices</code> and <code>audio_indices</code>.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>MiniMaxH3TransformerOutput</code> or <code>tuple</code></p> | |
| `}),a(l),a(o);var p=e(o,2);t(p,{title:"MiniMaxH3TransformerOutput",local:"diffusers.models.transformers.transformer_minimax_h3.MiniMaxH3TransformerOutput",headingTag:"h2"});var n=e(p,2),w=i(n);r(w,{name:"class diffusers.models.transformers.transformer_minimax_h3.MiniMaxH3TransformerOutput",anchor:"diffusers.models.transformers.transformer_minimax_h3.MiniMaxH3TransformerOutput",source:"https://github.com/huggingface/diffusers/blob/vr_14355/src/diffusers/models/transformers/transformer_minimax_h3.py#L40",parameters:[{name:"sample",val:": Tensor"},{name:"audio_sample",val:": Tensor"}],parametersDescription:[{anchor:"diffusers.models.transformers.transformer_minimax_h3.MiniMaxH3TransformerOutput.sample",description:`<strong>sample</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, num_video_tokens, in_channels * prod(patch_size))</code>) — | |
| The video velocity prediction for the rows addressed by <code>video_indices</code>, in the same order. Conditioning | |
| rows are returned unmasked — masking them out before the scheduler step is the caller’s job.`,name:"sample"},{anchor:"diffusers.models.transformers.transformer_minimax_h3.MiniMaxH3TransformerOutput.audio_sample",description:`<strong>audio_sample</strong> (<code>torch.Tensor</code> of shape <code>(batch_size, num_audio_tokens, audio_in_channels)</code>) — | |
| The audio velocity prediction for the rows addressed by <code>audio_indices</code>, in the same order.`,name:"audio_sample"}]}),g(2),a(n);var k=e(n,2);z(k,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/models/minimax_h3_transformer3d.md"}),g(2),M(T,s),N()}export{I as component}; | |
Xet Storage Details
- Size:
- 18.8 kB
- Xet hash:
- 94ef0b31739d424ebd6dc516aecca4318579526dd7498bd62566db1a21baf954
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.