Buckets:

download
raw
5.06 kB
import"../chunks/DsnmJJEf.js";import{i as v,h as b,C as T,H as m,D as l,E as w,s as D}from"../chunks/BtE7mKSK.js";import{p as y,o as E,s as e,f as L,a as f,b as A,c as u,d as p,r as h,n as F}from"../chunks/jDjavuwI.js";const C='{"title":"MiniMaxMusic3Transformer1DModel","local":"minimaxmusic3transformer1dmodel","sections":[{"title":"MiniMaxMusic3Transformer1DModel","local":"diffusers.MiniMaxMusic3Transformer1DModel","sections":[],"depth":2}],"depth":1}';var I=p('<meta name="hf:doc:metadata"/>'),P=p(`<p></p> <!> <!> <p>The 2.4B flow-matching Diffusion Transformer of <a href="https://huggingface.co/MiniMaxAI/MiniMax-Music3" rel="nofollow">MiniMax Music 3</a>. It
denoises 128-channel Flow-VAE audio latents conditioned on the per-frame hidden states of the model’s autoregressive
language-model stage, prepending the flow-matching timestep as an extra sequence token (a Stable-Audio-lineage
continuous transformer with partial rotary attention and GLU feedforwards).</p> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>The flow-matching diffusion transformer of MiniMax Music 3. It denoises Flow-VAE audio latents conditioned on
per-frame hidden states produced by the autoregressive language-model stage.</p> <p>Inputs are 1D latent sequences of shape <code>(batch, in_channels, length)</code>. The conditioning signal
(<code>encoder_hidden_states</code>, shape <code>(batch, length, condition_dim)</code>) must already be aligned to the latent timeline —
see <code>MiniMaxMusic3ConditionEncoder</code>. The flow-matching <code>timestep</code> runs from 0 (noise) to 1 (data).</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div></div> <!> <p></p>`,1);function q(M,g){y(g,!1),E(()=>{new URLSearchParams(window.location.search).get("fw")}),v();var a=P();b("g8i7ej",d=>{var c=I();D(c,"content",C),f(d,c)});var o=e(L(a),2);T(o,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var i=e(o,2);m(i,{title:"MiniMaxMusic3Transformer1DModel",local:"minimaxmusic3transformer1dmodel",headingTag:"h1"});var r=e(i,4);m(r,{title:"MiniMaxMusic3Transformer1DModel",local:"diffusers.MiniMaxMusic3Transformer1DModel",headingTag:"h2"});var n=e(r,2),s=u(n);l(s,{name:"class diffusers.MiniMaxMusic3Transformer1DModel",anchor:"diffusers.MiniMaxMusic3Transformer1DModel",source:"https://github.com/huggingface/diffusers/blob/vr_14386/src/diffusers/models/transformers/transformer_minimax_music3.py#L147",parameters:[{name:"in_channels",val:": int = 128"},{name:"condition_dim",val:": int = 2048"},{name:"num_layers",val:": int = 36"},{name:"num_attention_heads",val:": int = 32"},{name:"attention_head_dim",val:": int = 64"},{name:"ff_inner_dim",val:": int = 8192"},{name:"rotary_dim",val:": int = 32"},{name:"fourier_embedding_dim",val:": int = 256"}]});var t=e(s,6),_=u(t);l(_,{name:"forward",anchor:"diffusers.MiniMaxMusic3Transformer1DModel.forward",source:"https://github.com/huggingface/diffusers/blob/vr_14386/src/diffusers/models/transformers/transformer_minimax_music3.py#L196",parameters:[{name:"hidden_states",val:": Tensor"},{name:"timestep",val:": Tensor"},{name:"encoder_hidden_states",val:": Tensor"},{name:"return_dict",val:": bool = True"}],parametersDescription:[{anchor:"diffusers.MiniMaxMusic3Transformer1DModel.forward.hidden_states",description:`<strong>hidden_states</strong> (<code>torch.Tensor</code> of shape <code>(batch, in_channels, length)</code>) &#x2014;
Noisy Flow-VAE latents.`,name:"hidden_states"},{anchor:"diffusers.MiniMaxMusic3Transformer1DModel.forward.timestep",description:`<strong>timestep</strong> (<code>torch.Tensor</code> of shape <code>(batch,)</code>) &#x2014;
Flow-matching time in <code>[0, 1]</code>, where 0 is pure noise and 1 is data.`,name:"timestep"},{anchor:"diffusers.MiniMaxMusic3Transformer1DModel.forward.encoder_hidden_states",description:`<strong>encoder_hidden_states</strong> (<code>torch.Tensor</code> of shape <code>(batch, length, condition_dim)</code>) &#x2014;
Frame-aligned conditioning from <code>MiniMaxMusic3ConditionEncoder</code>. Pass zeros for the unconditional
branch of classifier-free guidance.`,name:"encoder_hidden_states"},{anchor:"diffusers.MiniMaxMusic3Transformer1DModel.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, defaults to <code>True</code>) &#x2014;
Whether to return a <a href="/docs/diffusers/pr_14386/en/api/models/hunyuan_video15_transformer_3d#diffusers.models.modeling_outputs.Transformer2DModelOutput">Transformer2DModelOutput</a> instead of a plain tuple.`,name:"return_dict"}],returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>The predicted flow-matching velocity with the same shape as <code>hidden_states</code>.</p>
`}),h(t),h(n);var x=e(n,2);w(x,{source:"https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/models/minimax_music3_transformer.md"}),F(2),f(M,a),A()}export{q as component};

Xet Storage Details

Size:
5.06 kB
·
Xet hash:
c6541e733c2654db914abc0d98c9c05e659f02d3cf08d5e1795d9360c1bf3c58

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.