Buckets:

download
raw
122 kB
<meta charset="utf-8" /><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;MiniMax-H3&quot;,&quot;local&quot;:&quot;minimax-h3&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Checkpoint layout&quot;,&quot;local&quot;:&quot;checkpoint-layout&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Two schedulers&quot;,&quot;local&quot;:&quot;two-schedulers&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Generation constraints&quot;,&quot;local&quot;:&quot;generation-constraints&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Memory&quot;,&quot;local&quot;:&quot;memory&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Text and keyframes&quot;,&quot;local&quot;:&quot;text-and-keyframes&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Omni-references&quot;,&quot;local&quot;:&quot;omni-references&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;A generation as a reference&quot;,&quot;local&quot;:&quot;a-generation-as-a-reference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3ModularPipeline&quot;,&quot;local&quot;:&quot;diffusers.MiniMaxH3ModularPipeline&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3Blocks&quot;,&quot;local&quot;:&quot;diffusers.MiniMaxH3Blocks&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3ImageReference&quot;,&quot;local&quot;:&quot;diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3VideoReference&quot;,&quot;local&quot;:&quot;diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3AudioReference&quot;,&quot;local&quot;:&quot;diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/>
<link href="/docs/diffusers/pr_14407/en/_app/immutable/entry/start.Ic04TDLP.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/D6cpoeLy.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/DK803DsY.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/entry/app.DrFCO7E8.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/DTwaC60R.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/BTASUwav.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/Y1jgPe5f.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/DsnmJJEf.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/nodes/0.UUeUJWYX.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/BzOvRKAw.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/nodes/190.CATsNqvL.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/CmJXCtRL.js" rel="modulepreload">
<link href="/docs/diffusers/pr_14407/en/_app/immutable/chunks/Bu2vAape.js" rel="modulepreload">
<!--1oh9kud--><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;MiniMax-H3&quot;,&quot;local&quot;:&quot;minimax-h3&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Checkpoint layout&quot;,&quot;local&quot;:&quot;checkpoint-layout&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Two schedulers&quot;,&quot;local&quot;:&quot;two-schedulers&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Generation constraints&quot;,&quot;local&quot;:&quot;generation-constraints&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Memory&quot;,&quot;local&quot;:&quot;memory&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Text and keyframes&quot;,&quot;local&quot;:&quot;text-and-keyframes&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Omni-references&quot;,&quot;local&quot;:&quot;omni-references&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;A generation as a reference&quot;,&quot;local&quot;:&quot;a-generation-as-a-reference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3ModularPipeline&quot;,&quot;local&quot;:&quot;diffusers.MiniMaxH3ModularPipeline&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3Blocks&quot;,&quot;local&quot;:&quot;diffusers.MiniMaxH3Blocks&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3ImageReference&quot;,&quot;local&quot;:&quot;diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3VideoReference&quot;,&quot;local&quot;:&quot;diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;MiniMaxH3AudioReference&quot;,&quot;local&quot;:&quot;diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2}],&quot;depth&quot;:1}"/><!---->
<link href="/docs/diffusers/pr_14407/en/_app/immutable/assets/0.tn0RQdqM.css" rel="modulepreload"> <!--[--><!--[0--><!--[--><!--[0--><!--[--><p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg><!----></button></div> <!--[-1--><!--]--></div><!----> <!--[0--><h1 class="relative group"><a id="minimax-h3" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#minimax-h3"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MiniMax-H3</span></h1><!--]--><!----> <blockquote class="tip"><p>MiniMax-H3 is not part of a diffusers release yet. Install diffusers from source to use it: <code>pip install git+https://github.com/huggingface/diffusers.git</code></p></blockquote> <p>MiniMax-H3 generates video and its soundtrack together. A single transformer denoises one packed sequence containing the text conditioning, conditioning media, and target video and audio latents. There is no separate vocoder and no audio post-hoc pass: video and audio come out of the same denoising loop.</p> <p>You can find the original MiniMax-H3 checkpoints under the <a href="https://huggingface.co/MiniMaxAI" rel="nofollow">MiniMaxAI</a> organization.</p> <p>MiniMax-H3 is integrated as <a href="../../modular_diffusers/overview">Modular Diffusers</a> blocks only, the way <a href="./anima">Anima</a> is: the blocks and their <a href="/docs/diffusers/pr_14407/en/api/pipelines/minimax_h3#diffusers.MiniMaxH3ModularPipeline">MiniMaxH3ModularPipeline</a> are the whole integration, and there is no <code>DiffusionPipeline</code> half.</p> <!--[1--><h2 class="relative group"><a id="checkpoint-layout" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#checkpoint-layout"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Checkpoint layout</span></h2><!--]--><!----> <p>MiniMax-H3 was released as two checkpoint partitions that share every component except the transformer, so the diffusers conversion puts both in <strong>one repository</strong>:</p> <table><thead><tr><th>Subfolder</th><th>Workflows</th></tr></thead><tbody><tr><td><code>transformer/</code></td><td><code>t2va</code> (text only), <code>fl2va</code> (first and/or last keyframe)</td></tr><tr><td><code>transformer_ref/</code></td><td><code>ref2va</code> (an ordered mix of image, video and audio references)</td></tr></tbody></table> <p>Everything but the transformer, i.e. the video VAE, the audio VAE, the Qwen3-VL conditioner, its tokenizer and processor, and the two schedulers, is shared and stored once.</p> <p>The conditioner is a <code>Qwen3VLForConditionalGeneration</code>, and MiniMax-H3 reads the <em>unnormalized</em> hidden state after its 50th decoder layer rather than the last one, so the full released checkpoint is used with its language-model head unused.</p> <p>All three tasks are workflows of the one <a href="/docs/diffusers/pr_14407/en/api/pipelines/minimax_h3#diffusers.MiniMaxH3Blocks">MiniMaxH3Blocks</a>, and the repository carries one <code>modular_model_index.json</code> naming every component with its own loading spec. To serve a single task, pass the workflow to <code>from_pretrained</code>: it keeps only that workflow’s blocks, so the pipeline’s signature (<code>pipe.doc</code>) documents exactly that task’s inputs, only that task’s components are declared, and <code>load_components</code> fetches exactly their subfolders — a <code>t2va</code> / <code>fl2va</code> pipeline never touches <code>transformer_ref/</code>, a <code>ref2va</code> one never touches <code>transformer/</code>.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ModularPipeline
pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, workflow=<span class="hljs-string">&quot;ref2va&quot;</span>)
pipe.load_components(dtype=torch.bfloat16)<!----></pre></div><!----> <blockquote class="tip"><p><code>pipe.doc</code> prints what the pipeline in front of you takes and returns — every input with its default, the components it expects and the outputs it produces. Pruned to one workflow it describes exactly that task, which is the quickest way to see what a request needs before making one.</p></blockquote> <p>To keep every workflow available on one pipeline instead, leave the <code>workflow</code> argument out: the pipeline then picks the workflow per call from the inputs it is passed.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!---->pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>)<!----></pre></div><!----> <p>The <em>loading</em> can still go one workflow at a time. This one call fetches <code>transformer/</code> and every shared component, which serves both <code>t2va</code> and <code>fl2va</code>:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!---->pipe.load_components(workflow=<span class="hljs-string">&quot;t2va&quot;</span>, dtype=torch.bfloat16)<!----></pre></div><!----> <p>A plain <code>load_components()</code> with no <code>workflow=</code> pulls <strong>both</strong> 61.7GB transformer partitions, which is what lets one pipeline serve all three workflows without another loading call. Pair it with a <a href="/docs/diffusers/pr_14407/en/api/modular_diffusers/pipeline_components#diffusers.ComponentsManager">ComponentsManager</a> and auto offloading: the weights live in host RAM and the manager moves onto the accelerator just what each step needs, so when the <code>ref2va</code> denoiser wants the device the strategy offloads whatever frees enough room. See <a href="#memory">Memory</a> for the recipes.</p> <!--[1--><h2 class="relative group"><a id="two-schedulers" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#two-schedulers"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Two schedulers</span></h2><!--]--><!----> <p>Video and audio latents step down two different schedules inside a single transformer call per step, which is why the blocks expect two <a href="/docs/diffusers/pr_14407/en/api/schedulers/minimax_h3#diffusers.MiniMaxH3Scheduler">MiniMaxH3Scheduler</a> instances: <code>scheduler</code> for the video latents (<code>shift=12.0</code> in the released checkpoints) and <code>audio_scheduler</code> for the audio latents (<code>shift=3.0</code>).</p> <p>Both transformer partitions are guidance-distilled, so this holds for every workflow: guidance is baked into the weights, there is no guider, no <code>negative_prompt</code> and no <code>guidance_scale</code>, and every step runs exactly one forward pass.</p> <!--[1--><h2 class="relative group"><a id="generation-constraints" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#generation-constraints"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Generation constraints</span></h2><!--]--><!----> <ul><li><strong>24 fps, 5 to 15 seconds.</strong> <code>num_frames</code> is snapped up to the next <code>17 * n + 5</code> the video VAE can decode, and the resulting duration has to stay in that window.</li> <li><strong>A 768 pixel short edge.</strong> <code>height</code> and <code>width</code> default to MiniMax-H3’s own canvas for the aspect ratio of the first keyframe (or 16:9 without one) and must be multiples of 32.</li> <li><strong>One generator, three draws.</strong> A request draws the keyframe or reference conditioning noise first, then the video noise, then the audio noise, all from the <code>generator</code> it is passed, so two runs from the same generator state return the same video and soundtrack. Passing <code>latents</code> or <code>audio_latents</code> replaces the corresponding draw.</li> <li><strong><code>num_inference_steps</code> counts sigma grid points</strong>, the terminal <code>0</code> included, so it drives one model evaluation less.</li></ul> <!--[1--><h2 class="relative group"><a id="memory" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#memory"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Memory</span></h2><!--]--><!----> <p>The transformer alone is 61.7 GB in bfloat16 and the Qwen3-VL conditioner is another 62.1 GB, so the loading recipe depends on the hardware. Smaller canvases are the biggest speed lever on every setup: <code>height</code> and <code>width</code> only have to be multiples of 32, and 960x544 runs about 2.3x faster per step than the trained 1344x768.</p> <p>On one 80 GB card, register the components in a <a href="/docs/diffusers/pr_14407/en/api/modular_diffusers/pipeline_components#diffusers.ComponentsManager">ComponentsManager</a> and let it move them on and off the accelerator:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ComponentsManager, ModularPipeline
manager = ComponentsManager()
pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, components_manager=manager)
pipe.load_components(workflow=<span class="hljs-string">&quot;t2va&quot;</span>, dtype=torch.bfloat16)
manager.enable_auto_cpu_offload(device=<span class="hljs-string">&quot;cuda&quot;</span>, memory_reserve_margin=<span class="hljs-string">&quot;12GB&quot;</span>)
pipe.transformer.set_attention_backend(<span class="hljs-string">&quot;_flash_3_hub&quot;</span>) <span class="hljs-comment"># Hopper, roughly 3x faster; kernels fetched from the Hub</span><!----></pre></div><!----> <p>On a consumer card (24 to 32 GB), quantize the two large components to int8 as they load and stream the transformer’s blocks from CPU RAM. Everything below uses supported loaders only, no patches, and works straight from the bfloat16 checkpoint:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> MiniMaxH3Transformer3DModel, ModularPipeline, TorchAoConfig
<span class="hljs-keyword">from</span> diffusers.hooks <span class="hljs-keyword">import</span> apply_group_offloading
<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> Qwen3VLForConditionalGeneration
<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TorchAoConfig <span class="hljs-keyword">as</span> TransformersTorchAoConfig
<span class="hljs-keyword">from</span> torchao.quantization <span class="hljs-keyword">import</span> Int8WeightOnlyConfig
pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>)
pipe.update_components(
transformer=MiniMaxH3Transformer3DModel.from_pretrained(
<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, subfolder=<span class="hljs-string">&quot;transformer&quot;</span>, dtype=torch.bfloat16,
quantization_config=TorchAoConfig(
Int8WeightOnlyConfig(version=<span class="hljs-number">2</span>),
modules_to_not_convert=[
<span class="hljs-string">&quot;proj_in&quot;</span>, <span class="hljs-string">&quot;audio_proj_in&quot;</span>, <span class="hljs-string">&quot;context_embedder&quot;</span>, <span class="hljs-string">&quot;time_embedder&quot;</span>, <span class="hljs-string">&quot;time_proj&quot;</span>,
<span class="hljs-string">&quot;token_refiner&quot;</span>, <span class="hljs-string">&quot;norm_out&quot;</span>, <span class="hljs-string">&quot;proj_out&quot;</span>, <span class="hljs-string">&quot;audio_proj_out&quot;</span>,
],
),
low_cpu_mem_usage=<span class="hljs-literal">False</span>,
),
text_encoder=Qwen3VLForConditionalGeneration.from_pretrained(
<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, subfolder=<span class="hljs-string">&quot;text_encoder&quot;</span>, dtype=torch.bfloat16,
quantization_config=TransformersTorchAoConfig(
Int8WeightOnlyConfig(version=<span class="hljs-number">2</span>),
modules_to_not_convert=[<span class="hljs-string">&quot;model.visual&quot;</span>, <span class="hljs-string">&quot;model.language_model.embed_tokens&quot;</span>, <span class="hljs-string">&quot;model.language_model.norm&quot;</span>, <span class="hljs-string">&quot;lm_head&quot;</span>],
),
),
)
pipe.load_components(workflow=<span class="hljs-string">&quot;t2va&quot;</span>, dtype=torch.bfloat16)
<span class="hljs-comment"># version=2 int8 tensors are pinnable, which streamed offload needs, and freezing removes the one autograd</span>
<span class="hljs-comment"># path the quantized tensors cannot serve.</span>
pipe.transformer.requires_grad_(<span class="hljs-literal">False</span>)
pipe.text_encoder.requires_grad_(<span class="hljs-literal">False</span>)
offload = <span class="hljs-built_in">dict</span>(onload_device=torch.device(<span class="hljs-string">&quot;cuda&quot;</span>), offload_device=torch.device(<span class="hljs-string">&quot;cpu&quot;</span>), use_stream=<span class="hljs-literal">True</span>)
pipe.transformer.enable_group_offload(offload_type=<span class="hljs-string">&quot;block_level&quot;</span>, num_blocks_per_group=<span class="hljs-number">1</span>, **offload)
apply_group_offloading(pipe.text_encoder.model, offload_type=<span class="hljs-string">&quot;leaf_level&quot;</span>, **offload)
pipe.vae.to(<span class="hljs-string">&quot;cuda&quot;</span>)
pipe.audio_vae.to(<span class="hljs-string">&quot;cuda&quot;</span>)<!----></pre></div><!----> <p>On 12 to 16 GB the same recipe works with the video VAE group offloaded too (<code>offload_type="leaf_level"</code>, no stream) and a small canvas such as 960x544. Expect the weights to live in host RAM: around 75 GB of it at int8.</p> <p>With two cards nothing has to be offloaded: split the pipeline in two and put each half on its own device.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ComponentsManager, ModularPipeline
workflow = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>).blocks.get_workflow(<span class="hljs-string">&quot;t2va&quot;</span>)
text_manager = ComponentsManager()
text_manager.enable_auto_cpu_offload(device=<span class="hljs-string">&quot;cuda:1&quot;</span>)
conditioner = workflow.sub_blocks.pop(<span class="hljs-string">&quot;text_encoder&quot;</span>).init_pipeline(
<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, components_manager=text_manager
)
conditioner.load_components(dtype=torch.bfloat16)
manager = ComponentsManager()
manager.enable_auto_cpu_offload(device=<span class="hljs-string">&quot;cuda:0&quot;</span>)
rest = workflow.init_pipeline(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, components_manager=manager)
rest.load_components(dtype=torch.bfloat16)
prompt = <span class="hljs-string">&quot;A red fox trotting through a snowy pine forest, snow crunching underfoot&quot;</span>
state = conditioner(prompt=prompt)
results = rest(
state=state,
num_frames=<span class="hljs-number">124</span>,
generator=torch.Generator().manual_seed(<span class="hljs-number">42</span>),
output=[<span class="hljs-string">&quot;videos&quot;</span>, <span class="hljs-string">&quot;audio&quot;</span>, <span class="hljs-string">&quot;sampling_rate&quot;</span>],
)<!----></pre></div><!----> <p>Two 80 GB cards run full bfloat16 this way: each half fits on its own card, so nothing is evicted back to host memory once it is resident. Two 48 GB cards do the same with the int8 loading above on both components.</p> <!--[1--><h2 class="relative group"><a id="text-and-keyframes" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#text-and-keyframes"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Text and keyframes</span></h2><!--]--><!----> <p><a href="/docs/diffusers/pr_14407/en/api/pipelines/minimax_h3#diffusers.MiniMaxH3Blocks">MiniMaxH3Blocks</a> covers text-to-video-and-audio and keyframe conditioning. A keyframe can be the frame the video starts from (<code>image</code>), the frame it ends on (<code>last_image</code>), or both.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ComponentsManager, ModularPipeline
<span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> load_image
<span class="hljs-keyword">from</span> diffusers.utils.export_utils <span class="hljs-keyword">import</span> encode_video
<span class="hljs-comment"># 61.7GB of transformer and 62.1GB of conditioner do not sit on one accelerator, so the components are</span>
<span class="hljs-comment"># registered in a manager that moves each one on and off as the blocks reach it. See [Memory](#memory).</span>
manager = ComponentsManager()
manager.enable_auto_cpu_offload(device=<span class="hljs-string">&quot;cuda&quot;</span>)
pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, components_manager=manager)
pipe.load_components(workflow=<span class="hljs-string">&quot;fl2va&quot;</span>, dtype=torch.bfloat16)
prompt = <span class="hljs-string">&quot;A red fox trotting through a snowy pine forest, snow crunching underfoot&quot;</span>
<span class="hljs-comment"># `output=` returns exactly the named outputs instead of the whole pipeline state.</span>
outputs = [<span class="hljs-string">&quot;videos&quot;</span>, <span class="hljs-string">&quot;audio&quot;</span>, <span class="hljs-string">&quot;sampling_rate&quot;</span>]
<span class="hljs-comment"># Text to video + audio.</span>
results = pipe(prompt=prompt, num_frames=<span class="hljs-number">124</span>, generator=torch.Generator().manual_seed(<span class="hljs-number">42</span>), output=outputs)
<span class="hljs-comment"># First frame (and optionally last frame) to video + audio. The canvas follows the first keyframe.</span>
image = load_image(
<span class="hljs-string">&quot;https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg&quot;</span>
)
results = pipe(prompt=prompt, image=image, num_frames=<span class="hljs-number">124</span>, generator=torch.Generator().manual_seed(<span class="hljs-number">42</span>), output=outputs)
encode_video(
results[<span class="hljs-string">&quot;videos&quot;</span>][<span class="hljs-number">0</span>],
fps=<span class="hljs-number">24</span>,
output_path=<span class="hljs-string">&quot;minimax_h3_fl2va.mp4&quot;</span>,
audio=results[<span class="hljs-string">&quot;audio&quot;</span>][<span class="hljs-number">0</span>],
audio_sample_rate=results[<span class="hljs-string">&quot;sampling_rate&quot;</span>],
)<!----></pre></div><!----> <p>Video and audio are generated jointly and come out of the call as separate outputs, <code>videos</code> and <code>audio</code>, next to the <code>sampling_rate</code> the soundtrack carries; muxing them into one file is left to the caller, e.g. with <a href="/docs/diffusers/pr_14407/en/api/utilities#diffusers.utils.encode_video">encode_video()</a>.</p> <!--[1--><h2 class="relative group"><a id="omni-references" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#omni-references"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>Omni-references</span></h2><!--]--><!----> <p>The <code>ref2va</code> workflow conditions on an ordered list of references: up to 9 images, 3 videos and 3 audio clips, 12 in total. The order is semantic. It labels the references in the prompt presentation (<code>"&lt;Picture 1>"</code>, <code>"&lt;Audio 1>"</code>, <code>"&lt;Video 1>"</code>) and it advances the shared audio/video rotary clock, so reordering the same references is a different request.</p> <p>Unlike the keyframes above, references do not bind the generated geometry: they are encoded at their own resolution and the target canvas defaults to MiniMax-H3’s 16:9.</p> <p>There is one reference class per modality, each holding in-memory media and the rate that media carries:</p> <table><thead><tr><th>reference</th><th>media</th><th>rate it declares</th></tr></thead><tbody><tr><td><a href="/docs/diffusers/pr_14407/en/api/pipelines/minimax_h3#diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference">MiniMaxH3ImageReference</a></td><td><code>image</code></td><td></td></tr><tr><td><a href="/docs/diffusers/pr_14407/en/api/pipelines/minimax_h3#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference">MiniMaxH3VideoReference</a></td><td><code>frames</code>, and <code>audio</code> for its own soundtrack</td><td><code>fps</code>, defaulting to MiniMax-H3’s 24.0, and <code>sample_rate</code></td></tr><tr><td><a href="/docs/diffusers/pr_14407/en/api/pipelines/minimax_h3#diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference">MiniMaxH3AudioReference</a></td><td><code>audio</code></td><td><code>sample_rate</code>, defaulting to the audio VAE’s own</td></tr></tbody></table> <p>The rates are what everything is resampled from: frames onto MiniMax-H3’s own 24 fps by dropping and duplicating whole frames, a waveform onto the audio VAE’s sample rate. Media at MiniMax-H3’s own rates flows through untouched, so only data produced at another rate has to say so.</p> <p>A reference is built one of two ways: decoded from a media file with the class’s <code>from_file</code> classmethod, or constructed directly from media the request already holds in memory — which is how a previous generation feeds back in (see <a href="#a-generation-as-a-reference">A generation as a reference</a>).</p> <p>The blocks never open a media file — decoding a path is the caller’s job, as it is everywhere else in the library. Each reference class does it through its <code>from_file</code> classmethod, which takes a path or a URL (video and audio through <a href="https://github.com/PyAV-Org/PyAV" rel="nofollow">PyAV</a>) and returns a reference carrying the rates the container reports — a video brings its frame rate and its soundtrack along. Prefer <code>from_file</code> over <a href="/docs/diffusers/pr_14407/en/api/utilities#diffusers.utils.load_video">load_video()</a>, which drops the frame rate: a reference built from frames whose real rate was lost is conditioned on at the wrong speed, and nothing raises.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ComponentsManager, ModularPipeline
<span class="hljs-keyword">from</span> diffusers.modular_pipelines.minimax_h3 <span class="hljs-keyword">import</span> (
MiniMaxH3AudioReference,
MiniMaxH3ImageReference,
MiniMaxH3VideoReference,
)
<span class="hljs-keyword">from</span> diffusers.utils <span class="hljs-keyword">import</span> load_video
<span class="hljs-keyword">from</span> diffusers.utils.export_utils <span class="hljs-keyword">import</span> encode_video
<span class="hljs-comment"># `ref2va` is a workflow of the one MiniMax-H3 pipeline; selecting it loads only the `transformer_ref/`</span>
<span class="hljs-comment"># checkpoint partition, and the manager moves each component on and off the accelerator in turn.</span>
manager = ComponentsManager()
manager.enable_auto_cpu_offload(device=<span class="hljs-string">&quot;cuda&quot;</span>)
pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, workflow=<span class="hljs-string">&quot;ref2va&quot;</span>, components_manager=manager)
pipe.load_components(dtype=torch.bfloat16)
subject = MiniMaxH3ImageReference.from_file(
<span class="hljs-string">&quot;https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg&quot;</span>
)
<span class="hljs-comment"># Decoding a file brings its rates along: the video its own frame rate and its soundtrack, the clip its sample rate.</span>
results = pipe(
prompt=<span class="hljs-string">&quot;The character speaks in time with the reference recording, natural lip movement&quot;</span>,
references=[
subject,
MiniMaxH3VideoReference.from_file(
<span class="hljs-string">&quot;https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/hiker.mp4&quot;</span>
),
MiniMaxH3AudioReference.from_file(<span class="hljs-string">&quot;voice.wav&quot;</span>),
],
num_frames=<span class="hljs-number">124</span>,
output=[<span class="hljs-string">&quot;videos&quot;</span>, <span class="hljs-string">&quot;audio&quot;</span>, <span class="hljs-string">&quot;sampling_rate&quot;</span>],
)
<span class="hljs-comment"># Frames a request already holds declare the rate they carry — `load_video` does not preserve it.</span>
motion = load_video(<span class="hljs-string">&quot;motion_ref.mp4&quot;</span>)
results = pipe(
prompt=<span class="hljs-string">&quot;The subject walks toward camera, matching the reference video&#x27;s shot rhythm&quot;</span>,
references=[subject, MiniMaxH3VideoReference(frames=motion, fps=<span class="hljs-number">30.0</span>)],
num_frames=<span class="hljs-number">124</span>,
output=[<span class="hljs-string">&quot;videos&quot;</span>, <span class="hljs-string">&quot;audio&quot;</span>, <span class="hljs-string">&quot;sampling_rate&quot;</span>],
)
encode_video(
results[<span class="hljs-string">&quot;videos&quot;</span>][<span class="hljs-number">0</span>],
fps=<span class="hljs-number">24</span>,
output_path=<span class="hljs-string">&quot;minimax_h3_ref2va.mp4&quot;</span>,
audio=results[<span class="hljs-string">&quot;audio&quot;</span>][<span class="hljs-number">0</span>],
audio_sample_rate=results[<span class="hljs-string">&quot;sampling_rate&quot;</span>],
)<!----></pre></div><!----> <p><code>num_frames</code> is required for <code>ref2va</code>. To generate a video exactly as long as a reference soundtrack, compute it from the clip — <code>round(samples / sample_rate * 24)</code> — and it is snapped up to the next <code>17 * n + 5</code> the video VAE can decode; the resulting duration must stay between the 5 and 15 seconds MiniMax-H3 generates.</p> <!--[2--><h3 class="relative group"><a id="a-generation-as-a-reference" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#a-generation-as-a-reference"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>A generation as a reference</span></h3><!--]--><!----> <p>Because the workflows share one pipeline, a <code>t2va</code> generation can be fed straight back as a <code>ref2va</code> reference — and the in-memory constructor is built for exactly this hand-off. The generated media is already at MiniMax-H3’s own rates (frames at 24 fps, the soundtrack at the audio VAE’s sample rate), so the reference needs no rate arguments and nothing is re-encoded through a lossy container on the way:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ComponentsManager, ModularPipeline
<span class="hljs-keyword">from</span> diffusers.modular_pipelines.minimax_h3 <span class="hljs-keyword">import</span> MiniMaxH3VideoReference
manager = ComponentsManager()
manager.enable_auto_cpu_offload(device=<span class="hljs-string">&quot;cuda&quot;</span>)
<span class="hljs-comment"># The full pipeline holds every workflow and picks one per call from the inputs. Loading without a</span>
<span class="hljs-comment"># `workflow=` brings both transformer partitions in one call, so the `ref2va` request that follows the</span>
<span class="hljs-comment"># `t2va` generation needs no further loading.</span>
pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, components_manager=manager)
pipe.load_components(dtype=torch.bfloat16)
results = pipe(
prompt=<span class="hljs-string">&quot;An astronaut hiking through the mountains, humming a tune&quot;</span>,
num_frames=<span class="hljs-number">124</span>,
output=[<span class="hljs-string">&quot;videos&quot;</span>, <span class="hljs-string">&quot;audio&quot;</span>, <span class="hljs-string">&quot;sampling_rate&quot;</span>],
)
reference = MiniMaxH3VideoReference(
frames=results[<span class="hljs-string">&quot;videos&quot;</span>][<span class="hljs-number">0</span>],
audio=results[<span class="hljs-string">&quot;audio&quot;</span>][<span class="hljs-number">0</span>],
sample_rate=results[<span class="hljs-string">&quot;sampling_rate&quot;</span>],
)
results = pipe(
prompt=<span class="hljs-string">&quot;The same astronaut now walks along a beach at sunset, humming the same tune&quot;</span>,
references=[reference],
num_frames=<span class="hljs-number">124</span>,
output=[<span class="hljs-string">&quot;videos&quot;</span>, <span class="hljs-string">&quot;audio&quot;</span>, <span class="hljs-string">&quot;sampling_rate&quot;</span>],
)<!----></pre></div><!----> <!--[1--><h2 class="relative group"><a id="diffusers.MiniMaxH3ModularPipeline" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.MiniMaxH3ModularPipeline"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MiniMaxH3ModularPipeline</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.MiniMaxH3ModularPipeline"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.</span><span class="font-semibold">MiniMaxH3ModularPipeline</span></span></h3><!----> <a id="diffusers.MiniMaxH3ModularPipeline" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.MiniMaxH3ModularPipeline"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/modular_pipeline.py#L149" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">blocks<span class="opacity-60">: diffusers.modular_pipelines.modular_pipeline.ModularPipelineBlocks | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">pretrained_model_name_or_path<span class="opacity-60">: str | os.PathLike | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">components_manager<span class="opacity-60">: diffusers.modular_pipelines.components_manager.ComponentsManager | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">collection<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">workflow<span class="opacity-60">: str | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">modular_config_dict<span class="opacity-60">: dict[str, typing.Any] | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">config_dict<span class="opacity-60">: dict[str, typing.Any] | None = None</span></span></span><span class="comma cursor-default"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">**kwargs<span class="opacity-60"></span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>A ModularPipeline for joint video + audio generation with MiniMax-H3: the <code>t2va</code> (text only) and <code>fl2va</code> (first
and/or last keyframe) workflows against the <code>transformer/</code> checkpoint partition, and the <code>ref2va</code> (omni-reference)
workflow against <code>transformer_ref/</code>. One repository holds both partitions, and selecting a workflow loads only its</p> <div class="relative group rounded-md"><a id="diffusers.MiniMaxH3ModularPipeline.example" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.MiniMaxH3ModularPipeline.example"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <p>own:</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!---->pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>, workflow=<span class="hljs-string">&quot;ref2va&quot;</span>)<!----></pre></div><!----><!----></div><!----> <p>MiniMax-H3 denoises <strong>one packed sequence</strong> that holds the text conditioning, the keyframe conditioning latents,
the audio latents and the video latents at once, which is why the blocks pass a row layout around rather than
per-modality tensors, and why the pipeline carries two schedulers (<code>shift = 12.0</code> for video, <code>shift = 3.0</code> for
audio) that are stepped inside a single transformer call.</p> <p>The checkpoint is guidance-distilled: guidance is baked into the weights, so there is no guider, no <code>negative_prompt</code> and no <code>guidance_scale</code>, and every step runs exactly one forward pass.</p> <p>MiniMax-H3 is modular only: this pipeline and its blocks are the whole integration, there is no <code>DiffusionPipeline</code> half. This module carries the model facts every block keys off — the config-derived geometry as properties, the
canvas and frame-count arithmetic as functions — and the conditioning, encoding and noise contracts live on the
blocks themselves.</p> <div class="relative group rounded-md"><a id="diffusers.MiniMaxH3ModularPipeline.example-2" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.MiniMaxH3ModularPipeline.example-2"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg><!----> <div class=" absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0 "><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent;"></div> Copied</div><!----></button><!----></div> <pre class="language-py "><!----><span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> diffusers <span class="hljs-keyword">import</span> ModularPipeline
pipe = ModularPipeline.from_pretrained(<span class="hljs-string">&quot;MiniMaxAI/MiniMax-H3&quot;</span>)
pipe.load_components(dtype=torch.bfloat16)<!----></pre></div><!----></div><!----> <blockquote class="warning"><p>> This is an experimental feature and is likely to change in the future.</p></blockquote></div> <!--[1--><h2 class="relative group"><a id="diffusers.MiniMaxH3Blocks" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.MiniMaxH3Blocks"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MiniMaxH3Blocks</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.MiniMaxH3Blocks"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.</span><span class="font-semibold">MiniMaxH3Blocks</span></span></h3><!----> <a id="diffusers.MiniMaxH3Blocks" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.MiniMaxH3Blocks"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/modular_blocks_minimax_h3.py#L659" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Auto Modular pipeline blocks for joint video + audio generation with MiniMax-H3: the <code>t2va</code> (text only), <code>fl2va</code> (first and/or last keyframe) and <code>ref2va</code> (omni-reference) workflows, selected on the <code>references</code> and keyframe
inputs. Without a <code>workflow=</code>, loading the components pulls <strong>both</strong> 61.7GB transformer partitions — pass one to
load only what the task needs.</p> <p>Supported workflows:</p> <ul><li><code>t2va</code>: requires <code>prompt</code></li> <li><code>fl2va</code>: requires <code>prompt</code>, <code>image</code> or <code>prompt</code>, <code>last_image</code></li> <li><code>ref2va</code>: requires <code>prompt</code>, <code>references</code></li></ul> <p>Components:
image_processor (<code>VaeImageProcessor</code>) text_encoder (<code>Qwen3VLForConditionalGeneration</code>) tokenizer
(<code>Qwen2Tokenizer</code>) processor (<code>Qwen3VLProcessor</code>) vae (<code>AutoencoderKLMiniMaxH3</code>) audio_vae
(<code>AutoencoderKLMiniMaxH3Audio</code>) scheduler (<code>MiniMaxH3Scheduler</code>) audio_scheduler (<code>MiniMaxH3Scheduler</code>)
transformer_ref (<code>MiniMaxH3Transformer3DModel</code>) transformer (<code>MiniMaxH3Transformer3DModel</code>) video_processor
(<code>VideoProcessor</code>)</p> <p>Configs:
canvas_short_edge (default: 768) canvas_max_pixels (default: 1032192) reference_image_short_edge (default:
2048)</p> <p>Inputs:
references (<code>list</code>, <em>optional</em>):
The references to condition on, <strong>in the order the model should read them</strong>: the order labels them in the
prompt presentation and lays them out on the shared rotary clock, so a different order is a different
request. One dataclass per modality, all holding in-memory media — a <code>MiniMaxH3ImageReference</code> (at most
9), a <code>MiniMaxH3VideoReference</code> at its own <code>fps</code> (at most 3, whose <code>audio</code> soundtrack is conditioned on
as well), or a <code>MiniMaxH3AudioReference</code> at its own <code>sample_rate</code> (at most 3) — for at most 12
references in total, and audio references cannot be the only ones. These blocks never open a media file:
decode with each class’s <code>from_file</code> classmethod, which brings the rates along.
height (<code>int</code>, <em>optional</em>):
Height of the generated video in pixels, a multiple of 32.
width (<code>int</code>, <em>optional</em>):
Width of the generated video in pixels, a multiple of 32.
num_frames (<code>int</code>, <em>optional</em>):
Number of frames to generate, at the fixed 24 fps. Snapped up to the next <code>17 * n + 5</code> the video VAE can
decode; the resulting duration must stay between 5 and 15 seconds. To generate a video as long as a
reference soundtrack, pass <code>round(samples / sample_rate * 24)</code>.
image (<code>Image</code>, <em>optional</em>):
Keyframe the video starts from. It is <em>stretched</em> onto the target canvas, which by default is derived
from its own aspect ratio.
last_image (<code>Image</code>, <em>optional</em>):
Keyframe the video ends on. Can be passed on its own to generate <em>up to</em> a frame. Combined with <code>image</code> it is the follower of the two and is cover-cropped onto the canvas.
prompt (<code>str</code>):
The prompt to guide generation, a single string.
normalized_references (<code>list</code>, <em>optional</em>):
The references normalized by the setup step, in packed order.
keyframes (<code>list</code>, <em>optional</em>):
The keyframes put onto the target canvas, in packed order.
condition_latents (<code>list</code>, <em>optional</em>):
The encoded video conditioning latents, one per image and video reference in packed order. Their shape is
where every reference block’s geometry comes from.
audio_condition_latents (<code>list</code>, <em>optional</em>):
The encoded audio conditioning rows, one per audio-bearing reference in packed order.
generator (<code>Generator</code>, <em>optional</em>):
The generator of the request. The conditioning noise is drawn from it first, one draw per condition,
before the noise of the generated rows.
latents (<code>Tensor</code>):
Pre-generated video noise of shape <code>(1, 24, num_latent_frames, latent_height, latent_width)</code>, used
instead of the draw.
audio_latents (<code>Tensor</code>):
Pre-generated audio noise of shape <code>(2, 32, num_audio_latents)</code>.
num_inference_steps (<code>int</code>):
The number of denoising steps.
*<em>denoiser_input_fields (<code>None</code>,</em>optional<em>):
The structural description of the packed sequence the transformer reads by name: <code>token_tags</code>, <code>position_ids</code> and the three row-index tensors.
attention_kwargs (<code>dict</code>,</em>optional<em>):
Additional kwargs for attention processors.
keyframe_anchors (<code>tuple</code>,</em>optional<em>, defaults to ()):
Which end of the video every keyframe is anchored to, in packed order.
output_type (<code>str</code>,</em>optional*, defaults to pil):
Output format: ‘pil’, ‘np’ or ‘pt’.</p> <p>Outputs:
videos (<code>list</code>):
The generated video.
audio (<code>Tensor</code>):
The generated soundtrack, of shape <code>(1, 2, num_samples)</code>.
sampling_rate (<code>int</code>):
Sample rate of the generated soundtrack in Hz.</p></div> <!--[1--><h2 class="relative group"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MiniMaxH3ImageReference</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.modular_pipelines.minimax_h3.</span><span class="font-semibold">MiniMaxH3ImageReference</span></span></h3><!----> <a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/references.py#L82" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">image<span class="opacity-60">: typing.Union[PIL.Image.Image, numpy.ndarray, torch.Tensor]</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference.image" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference.image"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>image</strong> (<code>PIL.Image.Image</code>, <code>np.ndarray</code> or <code>torch.Tensor</code>) &#x2014;
The reference image: an image, a <code>(height, width, 3)</code> array or a <code>(3, height, width)</code> tensor, <code>uint8</code> or
floating point over <code>[0, 1]</code>. It never binds the generated geometry &#x2014; it is encoded at a short edge of its
own, 2048 for the released checkpoint, whatever canvas the request generates at.<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>A subject, style or scene reference: at most 9 per request.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference.from_file"><!----><h4 class="!m-0"><span class="flex-1 rounded-xl py-0.5 break-all bg-gradient-to-r from-blue-50/60 to-white dark:from-gray-900 dark:to-gray-950 text-blue-700 dark:text-blue-300 font-medium px-2"><svg width="1em" height="1em" viewBox="0 0 32 33" class="mr-1 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg"><path d="M5.80566 18.3545C4.90766 17.4565 4.90766 16.0005 5.80566 15.1025L14.3768 6.53142C15.2748 5.63342 16.7307 5.63342 17.6287 6.53142L26.1999 15.1025C27.0979 16.0005 27.0979 17.4565 26.1999 18.3545L17.6287 26.9256C16.7307 27.8236 15.2748 27.8236 14.3768 26.9256L5.80566 18.3545Z" fill="currentColor" fill-opacity="0.25"/><path fill-rule="evenodd" clip-rule="evenodd" d="M16.4801 13.9619C16.4801 12.9761 16.7467 12.5436 16.9443 12.3296C17.1764 12.078 17.5731 11.8517 18.2275 11.707C18.8821 11.5623 19.638 11.5342 20.4038 11.5582C20.7804 11.57 21.1341 11.5932 21.4719 11.6156L21.5263 11.6193C21.8195 11.6389 22.1626 11.6618 22.4429 11.6618V7.40825C22.3209 7.40825 22.1219 7.39596 21.7544 7.37149C21.4202 7.34925 20.9976 7.32115 20.5371 7.30672C19.6286 7.27824 18.4672 7.29779 17.3093 7.55377C16.1512 7.8098 14.8404 8.33724 13.8181 9.4452C12.7612 10.5907 12.2266 12.1236 12.2266 13.9619V15.0127H10.6836V19.2662H12.2266V26.6332H16.4801V19.2662H20.3394V15.0127H16.4801V13.9619Z" fill="currentColor"/></svg>from_file</span></h4><!----> <a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference.from_file" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference.from_file"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/references.py#L98" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">media<span class="opacity-60"></span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference.from_file.media" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference.from_file.media"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>media</strong> (<code>str</code> or <code>os.PathLike</code>) &#x2014; Path or URL of the image.<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Load an image file into a <code>MiniMaxH3ImageReference</code>, through <a href="/docs/diffusers/pr_14407/en/api/utilities#diffusers.utils.load_image">load_image()</a>.</p></div></div> <!--[1--><h2 class="relative group"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MiniMaxH3VideoReference</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.modular_pipelines.minimax_h3.</span><span class="font-semibold">MiniMaxH3VideoReference</span></span></h3><!----> <a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/references.py#L110" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">frames<span class="opacity-60">: typing.Union[list[PIL.Image.Image], numpy.ndarray, torch.Tensor]</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">fps<span class="opacity-60">: float | None = None</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">audio<span class="opacity-60">: typing.Optional[torch.Tensor] = None</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">sample_rate<span class="opacity-60">: int | None = None</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.frames" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.frames"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>frames</strong> (<code>list[PIL.Image.Image]</code>, <code>np.ndarray</code> or <code>torch.Tensor</code>) &#x2014;
The reference frames: a list of images, a <code>(num_frames, height, width, 3)</code> array or a <code>(num_frames, 3, height, width)</code> tensor, <code>uint8</code> or floating point over <code>[0, 1]</code>.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.fps" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.fps"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>fps</strong> (<code>float</code>, <em>optional</em>, defaults to 24.0) &#x2014;
The frame rate <code>frames</code> carries, which is what places the reference&#x2019;s vision blocks on the conditioner&#x2019;s 2
fps grid. MiniMax-H3&#x2019;s own clock is 24 fps, so any other rate is resampled onto it by dropping and
duplicating whole frames &#x2014; which makes this the field to get right when the frames came from a file.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.audio" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.audio"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>audio</strong> (<code>torch.Tensor</code> of shape <code>(channels, num_samples)</code>, <em>optional</em>) &#x2014;
This video&#x2019;s soundtrack, mono or stereo, conditioned on as the reference&#x2019;s own rather than as a reference
of its own. Left out, the reference conditions on motion alone.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.sample_rate" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.sample_rate"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>sample_rate</strong> (<code>int</code>, <em>optional</em>) &#x2014;
The rate <code>audio</code> carries its samples at. Left out, it is the audio VAE&#x2019;s own, which leaves the samples
untouched; any other rate is resampled onto it.<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>A motion and camera reference: at most 3 per request, conditioned on together with its own soundtrack.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.from_file"><!----><h4 class="!m-0"><span class="flex-1 rounded-xl py-0.5 break-all bg-gradient-to-r from-blue-50/60 to-white dark:from-gray-900 dark:to-gray-950 text-blue-700 dark:text-blue-300 font-medium px-2"><svg width="1em" height="1em" viewBox="0 0 32 33" class="mr-1 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg"><path d="M5.80566 18.3545C4.90766 17.4565 4.90766 16.0005 5.80566 15.1025L14.3768 6.53142C15.2748 5.63342 16.7307 5.63342 17.6287 6.53142L26.1999 15.1025C27.0979 16.0005 27.0979 17.4565 26.1999 18.3545L17.6287 26.9256C16.7307 27.8236 15.2748 27.8236 14.3768 26.9256L5.80566 18.3545Z" fill="currentColor" fill-opacity="0.25"/><path fill-rule="evenodd" clip-rule="evenodd" d="M16.4801 13.9619C16.4801 12.9761 16.7467 12.5436 16.9443 12.3296C17.1764 12.078 17.5731 11.8517 18.2275 11.707C18.8821 11.5623 19.638 11.5342 20.4038 11.5582C20.7804 11.57 21.1341 11.5932 21.4719 11.6156L21.5263 11.6193C21.8195 11.6389 22.1626 11.6618 22.4429 11.6618V7.40825C22.3209 7.40825 22.1219 7.39596 21.7544 7.37149C21.4202 7.34925 20.9976 7.32115 20.5371 7.30672C19.6286 7.27824 18.4672 7.29779 17.3093 7.55377C16.1512 7.8098 14.8404 8.33724 13.8181 9.4452C12.7612 10.5907 12.2266 12.1236 12.2266 13.9619V15.0127H10.6836V19.2662H12.2266V26.6332H16.4801V19.2662H20.3394V15.0127H16.4801V13.9619Z" fill="currentColor"/></svg>from_file</span></h4><!----> <a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.from_file" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.from_file"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/references.py#L146" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">media<span class="opacity-60"></span></span></span><!--]--> <span>)</span> <!--[0--><span class="font-bold"></span> <span class="rounded hover:bg-gray-400 cursor-pointer"><!----><script context="module">export const metadata = 'undefined';</script><span><code>MiniMaxH3VideoReference</code></span><!----></span><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.from_file.media" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.from_file.media"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>media</strong> (<code>str</code> or <code>os.PathLike</code>) &#x2014; Path or URL of the video.<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[0--><div id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference.from_file.returns" class="flex items-center font-semibold space-x-3 text-base !mt-0 !mb-0 text-gray-800 rounded "><p class="text-base">Returns</p> <!--[0--><!----><script context="module">export const metadata = 'undefined';</script>
<p><code>MiniMaxH3VideoReference</code></p>
<!----><!--]--> <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700"></span></div> <p class="text-base"><!----><script context="module">export const metadata = 'undefined';</script>
<p>the <code>(num_frames, height, width, 3)</code> <code>uint8</code> frames at the frame rate the
container reports, carrying its soundtrack and that soundtrack’s own sample rate when it has an audio
stream.</p>
<!----></p><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Decode a video file into a <code>MiniMaxH3VideoReference</code>, at the resolution, the frame rate and the soundtrack it
carries.</p> <p>The rates land on the reference, which is the point of decoding this way rather than with <a href="/docs/diffusers/pr_14407/en/api/utilities#diffusers.utils.load_video">load_video()</a>: MiniMax-H3 resamples a reference onto its own 24 fps, so a frame rate lost on the way in
is a request conditioned at the wrong speed, with nothing to raise about it. A container whose metadata is
wrong is corrected by overriding <code>fps</code> or <code>sample_rate</code> on the returned reference.</p> <p>Needs <a href="https://github.com/PyAV-Org/PyAV" rel="nofollow">PyAV</a>.</p></div></div> <!--[1--><h2 class="relative group"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference"><span><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span>MiniMaxH3AudioReference</span></h2><!--]--><!----> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference"><!----><h3 class="!m-0"><span class="flex-1 break-all md:text-lg bg-gradient-to-r px-2.5 py-1.5 rounded-xl from-indigo-50/70 to-white dark:from-gray-900 dark:to-gray-950 dark:text-indigo-300 text-indigo-700"><svg class="mr-1.5 text-indigo-500 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" focusable="false" role="img" width=".8em" height=".8em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 24 24"><path class="uim-quaternary" d="M20.23 7.24L12 12L3.77 7.24a1.98 1.98 0 0 1 .7-.71L11 2.76c.62-.35 1.38-.35 2 0l6.53 3.77c.29.173.531.418.7.71z" opacity=".25" fill="currentColor"></path><path class="uim-tertiary" d="M12 12v9.5a2.09 2.09 0 0 1-.91-.21L4.5 17.48a2.003 2.003 0 0 1-1-1.73v-7.5a2.06 2.06 0 0 1 .27-1.01L12 12z" opacity=".5" fill="currentColor"></path><path class="uim-primary" d="M20.5 8.25v7.5a2.003 2.003 0 0 1-1 1.73l-6.62 3.82c-.275.13-.576.198-.88.2V12l8.23-4.76c.175.308.268.656.27 1.01z" fill="currentColor"></path></svg><span class="font-light">class</span> <span class="font-medium">diffusers.modular_pipelines.minimax_h3.</span><span class="font-semibold">MiniMaxH3AudioReference</span></span></h3><!----> <a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/references.py#L172" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">audio<span class="opacity-60">: Tensor</span></span></span><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">sample_rate<span class="opacity-60">: int | None = None</span></span></span><!--]--> <span>)</span> <!--[-1--><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.audio" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.audio"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>audio</strong> (<code>torch.Tensor</code> of shape <code>(channels, num_samples)</code>) &#x2014;
The reference waveform, mono or stereo.<!----></span></span></li><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.sample_rate" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.sample_rate"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>sample_rate</strong> (<code>int</code>, <em>optional</em>) &#x2014;
The rate <code>audio</code> carries its samples at. Left out, it is the audio VAE&#x2019;s own, which leaves the samples
untouched; any other rate is resampled onto it.<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--> <!--[-1--><!--]--></div></div><!----> <p>A voice or music reference: at most 3 per request, and never on its own — an audio reference has to be paired with
at least one image or video reference. It never reaches the conditioner and is encoded by the audio VAE alone.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><div><span class="group flex space-x-1.5 items-center text-gray-800 bg-gradient-to-r rounded-tr-lg -mt-4 -ml-4 pt-3 px-2.5" id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.from_file"><!----><h4 class="!m-0"><span class="flex-1 rounded-xl py-0.5 break-all bg-gradient-to-r from-blue-50/60 to-white dark:from-gray-900 dark:to-gray-950 text-blue-700 dark:text-blue-300 font-medium px-2"><svg width="1em" height="1em" viewBox="0 0 32 33" class="mr-1 inline-block -mt-0.5" xmlns="http://www.w3.org/2000/svg"><path d="M5.80566 18.3545C4.90766 17.4565 4.90766 16.0005 5.80566 15.1025L14.3768 6.53142C15.2748 5.63342 16.7307 5.63342 17.6287 6.53142L26.1999 15.1025C27.0979 16.0005 27.0979 17.4565 26.1999 18.3545L17.6287 26.9256C16.7307 27.8236 15.2748 27.8236 14.3768 26.9256L5.80566 18.3545Z" fill="currentColor" fill-opacity="0.25"/><path fill-rule="evenodd" clip-rule="evenodd" d="M16.4801 13.9619C16.4801 12.9761 16.7467 12.5436 16.9443 12.3296C17.1764 12.078 17.5731 11.8517 18.2275 11.707C18.8821 11.5623 19.638 11.5342 20.4038 11.5582C20.7804 11.57 21.1341 11.5932 21.4719 11.6156L21.5263 11.6193C21.8195 11.6389 22.1626 11.6618 22.4429 11.6618V7.40825C22.3209 7.40825 22.1219 7.39596 21.7544 7.37149C21.4202 7.34925 20.9976 7.32115 20.5371 7.30672C19.6286 7.27824 18.4672 7.29779 17.3093 7.55377C16.1512 7.8098 14.8404 8.33724 13.8181 9.4452C12.7612 10.5907 12.2266 12.1236 12.2266 13.9619V15.0127H10.6836V19.2662H12.2266V26.6332H16.4801V19.2662H20.3394V15.0127H16.4801V13.9619Z" fill="currentColor"/></svg>from_file</span></h4><!----> <a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.from_file" class="header-link invisible with-hover:group-hover:visible pr-2" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.from_file"><svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></a> <!--[0--><a class="!ml-auto !text-gray-400 !no-underline text-sm flex items-center" href="https://github.com/huggingface/diffusers/blob/vr_14407/src/diffusers/modular_pipelines/minimax_h3/references.py#L191" target="_blank"><span>&lt;</span> <span class="hidden md:block mx-0.5 hover:!underline">source</span> <span>></span></a><!--]--></span> <!--[0--><p class="font-mono text-xs md:text-sm !leading-relaxed !my-6"><span>(</span> <!--[--><span class="comma cursor-pointer"><span class="rounded hover:bg-black hover:text-white dark:hover:bg-white dark:hover:text-black">media<span class="opacity-60"></span></span></span><!--]--> <span>)</span> <!--[0--><span class="font-bold"></span> <span class="rounded hover:bg-gray-400 cursor-pointer"><!----><script context="module">export const metadata = 'undefined';</script><span><code>MiniMaxH3AudioReference</code></span><!----></span><!--]--></p><!--]--> <div class="!mb-10 relative docstring-details "><!--[-1--><!--]--> <!--[0--><p class="flex items-center font-semibold !mt-2 !mb-2 text-gray-800">Parameters <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700 ml-3"></span></p> <ul class="px-2"><!--[--><li class="text-base !pl-4 my-3 rounded "><span class="group flex space-x-1.5 items-start"><a id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.from_file.media" class="header-link block pr-0.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.from_file.media"><span><svg class="text-smd" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg><!----></span></a> <span><!----><strong>media</strong> (<code>str</code> or <code>os.PathLike</code>) &#x2014; Path or URL of the audio, or of a video whose soundtrack is taken.<!----></span></span></li><!--]--></ul><!--]--> <!--[-1--><!--]--> <!--[0--><div id="diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference.from_file.returns" class="flex items-center font-semibold space-x-3 text-base !mt-0 !mb-0 text-gray-800 rounded "><p class="text-base">Returns</p> <!--[0--><!----><script context="module">export const metadata = 'undefined';</script>
<p><code>MiniMaxH3AudioReference</code></p>
<!----><!--]--> <span class="flex-auto border-t-2 border-gray-100 dark:border-gray-700"></span></div> <p class="text-base"><!----><script context="module">export const metadata = 'undefined';</script>
<p>the <code>(channels, num_samples)</code> float32 waveform at the sample rate the
container reports.</p>
<!----></p><!--]--> <!--[-1--><!--]--></div></div><!----> <p>Decode an audio file into a <code>MiniMaxH3AudioReference</code>, at the sample rate it carries.</p> <p>Needs <a href="https://github.com/PyAV-Org/PyAV" rel="nofollow">PyAV</a>.</p></div></div> <a class="!text-gray-400 !no-underline text-sm flex items-center not-prose mt-4" href="https://github.com/huggingface/diffusers/blob/main/docs/source/en/api/pipelines/minimax_h3.md" target="_blank"><svg class="mr-1" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M31,16l-7,7l-1.41-1.41L28.17,16l-5.58-5.59L24,9l7,7z"></path><path d="M1,16l7-7l1.41,1.41L3.83,16l5.58,5.59L8,23l-7-7z"></path><path d="M12.419,25.484L17.639,6.552l1.932,0.518L14.351,26.002z"></path></svg><!----> <span><span class="underline">Update</span> on GitHub</span></a><!----> <p></p><!--]--><!----><!--]--><!--]--><!--]--> <!--[-1--><!--]--><!--]-->
<script>
{
__sveltekit_ipruxm = {
base: "/docs/diffusers/pr_14407/en",
assets: "/docs/diffusers/pr_14407/en"
};
const element = document.currentScript.parentElement;
Promise.all([
import("/docs/diffusers/pr_14407/en/_app/immutable/entry/start.Ic04TDLP.js"),
import("/docs/diffusers/pr_14407/en/_app/immutable/entry/app.DrFCO7E8.js")
]).then(([kit, app]) => {
kit.start(app, element, {
node_ids: [0, 190],
data: [null,null],
form: null,
error: null
});
});
}
</script>

Xet Storage Details

Size:
122 kB
·
Xet hash:
245aa30c6c1c68c1869fbd5525d1c6188d5dcd2b8eb5336b8838b92d352a5f26

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.