| """The denoising half of a **split** MiniMax-H3 deployment. |
| |
| MiniMax-H3 is ~196 GiB in bfloat16 and a ZeroGPU Space is evicted at 150 GB of storage, so `MiniMaxH3Blocks` is cut at |
| its `text_encoder` step: the 62.14 GiB Qwen3-VL conditioner runs in its own Space and everything else β the distilled |
| transformer and the two autoencoders β runs here. `prompt_embeds` + `text_token_tags` is the whole wire format between |
| the two halves. |
| |
| For a text-only (`t2va`) request the full `MiniMaxH3Blocks` sequence is `before_encode -> text_encoder -> vae_encoder |
| -> denoise -> decode`, and the two `Auto*` encoder steps select nothing without a keyframe or a reference. What is |
| left once the text encoder goes is exactly `MiniMaxH3CoreDenoiseStep -> MiniMaxH3DecodeStep`, and those two declare |
| only `transformer`, the two schedulers, both autoencoders and `video_processor` β so `load_components` resolves those |
| out of the checkpoint's `modular_model_index.json` and never fetches the conditioner or the `ref2va` partition. |
| |
| Adapted from `multimodalart/minimax-h3`, trimmed to the `t2va` (text -> video + audio) branch this Space serves. |
| """ |
|
|
| import torch |
| from diffusers.modular_pipelines.minimax_h3.modular_blocks_minimax_h3 import ( |
| MiniMaxH3CoreDenoiseStep, |
| MiniMaxH3DecodeStep, |
| ) |
| from diffusers.modular_pipelines.modular_pipeline import SequentialPipelineBlocks |
| from diffusers.modular_pipelines.modular_pipeline_utils import OutputParam |
|
|
|
|
| class MiniMaxH3GeneratorBlocks(SequentialPipelineBlocks): |
| """The denoising half of a split MiniMax-H3: the `t2va` branch of `MiniMaxH3Blocks` without its text encoder.""" |
|
|
| model_name = "minimax-h3" |
| block_classes = [MiniMaxH3CoreDenoiseStep, MiniMaxH3DecodeStep] |
| block_names = ["denoise", "decode"] |
|
|
| @property |
| def description(self): |
| return ( |
| "The denoising half of a split MiniMax-H3 deployment: the `t2va` branch of `MiniMaxH3Blocks` without its " |
| "text-encoder step, so `prompt_embeds` and `text_token_tags` come in as inputs and the 62.14 GiB Qwen3-VL " |
| "conditioner is never loaded here." |
| ) |
|
|
| @property |
| def outputs(self): |
| return [ |
| OutputParam.template("videos", description="The generated video."), |
| OutputParam( |
| "audio", |
| type_hint=torch.Tensor, |
| description="The generated soundtrack, of shape `(1, 2, num_samples)`.", |
| ), |
| OutputParam("sampling_rate", type_hint=int, description="Sample rate of the generated soundtrack in Hz."), |
| ] |
|
|