File size: 15,812 Bytes
92c8dff ae15fda 5d00781 b087f55 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 6963487 92c8dff 3b93b22 92c8dff 315be37 92c8dff 88575f3 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 92c8dff 6963487 92c8dff 3b93b22 211f492 3b93b22 92c8dff 3b93b22 92c8dff 3b93b22 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 | ---
license: apache-2.0
base_model:
- MiniMaxAI/MiniMax-H3
frameworks:
- ""
base_model_relation: quantized
---
# MiniMax-H3-NF4
This model is the **NF4 quantized version** of the video generation model [MiniMax-H3](https://modelscope.cn/models/MiniMax/MiniMax-H3). It utilizes the `bitsandbytes` 4-bit quantization scheme and is designed to be used with [DiffSynth-Studio](https://github.com/modelscope/DiffSynth-Studio), enabling model inference on devices with limited VRAM and RAM.
## Environment Setup
```shell
git clone https://github.com/modelscope/DiffSynth-Studio.git
cd DiffSynth-Studio
pip install -e ".[all]"
```
## Inference Code
### Enable VRAM Management
Run the following code to perform inference using DiffSynth-Studio. VRAM management will be automatically enabled. The actual VRAM usage depends on the available VRAM on your GPU; a minimum of 8GB VRAM is required to run.
#### FL2VA (Text-to-Video/Audio):
```python
import torch
from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
from diffsynth.utils.data.audio_video import write_video_audio
from modelscope import dataset_snapshot_download
from PIL import Image
vram_config = {
"offload_dtype": "disk",
"offload_device": "disk",
"onload_dtype": torch.bfloat16,
"onload_device": "cpu",
"preparing_dtype": torch.bfloat16,
"preparing_device": "cuda",
"computation_dtype": torch.bfloat16,
"computation_device": "cuda",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
device="cuda",
model_configs=[
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-fl2va-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
],
processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="FL2VA/processor/"),
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 4,
)
prompt = "A girl is very happy, she is speaking in english: “I enjoy working with Diffsynth-Studio, it's a perfect framework.”"
video, audio = pipe(
prompt=prompt,
height=480, width=832, num_frames=124, num_inference_steps=50, seed=0,
)
write_video_audio(
video=video, audio=audio,
output_path="t2va.mp4", fps=24, audio_sample_rate=32000,
)
```
#### Ref2VA (Reference-to-Video/Audio):
<details>
<summary>Expand Code</summary>
```python
import torch
from PIL import Image
from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
from diffsynth.utils.data.audio_video import write_video_audio
from diffsynth.utils.data.audio import read_audio
from diffsynth.utils.data import VideoData
from modelscope import dataset_snapshot_download
def align_frame_count(frame_count):
current = max(int(frame_count), 1)
while current % 17 != 5:
current += 1
return current
def read_video_with_fps(path, num_out_frames, height, width, fps=24):
video = VideoData(path, height=height, width=width)
frames = video.raw_data()
src_fps = float(video.data.reader.get_meta_data()["fps"])
out = []
for k in range(num_out_frames):
idx = int(round(k * src_fps / fps))
if idx >= len(frames):
break
out.append(frames[idx])
return out
vram_config = {
"offload_dtype": "disk",
"offload_device": "disk",
"onload_dtype": torch.bfloat16,
"onload_device": "cpu",
"preparing_dtype": torch.bfloat16,
"preparing_device": "cuda",
"computation_dtype": torch.bfloat16,
"computation_device": "cuda",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
device="cuda",
model_configs=[
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-ref2va-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
],
processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="Ref2VA/processor/"),
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5,
)
# Text + Reference Image -> Video + Audio
dataset_snapshot_download(dataset_id="DiffSynth-Studio/diffsynth_example_dataset", local_dir="data/diffsynth_example_dataset", allow_file_pattern="minimax_h3/MiniMax-H3-Ref2VA/*")
ref_image = Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")
prompt = "A website page, website UI design, website animation, video showing smooth webpage scrolling effect. A highly explosive and dynamic product official website style product landing page UI/UX demo video, the core display subject is product image 1. The page uses bold, powerful, tilted oversized sans-serif fonts for flamboyant typography. The background features dynamic light and shadow with extreme speed sense, dark carbon fiber or sports breathable mesh textures interweaving and changing. The video shows a tight-paced, powerful webpage downward scrolling effect, as well as strong visual zoom and color inversion UI interaction actions when hovering the mouse."
video, audio = pipe(
prompt=prompt,
height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
references=[{"type": "image", "image": Image.open("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/0.png").convert("RGB")}]
)
write_video_audio(
video=video, audio=audio,
output_path="ti2va.mp4", fps=24, audio_sample_rate=32000,
)
# Text + Reference Audio + Reference Video -> Video + Audio
ref_video = read_video_with_fps("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/video.mp4", 124, 480, 832)
ref_audio, sample_rate = read_audio("data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-Ref2VA/voice.mp3", duration=len(ref_video) / 24, resample=True, resample_rate=pipe.audio_vae.sample_rate)
prompt = "subject_definitions:\n<Subject 1> is the young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, an unbuttoned white shirt, and silver rings, holding a small black lamb in his arms in <Video 1>.\n<Video 1> is the source video for the editing task.\n<Audio 1> is the synchronized audio track of <Video 1>, providing the background music.\n<Audio 2> is the voice timbre reference for <Subject 1>'s voice, containing a spoken male voiceover.\n\nsummary:\n[video editing + audio reference + audio reuse] The target video is an edited version of <Video 1>. <Subject 1>, wearing a bright pink suit and holding a black lamb, stands in a grassy field with other white lambs in the background. The edit animates <Subject 1>'s face to speak the user-provided dialogue. <Audio 1> is partially reused as the continuous background music, while the target references the calm male voice timbre of <Audio 2> for <Subject 1>'s spoken lines.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1]): fully_preserved - the man retains his identity, wavy blonde hair, pink suit, white shirt, accessories, and the black lamb he holds, with his mouth newly animated to speak.\n<Video 1> (source video editing): fully_preserved - the original camera framing, warm golden hour lighting, grassy hill setting, and background white lambs are maintained while the central character is edited.\n<Audio 1>: partially_copy - the atmospheric background music from <Audio 1> is reused in the target video, mixed beneath the newly added spoken dialogue.\n<Audio 2>: reference - the target audio references the male voice timbre from <Audio 2> to generate <Subject 1>'s spoken dialogue.\n\ndetailed_description:\nThe target video is in realistic photographic style.\n[Shot 1] The shot begins from the source <Video 1>, showing <Subject 1>, a young man with short wavy blonde hair, wearing a bright pink suit jacket, matching pink trousers, and a casually unbuttoned white shirt. He stands confidently in a sunlit green pasture, gently holding a small black lamb securely in his arms. The warm, golden hour lighting casts soft shadows across his face and the bright pink fabric of his suit. Behind him, several white lambs stand and graze on the rolling grassy hill against a clear, pale blue sky. The atmospheric background music from <Audio 1> plays continuously throughout the scene. <Subject 1> physically speaks, his mouth movements naturally syncing to the new dialogue, with his voice timbre referencing the calm male delivery from <Audio 2>. Looking thoughtfully forward, <Subject 1> (S1) speaks softly, <d>[English] Follow the wind, live free.</d> As he delivers the line, he subtly shifts his weight, cradling the resting black lamb while the camera slowly pushes in. <Subject 1> (S1) continues his thought, <d>[English] Leave worries behind, enjoy the moment.</d> Exactly as his voice stops, his lips meet in a relaxed, peaceful smile, and his jaw ceases speaking motion. He then turns his gaze slightly away toward the horizon, gently stroking the black lamb's fleece with his fingers as the camera holds on this tranquil, sunlit state through the end of the video.\n\noverall_soundscape:\nThe soundscape consists of the continuous, atmospheric background music from <Audio 1>, overlaid with the clear, calm male dialogue spoken by the main character, referencing the voice timbre of <Audio 2>.\n\nnon_diegetic_music:\nThe atmospheric, sustained background music from <Audio 1> is reused as the continuous score, playing quietly beneath the spoken dialogue."
video, audio = pipe(
prompt=prompt,
height=480, width=832, num_frames=124, num_inference_steps=50, seed=42,
references=[
{"type": "video", "video": ref_video},
{"type": "audio", "audio": ref_audio, "sample_rate": sample_rate},
],
)
write_video_audio(
video=video, audio=audio,
output_path="tav2va.mp4", fps=24, audio_sample_rate=32000,
)
```
</details>
### Extreme Hardware Optimization
If your computing device has extremely limited performance, we support enabling direct disk-to-VRAM loading. With this configuration, tensors in the model are loaded from disk to VRAM one by one according to the computation order. This allows the model to run with only 8GB of RAM:
```diff
vram_config = {
+ "offload_dtype": "disk",
+ "offload_device": "disk",
+ "onload_dtype": "disk",
+ "onload_device": "disk",
+ "preparing_dtype": "disk",
+ "preparing_device": "disk",
+ "computation_dtype": torch.bfloat16,
+ "computation_device": "cuda",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
device="cuda",
model_configs=...,
processor_config=...,
+ vram_limit=0,
)
```
We also support running model inference on Mac M-series chips, although this is not recommended:
```diff
vram_config = {
+ "offload_dtype": "disk",
+ "offload_device": "disk",
+ "onload_dtype": "disk",
+ "onload_device": "disk",
+ "preparing_dtype": "disk",
+ "preparing_device": "disk",
+ "computation_dtype": torch.bfloat16,
+ "computation_device": "mps",
}
pipe = MiniMaxH3Pipeline.from_pretrained(
torch_dtype=torch.bfloat16,
+ device="mps",
model_configs=...,
processor_config=...,
+ vram_limit=0,
)
```
## Training Code
This quantized model supports LoRA training. Please follow the steps below to start the training program.
Download the sample dataset:
```shell
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "minimax_h3/MiniMax-H3-FL2VA/*" --local_dir ./data/diffsynth_example_dataset
```
**Training configuration suitable for Data Center GPUs (e.g., Nvidia H20):** Run the following script to start the LoRA training program. Requires 48GB VRAM.
```shell
accelerate launch examples/minimax_h3/model_training/train.py \
--dataset_base_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA \
--dataset_metadata_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/metadata.csv \
--data_file_keys "video,input_audio" \
--extra_inputs "input_audio" \
--height 480 \
--width 832 \
--num_frames 124 \
--dataset_repeat 100 \
--model_id_with_origin_paths "DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-text-encoder-nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-fl2va-nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:video_vae_nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:audio_vae_nf4.safetensors" \
--learning_rate 1e-4 \
--num_epochs 5 \
--remove_prefix_in_ckpt "pipe.dit." \
--output_path "./models/train/MiniMax-H3-T2VA-nf4" \
--lora_base_model "dit" \
--lora_target_modules "qkv_proj,out_proj" \
--lora_rank 32 \
--use_gradient_checkpointing \
--find_unused_parameters
```
**Training configuration suitable for Consumer GPUs (e.g., Nvidia RTX 4090):** Run the following scripts to start two-stage split training with gradient checkpointing offload. Requires 24GB VRAM.
```shell
accelerate launch examples/minimax_h3/model_training/train.py \
--dataset_base_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA \
--dataset_metadata_path data/diffsynth_example_dataset/minimax_h3/MiniMax-H3-FL2VA/metadata.csv \
--data_file_keys "video,input_audio" \
--extra_inputs "input_audio" \
--height 480 \
--width 832 \
--num_frames 124 \
--dataset_repeat 1 \
--model_id_with_origin_paths "DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-text-encoder-nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:video_vae_nf4.safetensors,DiffSynth-Studio/MiniMax-H3-NF4:audio_vae_nf4.safetensors" \
--learning_rate 1e-4 \
--num_epochs 1 \
--remove_prefix_in_ckpt "pipe.dit." \
--output_path "./models/train/MiniMax-H3-T2VA-nf4-split-cache" \
--lora_base_model "dit" \
--lora_target_modules "qkv_proj,out_proj" \
--lora_rank 32 \
--use_gradient_checkpointing \
--use_gradient_checkpointing_offload \
--task "sft:data_process"
accelerate launch examples/minimax_h3/model_training/train.py \
--dataset_base_path "./models/train/MiniMax-H3-T2VA-nf4-split-cache" \
--data_file_keys "video,input_audio" \
--extra_inputs "input_audio" \
--height 480 \
--width 832 \
--num_frames 124 \
--dataset_repeat 100 \
--model_id_with_origin_paths "DiffSynth-Studio/MiniMax-H3-NF4:minimax-h3-fl2va-nf4.safetensors" \
--learning_rate 1e-4 \
--num_epochs 5 \
--remove_prefix_in_ckpt "pipe.dit." \
--output_path "./models/train/MiniMax-H3-T2VA-nf4" \
--lora_base_model "dit" \
--lora_target_modules "qkv_proj,out_proj" \
--lora_rank 32 \
--use_gradient_checkpointing \
--use_gradient_checkpointing_offload \
--find_unused_parameters \
--task "sft:train"
```
## References
* DiffSynth-Studio Documentation: [Minimax-H3](https://diffsynth-studio-doc.readthedocs.io/en/latest/Model_Details/MiniMax-H3.html)
* DiffSynth-Studio Documentation: [VRAM Management](https://diffsynth-studio-doc.readthedocs.io/en/latest/Pipeline_Usage/VRAM_management.html)
* DiffSynth-Studio Documentation: [Two-Stage Split Training](https://diffsynth-studio-doc.readthedocs.io/en/latest/Training/Split_Training.html)
* DiffSynth-Studio Documentation: [Low VRAM Training](https://diffsynth-studio-doc.readthedocs.io/en/latest/Pipeline_Usage/Model_Training.html#low-vram-training)
|