MiniMax-H3-Workflows / MiniMax_H3_Turbo_v4_8step_T2VA_BennyDaBall.json
BennyDaBall's picture
Add Turbo v4 8-step T2VA workflow (euler/beta, 1344x768, core nodes only)
d108764 verified
Raw
History Blame Contribute Delete
22.7 kB
{
"id": "be3e373d-ffe6-4906-8a2e-048730e65911",
"revision": 0,
"last_node_id": 17,
"last_link_id": 18,
"nodes": [
{
"id": 1,
"type": "UNETLoader",
"pos": [
30,
60
],
"size": [
410,
82
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"localized_name": "MODEL",
"name": "MODEL",
"type": "MODEL",
"links": [
1
]
}
],
"properties": {
"Node name for S&R": "UNETLoader",
"models": [
{
"name": "minimax_h3_fl2va_pruned_int8_convrot.safetensors",
"url": "https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors",
"directory": "diffusion_models"
}
]
},
"widgets_values": [
"minimax_h3_fl2va_pruned_int8_convrot.safetensors",
"default"
]
},
{
"id": 2,
"type": "LoraLoaderModelOnly",
"pos": [
30,
200
],
"size": [
410,
82
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"localized_name": "model",
"name": "model",
"type": "MODEL",
"link": 1
}
],
"outputs": [
{
"localized_name": "MODEL",
"name": "MODEL",
"type": "MODEL",
"links": [
2,
3
]
}
],
"properties": {
"Node name for S&R": "LoraLoaderModelOnly",
"models": [
{
"name": "minimax_h3_turbo_v4_step600_ema_pruned_comfyui.safetensors",
"url": "https://huggingface.co/drbaph/MiniMax-H3-Turbo-Lora-ComfyUI/resolve/main/minimax_h3_turbo_v4_step600_ema_pruned_comfyui.safetensors",
"directory": "loras"
}
]
},
"widgets_values": [
"minimax_h3_turbo_v4_step600_ema_pruned_comfyui.safetensors",
1.0
]
},
{
"id": 3,
"type": "CLIPLoader",
"pos": [
30,
340
],
"size": [
410,
106
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"localized_name": "CLIP",
"name": "CLIP",
"type": "CLIP",
"links": [
4
]
}
],
"properties": {
"Node name for S&R": "CLIPLoader",
"models": [
{
"name": "qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors",
"url": "https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors",
"directory": "text_encoders"
}
]
},
"widgets_values": [
"qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors",
"minimax",
"default"
]
},
{
"id": 4,
"type": "VAELoader",
"pos": [
30,
500
],
"size": [
410,
58
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"localized_name": "VAE",
"name": "VAE",
"type": "VAE",
"links": [
5,
6
]
}
],
"properties": {
"Node name for S&R": "VAELoader",
"models": [
{
"name": "minimax_h3_video_vae_fp16.safetensors",
"url": "https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/vae/minimax_h3_video_vae_fp16.safetensors",
"directory": "vae"
}
]
},
"widgets_values": [
"minimax_h3_video_vae_fp16.safetensors"
]
},
{
"id": 5,
"type": "VAELoader",
"pos": [
30,
620
],
"size": [
410,
58
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"localized_name": "VAE",
"name": "VAE",
"type": "VAE",
"links": [
7
]
}
],
"properties": {
"Node name for S&R": "VAELoader",
"models": [
{
"name": "minimax_h3_audio_vae_fp32.safetensors",
"url": "https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/vae/minimax_h3_audio_vae_fp32.safetensors",
"directory": "vae"
}
]
},
"widgets_values": [
"minimax_h3_audio_vae_fp32.safetensors"
]
},
{
"id": 6,
"type": "MiniMaxH3ImageToVideo",
"pos": [
500,
60
],
"size": [
460,
660
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"localized_name": "clip",
"name": "clip",
"type": "CLIP",
"link": 4
},
{
"localized_name": "vae",
"name": "vae",
"type": "VAE",
"link": 5
},
{
"localized_name": "first_frame",
"name": "first_frame",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"localized_name": "last_frame",
"name": "last_frame",
"type": "IMAGE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"localized_name": "positive",
"name": "positive",
"type": "CONDITIONING",
"links": [
8
]
},
{
"localized_name": "LATENT",
"name": "LATENT",
"type": "LATENT",
"links": [
9
]
}
],
"properties": {
"Node name for S&R": "MiniMaxH3ImageToVideo"
},
"widgets_values": [
"Style contract: photoreal cinematic live-action, warm 1950s American diner interior, chrome counter, red vinyl stools, soft window daylight, natural skin tones. One single continuous locked shot, no scene cuts. A cheerful waitress in a mint-green 1950s diner uniform with a white apron and a small paper cap stands behind the counter facing camera; a slice of cherry pie on a white plate sits on the counter in front of her.\nTimed beats:\n0.0-0.6s: she wipes her hands on her apron, glancing down at the pie.\n0.6-1.4s: she slides the plate forward across the counter with a soft ceramic clink and looks up at camera with a warm smile.\n1.4-3.6s: (S1) says: <d>[English] Fresh out of the oven, sugar — careful, that plate's hot.</d> Her lips sync naturally to the line; she raises her eyebrows playfully on the word hot.\n3.6-5.1s: she rests both hands on the counter, tilts her head, smiling, steam rising gently from the pie.\nCamera: completely static and locked-off — no pan, no tilt, no zoom, no drift. Medium shot at counter height.\nAudio: at 0.0s quiet diner room tone with a faint refrigerator hum and distant clink of dishes; soft apron fabric rustle 0.0-0.6s; a ceramic plate slide and clink at 0.6s; her clear warm female voice speaking at 1.4-3.6s; a soft single counter-bell ding at 4.3s; room tone continues to the end. No music.\nNegative: no on-screen text, no captions, no watermark, no scene cuts, no camera motion, no morphing hands, no extra fingers, no second person, no music.",
1344,
768,
124
]
},
{
"id": 7,
"type": "RandomNoise",
"pos": [
1020,
60
],
"size": [
380,
90
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"localized_name": "NOISE",
"name": "NOISE",
"type": "NOISE",
"links": [
10
]
}
],
"properties": {
"Node name for S&R": "RandomNoise"
},
"widgets_values": [
84070802,
"fixed"
]
},
{
"id": 8,
"type": "KSamplerSelect",
"pos": [
1020,
210
],
"size": [
380,
58
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"localized_name": "SAMPLER",
"name": "SAMPLER",
"type": "SAMPLER",
"links": [
11
]
}
],
"properties": {
"Node name for S&R": "KSamplerSelect"
},
"widgets_values": [
"euler"
]
},
{
"id": 9,
"type": "BasicScheduler",
"pos": [
1020,
330
],
"size": [
380,
130
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"localized_name": "model",
"name": "model",
"type": "MODEL",
"link": 2
}
],
"outputs": [
{
"localized_name": "SIGMAS",
"name": "SIGMAS",
"type": "SIGMAS",
"links": [
12
]
}
],
"properties": {
"Node name for S&R": "BasicScheduler"
},
"widgets_values": [
"beta",
8,
1.0
]
},
{
"id": 10,
"type": "BasicGuider",
"pos": [
1020,
520
],
"size": [
380,
60
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"localized_name": "model",
"name": "model",
"type": "MODEL",
"link": 3
},
{
"localized_name": "conditioning",
"name": "conditioning",
"type": "CONDITIONING",
"link": 8
}
],
"outputs": [
{
"localized_name": "GUIDER",
"name": "GUIDER",
"type": "GUIDER",
"links": [
13
]
}
],
"properties": {
"Node name for S&R": "BasicGuider"
}
},
{
"id": 11,
"type": "SamplerCustomAdvanced",
"pos": [
1020,
640
],
"size": [
380,
140
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"localized_name": "noise",
"name": "noise",
"type": "NOISE",
"link": 10
},
{
"localized_name": "guider",
"name": "guider",
"type": "GUIDER",
"link": 13
},
{
"localized_name": "sampler",
"name": "sampler",
"type": "SAMPLER",
"link": 11
},
{
"localized_name": "sigmas",
"name": "sigmas",
"type": "SIGMAS",
"link": 12
},
{
"localized_name": "latent_image",
"name": "latent_image",
"type": "LATENT",
"link": 9
}
],
"outputs": [
{
"localized_name": "output",
"name": "output",
"type": "LATENT",
"links": [
14,
15
]
},
{
"localized_name": "denoised_output",
"name": "denoised_output",
"type": "LATENT",
"links": null
}
],
"properties": {
"Node name for S&R": "SamplerCustomAdvanced"
},
"widgets_values": []
},
{
"id": 12,
"type": "VAEDecode",
"pos": [
1460,
60
],
"size": [
270,
66
],
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"localized_name": "samples",
"name": "samples",
"type": "LATENT",
"link": 14
},
{
"localized_name": "vae",
"name": "vae",
"type": "VAE",
"link": 6
}
],
"outputs": [
{
"localized_name": "IMAGE",
"name": "IMAGE",
"type": "IMAGE",
"links": [
16
]
}
],
"properties": {
"Node name for S&R": "VAEDecode"
}
},
{
"id": 13,
"type": "VAEDecodeAudio",
"pos": [
1460,
180
],
"size": [
270,
66
],
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"localized_name": "samples",
"name": "samples",
"type": "LATENT",
"link": 15
},
{
"localized_name": "vae",
"name": "vae",
"type": "VAE",
"link": 7
}
],
"outputs": [
{
"localized_name": "AUDIO",
"name": "AUDIO",
"type": "AUDIO",
"links": [
17
]
}
],
"properties": {
"Node name for S&R": "VAEDecodeAudio"
}
},
{
"id": 14,
"type": "CreateVideo",
"pos": [
1460,
300
],
"size": [
270,
110
],
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"localized_name": "images",
"name": "images",
"type": "IMAGE",
"link": 16
},
{
"localized_name": "audio",
"name": "audio",
"type": "AUDIO",
"link": 17,
"shape": 7
}
],
"outputs": [
{
"localized_name": "VIDEO",
"name": "VIDEO",
"type": "VIDEO",
"links": [
18
]
}
],
"properties": {
"Node name for S&R": "CreateVideo"
},
"widgets_values": [
24,
8
]
},
{
"id": 15,
"type": "SaveVideo",
"pos": [
1460,
470
],
"size": [
580,
330
],
"flags": {},
"order": 14,
"mode": 0,
"inputs": [
{
"localized_name": "video",
"name": "video",
"type": "VIDEO",
"link": 18
}
],
"outputs": [
{
"localized_name": "video",
"name": "video",
"type": "VIDEO",
"links": null
}
],
"properties": {
"Node name for S&R": "SaveVideo"
},
"widgets_values": [
"video/H3_turbo_v4",
"auto",
"auto"
]
},
{
"id": 16,
"type": "MarkdownNote",
"pos": [
-520,
60
],
"size": [
470,
830
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"Node name for S&R": "MarkdownNote"
},
"widgets_values": [
"## MiniMax H3 + Turbo v4 LoRA — 8-step video+audio\n\nText → video with **native stereo audio** (voice, SFX, room tone in one pass), ~**1.9× faster** than the stock 20-step recipe. Verified with a same-seed A/B against the 20-step baseline: same choreography, healthy audio (peak ≈ −9 dB, zero clipping).\n\n**The recipe**\n- Turbo v4 LoRA at strength **1.0** via the built-in *LoraLoaderModelOnly* — this graph is 100% core ComfyUI nodes, no custom node packs\n- **8 steps · euler · beta** — no sigma-shift node needed; core defaults carry the dual video/audio clock at this step count\n- CFG-free (BasicGuider): there is no negative-prompt input — write a `Negative:` line inside the prompt itself (see the demo prompt)\n- **1344×768 @ 24 fps**. `length` is frames on a 17k+5 grid: **124 ≈ 5 s**, 243 ≈ 10 s, 362 ≈ 15 s (model max)\n- Seed is **fixed** so your first run reproduces the demo clip; set it to *randomize* for new takes\n\n**Tested / not tested**\n- Verified on 5-second clips at 1344×768. Higher res (1920×1088) and longer clips are untested with this LoRA — earlier turbo versions fell apart there, so treat that as experimental\n- First run: watch the console — you should see **zero** \"lora key not loaded\" warnings on the pruned int8 checkpoint\n- Few-step audio blowout is the classic turbo failure mode; this v4 @ 8 steps is the combo that passed. If you drop steps further, check your audio peaks\n\n**Prompt pattern that works**\nStyle contract → timed beats (`0.0-0.6s: ...`) → camera lock line → an audio timeline starting with room tone at 0.0s → `Negative:` line. Dialogue inside `<d>[English] ... </d>` is spoken verbatim with lip-sync. Give the first spoken line ≥1.2 s of lead-in and anchor t=0 with room tone.\n\n*LoRA: larryvrh's MiniMax-H3-Turbo-Lora (v4_step600 EMA), ComfyUI convert by drbaph. Workflow shared by @BennyDaBall_OG.*"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 17,
"type": "MarkdownNote",
"pos": [
-1030,
60
],
"size": [
480,
830
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"Node name for S&R": "MarkdownNote"
},
"widgets_values": [
"## Model downloads\n\nComfyUI **≥ 0.30** (MiniMax H3 nodes are built in — update first). When you load this workflow, ComfyUI should offer to download any missing models automatically; links below if you'd rather grab them yourself. All base files are the official Comfy-Org releases.\n\n**diffusion_models** (19.5 GB)\n- [minimax_h3_fl2va_pruned_int8_convrot.safetensors](https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors)\n\n**text_encoders** (14.6 GB)\n- [qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors](https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors)\n\n**vae**\n- [minimax_h3_video_vae_fp16.safetensors](https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/vae/minimax_h3_video_vae_fp16.safetensors)\n- [minimax_h3_audio_vae_fp32.safetensors](https://huggingface.co/Comfy-Org/MiniMax-H3/resolve/main/vae/minimax_h3_audio_vae_fp32.safetensors)\n\n**loras** (620 MB)\n- [minimax_h3_turbo_v4_step600_ema_pruned_comfyui.safetensors](https://huggingface.co/drbaph/MiniMax-H3-Turbo-Lora-ComfyUI/blob/main/minimax_h3_turbo_v4_step600_ema_pruned_comfyui.safetensors)\n from [drbaph/MiniMax-H3-Turbo-Lora-ComfyUI](https://huggingface.co/drbaph/MiniMax-H3-Turbo-Lora-ComfyUI)\n\n**Folder placement**\n\n```\nComfyUI/models/\n├── diffusion_models/ minimax_h3_fl2va_pruned_int8_convrot.safetensors\n├── text_encoders/ qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors\n├── vae/ minimax_h3_video_vae_fp16.safetensors\n│ minimax_h3_audio_vae_fp32.safetensors\n└── loras/ minimax_h3_turbo_v4_step600_ema_pruned_comfyui.safetensors\n```"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
1,
1,
0,
2,
0,
"MODEL"
],
[
2,
2,
0,
9,
0,
"MODEL"
],
[
3,
2,
0,
10,
0,
"MODEL"
],
[
4,
3,
0,
6,
0,
"CLIP"
],
[
5,
4,
0,
6,
1,
"VAE"
],
[
6,
4,
0,
12,
1,
"VAE"
],
[
7,
5,
0,
13,
1,
"VAE"
],
[
8,
6,
0,
10,
1,
"CONDITIONING"
],
[
9,
6,
1,
11,
4,
"LATENT"
],
[
10,
7,
0,
11,
0,
"NOISE"
],
[
11,
8,
0,
11,
2,
"SAMPLER"
],
[
12,
9,
0,
11,
3,
"SIGMAS"
],
[
13,
10,
0,
11,
1,
"GUIDER"
],
[
14,
11,
0,
12,
0,
"LATENT"
],
[
15,
11,
0,
13,
0,
"LATENT"
],
[
16,
12,
0,
14,
0,
"IMAGE"
],
[
17,
13,
0,
14,
1,
"AUDIO"
],
[
18,
14,
0,
15,
0,
"VIDEO"
]
],
"groups": [
{
"id": 1,
"title": "1 · Models",
"bounding": [
10,
-10,
450,
700
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
},
{
"id": 2,
"title": "2 · Prompt → video+audio conditioning",
"bounding": [
480,
-10,
500,
740
],
"color": "#8A8",
"font_size": 24,
"flags": {}
},
{
"id": 3,
"title": "3 · Turbo sampling — 8 steps, euler/beta",
"bounding": [
1000,
-10,
420,
800
],
"color": "#b58b2a",
"font_size": 24,
"flags": {}
},
{
"id": 4,
"title": "4 · Decode + save (video with audio)",
"bounding": [
1440,
-10,
620,
820
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.42,
"offset": [
1090,
80
]
},
"workflow_info": {
"name": "MiniMax H3 + Turbo v4 LoRA (8-step T2VA)",
"author": "@BennyDaBall_OG",
"recipe": "turbo v4_step600 EMA @1.0, 8 steps euler/beta, 1344x768, 124f@24fps"
}
},
"version": 0.4
}