MiniMax-H3 / MiniMax-Long-Video-&-Extend-Video /4-Video-Extend-with-Reference-Image /4-Video-Extend-with-Reference-Image.json
| { | |
| "id": "76b5f713-5824-40ae-a2b7-f02ad62c256c", | |
| "revision": 0, | |
| "last_node_id": 951, | |
| "last_link_id": 308, | |
| "nodes": [ | |
| { | |
| "id": 311, | |
| "type": "BasicGuider", | |
| "pos": [ | |
| 2207.9859999999985, | |
| 50.73255834960937 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 71, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 65 | |
| }, | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 66 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "GUIDER", | |
| "type": "GUIDER", | |
| "links": [ | |
| 68 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicGuider" | |
| } | |
| }, | |
| { | |
| "id": 312, | |
| "type": "KSamplerSelect", | |
| "pos": [ | |
| 2207.9859999999985, | |
| 165.25400000000005 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 0, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "SAMPLER", | |
| "type": "SAMPLER", | |
| "links": [] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "KSamplerSelect" | |
| }, | |
| "widgets_values": [ | |
| "res_multistep" | |
| ], | |
| "widgets_values_named": { | |
| "sampler_name": "res_multistep" | |
| } | |
| }, | |
| { | |
| "id": 313, | |
| "type": "BasicScheduler", | |
| "pos": [ | |
| 2207.985999999998, | |
| 284.2119999999998 | |
| ], | |
| "size": [ | |
| 270, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 46, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 70 | |
| }, | |
| { | |
| "name": "steps", | |
| "type": "INT", | |
| "widget": { | |
| "name": "steps" | |
| }, | |
| "link": 215 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "SIGMAS", | |
| "type": "SIGMAS", | |
| "links": [ | |
| 71 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicScheduler" | |
| }, | |
| "widgets_values": [ | |
| "simple", | |
| 20, | |
| 1 | |
| ], | |
| "widgets_values_named": { | |
| "scheduler": "simple", | |
| "steps": 20, | |
| "denoise": 1 | |
| } | |
| }, | |
| { | |
| "id": 323, | |
| "type": "VHS_VideoCombine", | |
| "pos": [ | |
| 3570, | |
| -40 | |
| ], | |
| "size": [ | |
| 420, | |
| 334 | |
| ], | |
| "flags": {}, | |
| "order": 76, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 80 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 81 | |
| }, | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "Filenames", | |
| "type": "VHS_FILENAMES", | |
| "links": null | |
| } | |
| ], | |
| "title": "CLIP 4 PREVIEW β TEMP ONLY", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "1.7.7", | |
| "Node name for S&R": "VHS_VideoCombine" | |
| }, | |
| "widgets_values": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip04", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip04", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| } | |
| }, | |
| { | |
| "id": 411, | |
| "type": "BasicGuider", | |
| "pos": [ | |
| 2200, | |
| 875 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 82, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 96 | |
| }, | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 97 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "GUIDER", | |
| "type": "GUIDER", | |
| "links": [ | |
| 99 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicGuider" | |
| } | |
| }, | |
| { | |
| "id": 412, | |
| "type": "KSamplerSelect", | |
| "pos": [ | |
| 2200, | |
| 970 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 1, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "SAMPLER", | |
| "type": "SAMPLER", | |
| "links": [] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "KSamplerSelect" | |
| }, | |
| "widgets_values": [ | |
| "res_multistep" | |
| ], | |
| "widgets_values_named": { | |
| "sampler_name": "res_multistep" | |
| } | |
| }, | |
| { | |
| "id": 413, | |
| "type": "BasicScheduler", | |
| "pos": [ | |
| 2200, | |
| 1065 | |
| ], | |
| "size": [ | |
| 270, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 47, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 101 | |
| }, | |
| { | |
| "name": "steps", | |
| "type": "INT", | |
| "widget": { | |
| "name": "steps" | |
| }, | |
| "link": 216 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "SIGMAS", | |
| "type": "SIGMAS", | |
| "links": [ | |
| 102 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicScheduler" | |
| }, | |
| "widgets_values": [ | |
| "simple", | |
| 20, | |
| 1 | |
| ], | |
| "widgets_values_named": { | |
| "scheduler": "simple", | |
| "steps": 20, | |
| "denoise": 1 | |
| } | |
| }, | |
| { | |
| "id": 423, | |
| "type": "VHS_VideoCombine", | |
| "pos": [ | |
| 3570, | |
| 810 | |
| ], | |
| "size": [ | |
| 420, | |
| 334 | |
| ], | |
| "flags": {}, | |
| "order": 87, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 111 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 112 | |
| }, | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "Filenames", | |
| "type": "VHS_FILENAMES", | |
| "links": null | |
| } | |
| ], | |
| "title": "CLIP 5 PREVIEW β TEMP ONLY", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "1.7.7", | |
| "Node name for S&R": "VHS_VideoCombine" | |
| }, | |
| "widgets_values": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip05", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip05", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| } | |
| }, | |
| { | |
| "id": 511, | |
| "type": "BasicGuider", | |
| "pos": [ | |
| 2200, | |
| 1725 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 93, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 127 | |
| }, | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 128 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "GUIDER", | |
| "type": "GUIDER", | |
| "links": [ | |
| 130 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicGuider" | |
| } | |
| }, | |
| { | |
| "id": 512, | |
| "type": "KSamplerSelect", | |
| "pos": [ | |
| 2200, | |
| 1820 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 2, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "SAMPLER", | |
| "type": "SAMPLER", | |
| "links": [] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "KSamplerSelect" | |
| }, | |
| "widgets_values": [ | |
| "res_multistep" | |
| ], | |
| "widgets_values_named": { | |
| "sampler_name": "res_multistep" | |
| } | |
| }, | |
| { | |
| "id": 513, | |
| "type": "BasicScheduler", | |
| "pos": [ | |
| 2200, | |
| 1915 | |
| ], | |
| "size": [ | |
| 270, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 48, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 132 | |
| }, | |
| { | |
| "name": "steps", | |
| "type": "INT", | |
| "widget": { | |
| "name": "steps" | |
| }, | |
| "link": 217 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "SIGMAS", | |
| "type": "SIGMAS", | |
| "links": [ | |
| 133 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicScheduler" | |
| }, | |
| "widgets_values": [ | |
| "simple", | |
| 20, | |
| 1 | |
| ], | |
| "widgets_values_named": { | |
| "scheduler": "simple", | |
| "steps": 20, | |
| "denoise": 1 | |
| } | |
| }, | |
| { | |
| "id": 531, | |
| "type": "AudioConcat", | |
| "pos": [ | |
| 3251.1190199999996, | |
| 2160.799652198488 | |
| ], | |
| "size": [ | |
| 280, | |
| 120 | |
| ], | |
| "flags": {}, | |
| "order": 99, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "audio1", | |
| "type": "AUDIO", | |
| "link": 146 | |
| }, | |
| { | |
| "name": "audio2", | |
| "type": "AUDIO", | |
| "link": 147 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 177 | |
| ] | |
| } | |
| ], | |
| "title": "CUMULATIVE AUDIO through Clip 6", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "AudioConcat" | |
| }, | |
| "widgets_values": [ | |
| "after" | |
| ], | |
| "widgets_values_named": { | |
| "direction": "after" | |
| } | |
| }, | |
| { | |
| "id": 611, | |
| "type": "BasicGuider", | |
| "pos": [ | |
| 2200, | |
| 2575 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 104, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 158 | |
| }, | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 159 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "GUIDER", | |
| "type": "GUIDER", | |
| "links": [ | |
| 161 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicGuider" | |
| } | |
| }, | |
| { | |
| "id": 612, | |
| "type": "KSamplerSelect", | |
| "pos": [ | |
| 2200, | |
| 2670 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 3, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "SAMPLER", | |
| "type": "SAMPLER", | |
| "links": [] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "KSamplerSelect" | |
| }, | |
| "widgets_values": [ | |
| "res_multistep" | |
| ], | |
| "widgets_values_named": { | |
| "sampler_name": "res_multistep" | |
| } | |
| }, | |
| { | |
| "id": 613, | |
| "type": "BasicScheduler", | |
| "pos": [ | |
| 2200, | |
| 2765 | |
| ], | |
| "size": [ | |
| 270, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 49, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 163 | |
| }, | |
| { | |
| "name": "steps", | |
| "type": "INT", | |
| "widget": { | |
| "name": "steps" | |
| }, | |
| "link": 218 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "SIGMAS", | |
| "type": "SIGMAS", | |
| "links": [ | |
| 164 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicScheduler" | |
| }, | |
| "widgets_values": [ | |
| "simple", | |
| 20, | |
| 1 | |
| ], | |
| "widgets_values_named": { | |
| "scheduler": "simple", | |
| "steps": 20, | |
| "denoise": 1 | |
| } | |
| }, | |
| { | |
| "id": 800, | |
| "type": "VHS_VideoCombine", | |
| "pos": [ | |
| 4150, | |
| 2520 | |
| ], | |
| "size": [ | |
| 420, | |
| 578.7692307692307 | |
| ], | |
| "flags": {}, | |
| "order": 112, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 179 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 180 | |
| }, | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "Filenames", | |
| "type": "VHS_FILENAMES", | |
| "links": null | |
| } | |
| ], | |
| "title": "FINAL STITCHED VIDEO β SAVE + PREVIEW", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "1.7.7", | |
| "Node name for S&R": "VHS_VideoCombine" | |
| }, | |
| "widgets_values": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "video/input_video_extension", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 19, | |
| "save_metadata": true, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": true, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": { | |
| "filename": "input_video_extension_00003-audio.mp4", | |
| "subfolder": "video", | |
| "type": "output", | |
| "format": "video/h264-mp4", | |
| "frame_rate": 24, | |
| "workflow": "input_video_extension_00003.png", | |
| "fullpath": "E:\\ComfyUI_windows_portable_nvidia\\ComfyUI_windows_portable\\ComfyUI\\output\\video\\input_video_extension_00003-audio.mp4" | |
| } | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "video/input_video_extension", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 19, | |
| "save_metadata": true, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": true, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": { | |
| "filename": "input_video_extension_00003-audio.mp4", | |
| "subfolder": "video", | |
| "type": "output", | |
| "format": "video/h264-mp4", | |
| "frame_rate": 24, | |
| "workflow": "input_video_extension_00003.png", | |
| "fullpath": "E:\\ComfyUI_windows_portable_nvidia\\ComfyUI_windows_portable\\ComfyUI\\output\\video\\input_video_extension_00003-audio.mp4" | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "id": 920, | |
| "type": "Any Switch (rgthree)", | |
| "pos": [ | |
| -970, | |
| 1030 | |
| ], | |
| "size": [ | |
| 760, | |
| 200 | |
| ], | |
| "flags": {}, | |
| "order": 56, | |
| "mode": 2, | |
| "inputs": [ | |
| { | |
| "dir": 3, | |
| "name": "any_01", | |
| "type": "AUDIO", | |
| "link": 263 | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_02", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_03", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_04", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_05", | |
| "type": "AUDIO", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "dir": 4, | |
| "label": "AUDIO", | |
| "name": "*", | |
| "type": "AUDIO", | |
| "links": [ | |
| 200 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 3 β FULL AUDIO OF CLIP 2", | |
| "properties": { | |
| "cnr_id": "rgthree-comfy", | |
| "ver": "6b76ee6f2c5a007710b5a16f97c94330d6ecc871", | |
| "Node name for S&R": "Any Switch (rgthree)" | |
| } | |
| }, | |
| { | |
| "id": 921, | |
| "type": "Any Switch (rgthree)", | |
| "pos": [ | |
| -970, | |
| 1310 | |
| ], | |
| "size": [ | |
| 760, | |
| 200 | |
| ], | |
| "flags": {}, | |
| "order": 67, | |
| "mode": 2, | |
| "inputs": [ | |
| { | |
| "dir": 3, | |
| "name": "any_01", | |
| "type": "AUDIO", | |
| "link": 201 | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_02", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_03", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_04", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_05", | |
| "type": "AUDIO", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "dir": 4, | |
| "label": "AUDIO", | |
| "name": "*", | |
| "type": "AUDIO", | |
| "links": [ | |
| 202 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 4 β FULL AUDIO OF CLIP 3", | |
| "properties": { | |
| "cnr_id": "rgthree-comfy", | |
| "ver": "6b76ee6f2c5a007710b5a16f97c94330d6ecc871", | |
| "Node name for S&R": "Any Switch (rgthree)" | |
| } | |
| }, | |
| { | |
| "id": 922, | |
| "type": "Any Switch (rgthree)", | |
| "pos": [ | |
| -970, | |
| 1590 | |
| ], | |
| "size": [ | |
| 760, | |
| 200 | |
| ], | |
| "flags": {}, | |
| "order": 78, | |
| "mode": 2, | |
| "inputs": [ | |
| { | |
| "dir": 3, | |
| "name": "any_01", | |
| "type": "AUDIO", | |
| "link": 203 | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_02", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_03", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_04", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_05", | |
| "type": "AUDIO", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "dir": 4, | |
| "label": "AUDIO", | |
| "name": "*", | |
| "type": "AUDIO", | |
| "links": [ | |
| 204 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 5 β FULL AUDIO OF CLIP 4", | |
| "properties": { | |
| "cnr_id": "rgthree-comfy", | |
| "ver": "6b76ee6f2c5a007710b5a16f97c94330d6ecc871", | |
| "Node name for S&R": "Any Switch (rgthree)" | |
| } | |
| }, | |
| { | |
| "id": 923, | |
| "type": "Any Switch (rgthree)", | |
| "pos": [ | |
| -970, | |
| 1870 | |
| ], | |
| "size": [ | |
| 760, | |
| 200 | |
| ], | |
| "flags": {}, | |
| "order": 89, | |
| "mode": 2, | |
| "inputs": [ | |
| { | |
| "dir": 3, | |
| "name": "any_01", | |
| "type": "AUDIO", | |
| "link": 205 | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_02", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_03", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_04", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_05", | |
| "type": "AUDIO", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "dir": 4, | |
| "label": "AUDIO", | |
| "name": "*", | |
| "type": "AUDIO", | |
| "links": [ | |
| 206 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 6 β FULL AUDIO OF CLIP 5", | |
| "properties": { | |
| "cnr_id": "rgthree-comfy", | |
| "ver": "6b76ee6f2c5a007710b5a16f97c94330d6ecc871", | |
| "Node name for S&R": "Any Switch (rgthree)" | |
| } | |
| }, | |
| { | |
| "id": 924, | |
| "type": "Any Switch (rgthree)", | |
| "pos": [ | |
| -970, | |
| 2150 | |
| ], | |
| "size": [ | |
| 760, | |
| 200 | |
| ], | |
| "flags": {}, | |
| "order": 100, | |
| "mode": 2, | |
| "inputs": [ | |
| { | |
| "dir": 3, | |
| "name": "any_01", | |
| "type": "AUDIO", | |
| "link": 207 | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_02", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_03", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_04", | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "dir": 3, | |
| "name": "any_05", | |
| "type": "AUDIO", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "dir": 4, | |
| "label": "AUDIO", | |
| "name": "*", | |
| "type": "AUDIO", | |
| "links": [ | |
| 208 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 7 β FULL AUDIO OF CLIP 6", | |
| "properties": { | |
| "cnr_id": "rgthree-comfy", | |
| "ver": "6b76ee6f2c5a007710b5a16f97c94330d6ecc871", | |
| "Node name for S&R": "Any Switch (rgthree)" | |
| } | |
| }, | |
| { | |
| "id": 932, | |
| "type": "PathchSageAttentionKJ", | |
| "pos": [ | |
| -2465.735974619766, | |
| 168.78800133362822 | |
| ], | |
| "size": [ | |
| 270, | |
| 82 | |
| ], | |
| "flags": {}, | |
| "order": 33, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 2 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "MODEL", | |
| "type": "MODEL", | |
| "links": [ | |
| 209 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "05ba05bf52331622bd3395716c505c29aa1fff76", | |
| "Node name for S&R": "PathchSageAttentionKJ" | |
| }, | |
| "widgets_values": [ | |
| "auto", | |
| false | |
| ], | |
| "widgets_values_named": { | |
| "sage_attention": "auto", | |
| "allow_compile": false | |
| } | |
| }, | |
| { | |
| "id": 933, | |
| "type": "MiniMaxH3MemoryEfficientSageAttentionPatch", | |
| "pos": [ | |
| -2471.410740633322, | |
| 300.3247479658497 | |
| ], | |
| "size": [ | |
| 320.323828125, | |
| 26 | |
| ], | |
| "flags": {}, | |
| "order": 36, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 209 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "links": [ | |
| 210 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "05ba05bf52331622bd3395716c505c29aa1fff76", | |
| "Node name for S&R": "MiniMaxH3MemoryEfficientSageAttentionPatch" | |
| } | |
| }, | |
| { | |
| "id": 934, | |
| "type": "SolAttnPatch", | |
| "pos": [ | |
| -2473.9242947003318, | |
| 396.73281077576706 | |
| ], | |
| "size": [ | |
| 270, | |
| 342 | |
| ], | |
| "flags": {}, | |
| "order": 38, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 210 | |
| }, | |
| { | |
| "name": "tau_profile", | |
| "shape": 7, | |
| "type": "STRING", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "MODEL", | |
| "type": "MODEL", | |
| "links": [ | |
| 211 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "aux_id": "kijai/ComfyUI-SolAttn_triton", | |
| "ver": "0e334dc981cfe3b0ed926ee13ad43f64914b7f5b", | |
| "Node name for S&R": "SolAttnPatch" | |
| }, | |
| "widgets_values": [ | |
| 1.4, | |
| 0.2, | |
| 0.9, | |
| 4096, | |
| true, | |
| "exact_kv_and_rows", | |
| false, | |
| "2d_frame", | |
| true, | |
| false, | |
| false, | |
| "33-35, 39-42" | |
| ], | |
| "widgets_values_named": { | |
| "tau": 1.4, | |
| "start_percent": 0.2, | |
| "end_percent": 0.9, | |
| "min_tokens": 4096, | |
| "int8_qk": true, | |
| "sink_conditioning": "exact_kv_and_rows", | |
| "morton": false, | |
| "morton_curve": "2d_frame", | |
| "int8_pv": true, | |
| "verbose": false, | |
| "use_tma": false, | |
| "dense_blocks": "33-35, 39-42" | |
| } | |
| }, | |
| { | |
| "id": 937, | |
| "type": "KSamplerSelect", | |
| "pos": [ | |
| -2445.1219754873764, | |
| -50.79805929653507 | |
| ], | |
| "size": [ | |
| 300.1692791748046, | |
| 71.77463958740236 | |
| ], | |
| "flags": {}, | |
| "order": 4, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "SAMPLER", | |
| "type": "SAMPLER", | |
| "links": [ | |
| 12, | |
| 38, | |
| 69, | |
| 100, | |
| 131, | |
| 162 | |
| ] | |
| } | |
| ], | |
| "title": "Global KSamplerSelect", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "KSamplerSelect" | |
| }, | |
| "widgets_values": [ | |
| "res_multistep" | |
| ], | |
| "widgets_values_named": { | |
| "sampler_name": "res_multistep" | |
| }, | |
| "color": "#232", | |
| "bgcolor": "#353" | |
| }, | |
| { | |
| "id": 931, | |
| "type": "Fast Groups Muter (rgthree)", | |
| "pos": [ | |
| -2874.380140068118, | |
| 1093.8016528925612 | |
| ], | |
| "size": [ | |
| 650, | |
| 180 | |
| ], | |
| "flags": {}, | |
| "order": 5, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "OPT_CONNECTION", | |
| "type": "*", | |
| "links": null | |
| } | |
| ], | |
| "title": "FULL PREVIOUS AUDIO REF β TOGGLE HERE", | |
| "properties": { | |
| "matchColors": "", | |
| "matchTitle": "OPTIONAL FULL PREVIOUS AUDIO REF", | |
| "showNav": true, | |
| "showAllGraphs": true, | |
| "sort": "position", | |
| "customSortAlphabet": "", | |
| "toggleRestriction": "default" | |
| } | |
| }, | |
| { | |
| "id": 939, | |
| "type": "PrimitiveInt", | |
| "pos": [ | |
| -3223.401266584812, | |
| -546.2077908528505 | |
| ], | |
| "size": [ | |
| 300, | |
| 82 | |
| ], | |
| "flags": {}, | |
| "order": 6, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "INT", | |
| "type": "INT", | |
| "links": [ | |
| 219, | |
| 220, | |
| 221, | |
| 222, | |
| 223, | |
| 224, | |
| 225, | |
| 226, | |
| 227, | |
| 228, | |
| 247 | |
| ] | |
| } | |
| ], | |
| "title": "GLOBAL CONTEXT FRAMES", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.31.0", | |
| "Node name for S&R": "PrimitiveInt" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "value": 39, | |
| "fixed": "fixed" | |
| }, | |
| "color": "#232", | |
| "bgcolor": "#353" | |
| }, | |
| { | |
| "id": 940, | |
| "type": "PrimitiveInt", | |
| "pos": [ | |
| -3222.9459103852914, | |
| -420.2620726802107 | |
| ], | |
| "size": [ | |
| 300, | |
| 82 | |
| ], | |
| "flags": {}, | |
| "order": 7, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "INT", | |
| "type": "INT", | |
| "links": [ | |
| 229, | |
| 231, | |
| 233, | |
| 235, | |
| 237, | |
| 250 | |
| ] | |
| } | |
| ], | |
| "title": "GLOBAL VIDEO CROSSFADE", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.31.0", | |
| "Node name for S&R": "PrimitiveInt" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "value": 39, | |
| "fixed": "fixed" | |
| }, | |
| "color": "#232", | |
| "bgcolor": "#353" | |
| }, | |
| { | |
| "id": 902, | |
| "type": "Note", | |
| "pos": [ | |
| -2500, | |
| -796.1266802062987 | |
| ], | |
| "size": [ | |
| 410, | |
| 170 | |
| ], | |
| "flags": {}, | |
| "order": 8, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [], | |
| "title": "Optional clip switch instructions", | |
| "properties": { | |
| "Node name for S&R": "Note" | |
| }, | |
| "widgets_values": [ | |
| "Optional groups are BYPASSED by default.\n\nTurn them on from top to bottom only: Clip 3, then 4, then 5, then 6, then 7." | |
| ], | |
| "widgets_values_named": { | |
| "text": "Optional groups are BYPASSED by default.\n\nTurn them on from top to bottom only: Clip 3, then 4, then 5, then 6, then 7." | |
| }, | |
| "color": "#432", | |
| "bgcolor": "#653" | |
| }, | |
| { | |
| "id": 200, | |
| "type": "MiniMaxH3ReferenceToVideo", | |
| "pos": [ | |
| 1204.5299739800844, | |
| -896.1025657999999 | |
| ], | |
| "size": [ | |
| 480, | |
| 430 | |
| ], | |
| "flags": {}, | |
| "order": 58, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "clip", | |
| "type": "CLIP", | |
| "link": 24 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 25 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "type": "VAE", | |
| "link": 184 | |
| }, | |
| { | |
| "label": "ref_image_0", | |
| "name": "ref_images.ref_image_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 185 | |
| }, | |
| { | |
| "label": "ref_image_1", | |
| "name": "ref_images.ref_image_1", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 186 | |
| }, | |
| { | |
| "label": "ref_image_2", | |
| "name": "ref_images.ref_image_2", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_0", | |
| "name": "ref_videos.ref_video_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_audio_0", | |
| "name": "ref_video_audios.ref_video_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_audio_0", | |
| "name": "ref_audios.ref_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 200 | |
| }, | |
| { | |
| "label": "ref_audio_1", | |
| "name": "ref_audios.ref_audio_1", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "widget": { | |
| "name": "width" | |
| }, | |
| "link": 270 | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "widget": { | |
| "name": "height" | |
| }, | |
| "link": 271 | |
| }, | |
| { | |
| "name": "length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "length" | |
| }, | |
| "link": 28 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "positive", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 29 | |
| ] | |
| }, | |
| { | |
| "name": "LATENT", | |
| "type": "LATENT", | |
| "links": [ | |
| 31, | |
| 41 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 3 β CONTINUATION + GLOBAL REFS", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "MiniMaxH3ReferenceToVideo" | |
| }, | |
| "widgets_values": [ | |
| "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| 960, | |
| 544, | |
| 362, | |
| "match" | |
| ], | |
| "widgets_values_named": { | |
| "prompt": "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| "width": 960, | |
| "height": 544, | |
| "length": 362, | |
| "ref_image_size": "match" | |
| }, | |
| "color": "#346434", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 211, | |
| "type": "BasicGuider", | |
| "pos": [ | |
| 2207.7100276163815, | |
| -801.6066912841799 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 60, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 34 | |
| }, | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 35 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "GUIDER", | |
| "type": "GUIDER", | |
| "links": [ | |
| 37 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicGuider" | |
| } | |
| }, | |
| { | |
| "id": 212, | |
| "type": "KSamplerSelect", | |
| "pos": [ | |
| 2211.115013808189, | |
| -665.4666912841801 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 9, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "SAMPLER", | |
| "type": "SAMPLER", | |
| "links": [] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "KSamplerSelect" | |
| }, | |
| "widgets_values": [ | |
| "res_multistep" | |
| ], | |
| "widgets_values_named": { | |
| "sampler_name": "res_multistep" | |
| } | |
| }, | |
| { | |
| "id": 201, | |
| "type": "MiniMaxH3MotionContext", | |
| "pos": [ | |
| 1698.546881079664, | |
| -887.0085130199157 | |
| ], | |
| "size": [ | |
| 488.6724609375, | |
| 350 | |
| ], | |
| "flags": {}, | |
| "order": 59, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 29 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 30 | |
| }, | |
| { | |
| "name": "latent", | |
| "type": "LATENT", | |
| "link": 31 | |
| }, | |
| { | |
| "name": "context_frames", | |
| "type": "IMAGE", | |
| "link": 260 | |
| }, | |
| { | |
| "name": "context_latent", | |
| "shape": 7, | |
| "type": "LATENT", | |
| "link": 33 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "context_length" | |
| }, | |
| "link": 219 | |
| }, | |
| { | |
| "name": "audio_context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "audio_context_length" | |
| }, | |
| "link": 220 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 35 | |
| ] | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "links": [ | |
| 48 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 3 Motion Context β GLOBAL / video / head / GLOBAL / timeline", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContext" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "video", | |
| "head", | |
| "disabled", | |
| 39, | |
| "timeline" | |
| ], | |
| "widgets_values_named": { | |
| "context_length": 39, | |
| "encode_mode": "video", | |
| "anchor_mode": "head", | |
| "crop": "disabled", | |
| "audio_context_length": 39, | |
| "audio_mode": "timeline" | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 213, | |
| "type": "BasicScheduler", | |
| "pos": [ | |
| 2209.63757960819, | |
| -543.6683225240113 | |
| ], | |
| "size": [ | |
| 270, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 45, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 39 | |
| }, | |
| { | |
| "name": "steps", | |
| "type": "INT", | |
| "widget": { | |
| "name": "steps" | |
| }, | |
| "link": 214 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "SIGMAS", | |
| "type": "SIGMAS", | |
| "links": [ | |
| 40 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicScheduler" | |
| }, | |
| "widgets_values": [ | |
| "simple", | |
| 20, | |
| 1 | |
| ], | |
| "widgets_values_named": { | |
| "scheduler": "simple", | |
| "steps": 20, | |
| "denoise": 1 | |
| } | |
| }, | |
| { | |
| "id": 214, | |
| "type": "SamplerCustomAdvanced", | |
| "pos": [ | |
| 2490.8021599367203, | |
| -838.8733087158201 | |
| ], | |
| "size": [ | |
| 300, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 61, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "noise", | |
| "type": "NOISE", | |
| "link": 36 | |
| }, | |
| { | |
| "name": "guider", | |
| "type": "GUIDER", | |
| "link": 37 | |
| }, | |
| { | |
| "name": "sampler", | |
| "type": "SAMPLER", | |
| "link": 38 | |
| }, | |
| { | |
| "name": "sigmas", | |
| "type": "SIGMAS", | |
| "link": 40 | |
| }, | |
| { | |
| "name": "latent_image", | |
| "type": "LATENT", | |
| "link": 41 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "output", | |
| "type": "LATENT", | |
| "links": [ | |
| 42, | |
| 44, | |
| 64 | |
| ] | |
| }, | |
| { | |
| "name": "denoised_output", | |
| "type": "LATENT", | |
| "links": [] | |
| } | |
| ], | |
| "title": "Clip 3 Sampler", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "SamplerCustomAdvanced" | |
| } | |
| }, | |
| { | |
| "id": 220, | |
| "type": "VAEDecode", | |
| "pos": [ | |
| 2809.726354679665, | |
| -871.6923026000001 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 62, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 42 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 43 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 46 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 3 Frames", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecode" | |
| } | |
| }, | |
| { | |
| "id": 221, | |
| "type": "VAEDecodeAudio", | |
| "pos": [ | |
| 2805.8291583601695, | |
| -740.7093286398311 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 63, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 44 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 45 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 47 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 3 Audio", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecodeAudio" | |
| } | |
| }, | |
| { | |
| "id": 222, | |
| "type": "MiniMaxH3MotionContextTrim", | |
| "pos": [ | |
| 3061.419402038319, | |
| -851.6425452300415 | |
| ], | |
| "size": [ | |
| 499.480078125, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 64, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 46 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 47 | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "widget": { | |
| "name": "trim_frames" | |
| }, | |
| "link": 48 | |
| }, | |
| { | |
| "name": "video_crossfade_frames", | |
| "shape": 7, | |
| "type": "INT", | |
| "widget": { | |
| "name": "video_crossfade_frames" | |
| }, | |
| "link": 229 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 49, | |
| 63 | |
| ] | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 50, | |
| 54, | |
| 201 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 52 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_frames", | |
| "type": "INT", | |
| "links": [ | |
| 230 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 3 β TRIM AUDIO / PREP VIDEO OVERLAP + MATCH AUDIO TAIL", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContextTrim" | |
| }, | |
| "widgets_values": [ | |
| 0, | |
| 24, | |
| true, | |
| 0 | |
| ], | |
| "widgets_values_named": { | |
| "trim_frames": 0, | |
| "fps": 24, | |
| "match_tail": true, | |
| "video_crossfade_frames": 0 | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 223, | |
| "type": "VHS_VideoCombine", | |
| "pos": [ | |
| 3576.4955650796624, | |
| -892.5982498199158 | |
| ], | |
| "size": [ | |
| 420, | |
| 334 | |
| ], | |
| "flags": {}, | |
| "order": 65, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 49 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 50 | |
| }, | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "Filenames", | |
| "type": "VHS_FILENAMES", | |
| "links": null | |
| } | |
| ], | |
| "title": "CLIP 3 PREVIEW β TEMP ONLY", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "1.7.7", | |
| "Node name for S&R": "VHS_VideoCombine" | |
| }, | |
| "widgets_values": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip03", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip03", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| } | |
| }, | |
| { | |
| "id": 230, | |
| "type": "ImageBatchExtendWithOverlap", | |
| "pos": [ | |
| 3149.785730800004, | |
| -586.7088585007582 | |
| ], | |
| "size": [ | |
| 378.9283203125, | |
| 155 | |
| ], | |
| "flags": {}, | |
| "order": 68, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "link": 261 | |
| }, | |
| { | |
| "name": "new_images", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 52 | |
| }, | |
| { | |
| "name": "overlap", | |
| "type": "INT", | |
| "widget": { | |
| "name": "overlap" | |
| }, | |
| "link": 230 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "start_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "extended_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 82 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "073efb07419f56cc714e099a82e49fbc23ad9263", | |
| "Node name for S&R": "ImageBatchExtendWithOverlap" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "source", | |
| "linear_blend" | |
| ], | |
| "widgets_values_named": { | |
| "overlap": 39, | |
| "overlap_side": "source", | |
| "overlap_mode": "linear_blend" | |
| } | |
| }, | |
| { | |
| "id": 231, | |
| "type": "AudioConcat", | |
| "pos": [ | |
| 3207.9680232015157, | |
| -386.7739116000001 | |
| ], | |
| "size": [ | |
| 280, | |
| 120 | |
| ], | |
| "flags": {}, | |
| "order": 66, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "audio1", | |
| "type": "AUDIO", | |
| "link": 262 | |
| }, | |
| { | |
| "name": "audio2", | |
| "type": "AUDIO", | |
| "link": 54 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 84 | |
| ] | |
| } | |
| ], | |
| "title": "CUMULATIVE AUDIO through Clip 3", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "AudioConcat" | |
| }, | |
| "widgets_values": [ | |
| "after" | |
| ], | |
| "widgets_values_named": { | |
| "direction": "after" | |
| } | |
| }, | |
| { | |
| "id": 300, | |
| "type": "MiniMaxH3ReferenceToVideo", | |
| "pos": [ | |
| 1204.5298550398315, | |
| -50 | |
| ], | |
| "size": [ | |
| 480, | |
| 430 | |
| ], | |
| "flags": {}, | |
| "order": 69, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "clip", | |
| "type": "CLIP", | |
| "link": 55 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 56 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "type": "VAE", | |
| "link": 187 | |
| }, | |
| { | |
| "label": "ref_image_0", | |
| "name": "ref_images.ref_image_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 188 | |
| }, | |
| { | |
| "label": "ref_image_1", | |
| "name": "ref_images.ref_image_1", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 189 | |
| }, | |
| { | |
| "label": "ref_image_2", | |
| "name": "ref_images.ref_image_2", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_0", | |
| "name": "ref_videos.ref_video_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_audio_0", | |
| "name": "ref_video_audios.ref_video_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_audio_0", | |
| "name": "ref_audios.ref_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 202 | |
| }, | |
| { | |
| "label": "ref_audio_1", | |
| "name": "ref_audios.ref_audio_1", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "widget": { | |
| "name": "width" | |
| }, | |
| "link": 272 | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "widget": { | |
| "name": "height" | |
| }, | |
| "link": 273 | |
| }, | |
| { | |
| "name": "length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "length" | |
| }, | |
| "link": 59 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "positive", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 60 | |
| ] | |
| }, | |
| { | |
| "name": "LATENT", | |
| "type": "LATENT", | |
| "links": [ | |
| 62, | |
| 72 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 4 β CONTINUATION + GLOBAL REFS", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "MiniMaxH3ReferenceToVideo" | |
| }, | |
| "widgets_values": [ | |
| "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| 960, | |
| 544, | |
| 362, | |
| "match" | |
| ], | |
| "widgets_values_named": { | |
| "prompt": "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| "width": 960, | |
| "height": 544, | |
| "length": 362, | |
| "ref_image_size": "match" | |
| }, | |
| "color": "#346434", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 301, | |
| "type": "MiniMaxH3MotionContext", | |
| "pos": [ | |
| 1712.8376713800842, | |
| -44.80338141991577 | |
| ], | |
| "size": [ | |
| 489.0689453125, | |
| 350 | |
| ], | |
| "flags": {}, | |
| "order": 70, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 60 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 61 | |
| }, | |
| { | |
| "name": "latent", | |
| "type": "LATENT", | |
| "link": 62 | |
| }, | |
| { | |
| "name": "context_frames", | |
| "type": "IMAGE", | |
| "link": 63 | |
| }, | |
| { | |
| "name": "context_latent", | |
| "shape": 7, | |
| "type": "LATENT", | |
| "link": 64 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "context_length" | |
| }, | |
| "link": 221 | |
| }, | |
| { | |
| "name": "audio_context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "audio_context_length" | |
| }, | |
| "link": 222 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 66 | |
| ] | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "links": [ | |
| 79 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 4 Motion Context β GLOBAL / video / head / GLOBAL / timeline", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContext" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "video", | |
| "head", | |
| "disabled", | |
| 39, | |
| "timeline" | |
| ], | |
| "widgets_values_named": { | |
| "context_length": 39, | |
| "encode_mode": "video", | |
| "anchor_mode": "head", | |
| "crop": "disabled", | |
| "audio_context_length": 39, | |
| "audio_mode": "timeline" | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 314, | |
| "type": "SamplerCustomAdvanced", | |
| "pos": [ | |
| 2498.427527120338, | |
| 25.196618580084227 | |
| ], | |
| "size": [ | |
| 300, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 72, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "noise", | |
| "type": "NOISE", | |
| "link": 67 | |
| }, | |
| { | |
| "name": "guider", | |
| "type": "GUIDER", | |
| "link": 68 | |
| }, | |
| { | |
| "name": "sampler", | |
| "type": "SAMPLER", | |
| "link": 69 | |
| }, | |
| { | |
| "name": "sigmas", | |
| "type": "SIGMAS", | |
| "link": 71 | |
| }, | |
| { | |
| "name": "latent_image", | |
| "type": "LATENT", | |
| "link": 72 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "output", | |
| "type": "LATENT", | |
| "links": [ | |
| 73, | |
| 75, | |
| 95 | |
| ] | |
| }, | |
| { | |
| "name": "denoised_output", | |
| "type": "LATENT", | |
| "links": [] | |
| } | |
| ], | |
| "title": "Clip 4 Sampler", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "SamplerCustomAdvanced" | |
| } | |
| }, | |
| { | |
| "id": 320, | |
| "type": "VAEDecode", | |
| "pos": [ | |
| 2812.3247234398323, | |
| -13.897434199999998 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 73, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 73 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 74 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 77 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 4 Frames", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecode" | |
| } | |
| }, | |
| { | |
| "id": 321, | |
| "type": "VAEDecodeAudio", | |
| "pos": [ | |
| 2808.427289239832, | |
| 123.58122378008423 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 74, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 75 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 76 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 78 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 4 Audio", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecodeAudio" | |
| } | |
| }, | |
| { | |
| "id": 322, | |
| "type": "MiniMaxH3MotionContextTrim", | |
| "pos": [ | |
| 3057.1794363531267, | |
| -2.8033814199157723 | |
| ], | |
| "size": [ | |
| 499.8765625, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 75, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 77 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 78 | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "widget": { | |
| "name": "trim_frames" | |
| }, | |
| "link": 79 | |
| }, | |
| { | |
| "name": "video_crossfade_frames", | |
| "shape": 7, | |
| "type": "INT", | |
| "widget": { | |
| "name": "video_crossfade_frames" | |
| }, | |
| "link": 231 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 80, | |
| 94 | |
| ] | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 81, | |
| 85, | |
| 203 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 83 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_frames", | |
| "type": "INT", | |
| "links": [ | |
| 232 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 4 β TRIM AUDIO / PREP VIDEO OVERLAP + MATCH AUDIO TAIL", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContextTrim" | |
| }, | |
| "widgets_values": [ | |
| 0, | |
| 24, | |
| true, | |
| 0 | |
| ], | |
| "widgets_values_named": { | |
| "trim_frames": 0, | |
| "fps": 24, | |
| "match_tail": true, | |
| "video_crossfade_frames": 0 | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 330, | |
| "type": "ImageBatchExtendWithOverlap", | |
| "pos": [ | |
| 3147.675342760168, | |
| 256.80336279647827 | |
| ], | |
| "size": [ | |
| 379.0376953125, | |
| 155 | |
| ], | |
| "flags": {}, | |
| "order": 79, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "link": 82 | |
| }, | |
| { | |
| "name": "new_images", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 83 | |
| }, | |
| { | |
| "name": "overlap", | |
| "type": "INT", | |
| "widget": { | |
| "name": "overlap" | |
| }, | |
| "link": 232 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "start_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "extended_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 113 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "073efb07419f56cc714e099a82e49fbc23ad9263", | |
| "Node name for S&R": "ImageBatchExtendWithOverlap" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "source", | |
| "linear_blend" | |
| ], | |
| "widgets_values_named": { | |
| "overlap": 39, | |
| "overlap_side": "source", | |
| "overlap_mode": "linear_blend" | |
| } | |
| }, | |
| { | |
| "id": 331, | |
| "type": "AudioConcat", | |
| "pos": [ | |
| 3199.282002153125, | |
| 471.5555471566468 | |
| ], | |
| "size": [ | |
| 280, | |
| 120 | |
| ], | |
| "flags": {}, | |
| "order": 77, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "audio1", | |
| "type": "AUDIO", | |
| "link": 84 | |
| }, | |
| { | |
| "name": "audio2", | |
| "type": "AUDIO", | |
| "link": 85 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 115 | |
| ] | |
| } | |
| ], | |
| "title": "CUMULATIVE AUDIO through Clip 4", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "AudioConcat" | |
| }, | |
| "widgets_values": [ | |
| "after" | |
| ], | |
| "widgets_values_named": { | |
| "direction": "after" | |
| } | |
| }, | |
| { | |
| "id": 400, | |
| "type": "MiniMaxH3ReferenceToVideo", | |
| "pos": [ | |
| 1192.8376713800847, | |
| 805.1964996398316 | |
| ], | |
| "size": [ | |
| 480, | |
| 430 | |
| ], | |
| "flags": {}, | |
| "order": 80, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "clip", | |
| "type": "CLIP", | |
| "link": 86 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 87 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "type": "VAE", | |
| "link": 190 | |
| }, | |
| { | |
| "label": "ref_image_0", | |
| "name": "ref_images.ref_image_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 191 | |
| }, | |
| { | |
| "label": "ref_image_1", | |
| "name": "ref_images.ref_image_1", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 192 | |
| }, | |
| { | |
| "label": "ref_image_2", | |
| "name": "ref_images.ref_image_2", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_0", | |
| "name": "ref_videos.ref_video_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_audio_0", | |
| "name": "ref_video_audios.ref_video_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_audio_0", | |
| "name": "ref_audios.ref_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 204 | |
| }, | |
| { | |
| "label": "ref_audio_1", | |
| "name": "ref_audios.ref_audio_1", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "widget": { | |
| "name": "width" | |
| }, | |
| "link": 274 | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "widget": { | |
| "name": "height" | |
| }, | |
| "link": 275 | |
| }, | |
| { | |
| "name": "length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "length" | |
| }, | |
| "link": 90 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "positive", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 91 | |
| ] | |
| }, | |
| { | |
| "name": "LATENT", | |
| "type": "LATENT", | |
| "links": [ | |
| 93, | |
| 103 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 5 β CONTINUATION + GLOBAL REFS", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "MiniMaxH3ReferenceToVideo" | |
| }, | |
| "widgets_values": [ | |
| "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| 960, | |
| 544, | |
| 362, | |
| "match" | |
| ], | |
| "widgets_values_named": { | |
| "prompt": "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| "width": 960, | |
| "height": 544, | |
| "length": 362, | |
| "ref_image_size": "match" | |
| }, | |
| "color": "#346434", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 401, | |
| "type": "MiniMaxH3MotionContext", | |
| "pos": [ | |
| 1686.854697419916, | |
| 811.6923026000001 | |
| ], | |
| "size": [ | |
| 489.0689453125, | |
| 350 | |
| ], | |
| "flags": {}, | |
| "order": 81, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 91 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 92 | |
| }, | |
| { | |
| "name": "latent", | |
| "type": "LATENT", | |
| "link": 93 | |
| }, | |
| { | |
| "name": "context_frames", | |
| "type": "IMAGE", | |
| "link": 94 | |
| }, | |
| { | |
| "name": "context_latent", | |
| "shape": 7, | |
| "type": "LATENT", | |
| "link": 95 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "context_length" | |
| }, | |
| "link": 223 | |
| }, | |
| { | |
| "name": "audio_context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "audio_context_length" | |
| }, | |
| "link": 224 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 97 | |
| ] | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "links": [ | |
| 110 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 5 Motion Context β GLOBAL / video / head / GLOBAL / timeline", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContext" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "video", | |
| "head", | |
| "disabled", | |
| 39, | |
| "timeline" | |
| ], | |
| "widgets_values_named": { | |
| "context_length": 39, | |
| "encode_mode": "video", | |
| "anchor_mode": "head", | |
| "crop": "disabled", | |
| "audio_context_length": 39, | |
| "audio_mode": "timeline" | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 414, | |
| "type": "SamplerCustomAdvanced", | |
| "pos": [ | |
| 2485.4359212000013, | |
| 875.1964996398316 | |
| ], | |
| "size": [ | |
| 300, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 83, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "noise", | |
| "type": "NOISE", | |
| "link": 98 | |
| }, | |
| { | |
| "name": "guider", | |
| "type": "GUIDER", | |
| "link": 99 | |
| }, | |
| { | |
| "name": "sampler", | |
| "type": "SAMPLER", | |
| "link": 100 | |
| }, | |
| { | |
| "name": "sigmas", | |
| "type": "SIGMAS", | |
| "link": 102 | |
| }, | |
| { | |
| "name": "latent_image", | |
| "type": "LATENT", | |
| "link": 103 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "output", | |
| "type": "LATENT", | |
| "links": [ | |
| 104, | |
| 106, | |
| 126 | |
| ] | |
| }, | |
| { | |
| "name": "denoised_output", | |
| "type": "LATENT", | |
| "links": [] | |
| } | |
| ], | |
| "title": "Clip 5 Sampler", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "SamplerCustomAdvanced" | |
| } | |
| }, | |
| { | |
| "id": 420, | |
| "type": "VAEDecode", | |
| "pos": [ | |
| 2790.2394215601685, | |
| 834.8032624796631 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 84, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 104 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 105 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 108 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 5 Frames", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecode" | |
| } | |
| }, | |
| { | |
| "id": 421, | |
| "type": "VAEDecodeAudio", | |
| "pos": [ | |
| 2792.8375524398316, | |
| 963.1878676796631 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 85, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 106 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 107 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 109 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 5 Audio", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecodeAudio" | |
| } | |
| }, | |
| { | |
| "id": 422, | |
| "type": "MiniMaxH3MotionContextTrim", | |
| "pos": [ | |
| 3043.1282828576705, | |
| 839.435917628835 | |
| ], | |
| "size": [ | |
| 499.8765625, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 86, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 108 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 109 | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "widget": { | |
| "name": "trim_frames" | |
| }, | |
| "link": 110 | |
| }, | |
| { | |
| "name": "video_crossfade_frames", | |
| "shape": 7, | |
| "type": "INT", | |
| "widget": { | |
| "name": "video_crossfade_frames" | |
| }, | |
| "link": 233 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 111, | |
| 125 | |
| ] | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 112, | |
| 116, | |
| 205 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 114 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_frames", | |
| "type": "INT", | |
| "links": [ | |
| 234 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 5 β TRIM AUDIO / PREP VIDEO OVERLAP + MATCH AUDIO TAIL", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContextTrim" | |
| }, | |
| "widgets_values": [ | |
| 0, | |
| 24, | |
| true, | |
| 0 | |
| ], | |
| "widgets_values_named": { | |
| "trim_frames": 0, | |
| "fps": 24, | |
| "match_tail": true, | |
| "video_crossfade_frames": 0 | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 430, | |
| "type": "ImageBatchExtendWithOverlap", | |
| "pos": [ | |
| 3135.1592511307404, | |
| 1102.9697190363634 | |
| ], | |
| "size": [ | |
| 379.4341796875, | |
| 155 | |
| ], | |
| "flags": {}, | |
| "order": 90, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "link": 113 | |
| }, | |
| { | |
| "name": "new_images", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 114 | |
| }, | |
| { | |
| "name": "overlap", | |
| "type": "INT", | |
| "widget": { | |
| "name": "overlap" | |
| }, | |
| "link": 234 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "start_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "extended_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 144 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "073efb07419f56cc714e099a82e49fbc23ad9263", | |
| "Node name for S&R": "ImageBatchExtendWithOverlap" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "source", | |
| "linear_blend" | |
| ], | |
| "widgets_values_named": { | |
| "overlap": 39, | |
| "overlap_side": "source", | |
| "overlap_mode": "linear_blend" | |
| } | |
| }, | |
| { | |
| "id": 431, | |
| "type": "AudioConcat", | |
| "pos": [ | |
| 3215.3550634611383, | |
| 1311.7374430195728 | |
| ], | |
| "size": [ | |
| 280, | |
| 120 | |
| ], | |
| "flags": {}, | |
| "order": 88, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "audio1", | |
| "type": "AUDIO", | |
| "link": 115 | |
| }, | |
| { | |
| "name": "audio2", | |
| "type": "AUDIO", | |
| "link": 116 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 146 | |
| ] | |
| } | |
| ], | |
| "title": "CUMULATIVE AUDIO through Clip 5", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "AudioConcat" | |
| }, | |
| "widgets_values": [ | |
| "after" | |
| ], | |
| "widgets_values_named": { | |
| "direction": "after" | |
| } | |
| }, | |
| { | |
| "id": 500, | |
| "type": "MiniMaxH3ReferenceToVideo", | |
| "pos": [ | |
| 1203.2307896000004, | |
| 1644.8035003601685 | |
| ], | |
| "size": [ | |
| 480, | |
| 430 | |
| ], | |
| "flags": {}, | |
| "order": 91, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "clip", | |
| "type": "CLIP", | |
| "link": 117 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 118 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "type": "VAE", | |
| "link": 193 | |
| }, | |
| { | |
| "label": "ref_image_0", | |
| "name": "ref_images.ref_image_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 194 | |
| }, | |
| { | |
| "label": "ref_image_1", | |
| "name": "ref_images.ref_image_1", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 195 | |
| }, | |
| { | |
| "label": "ref_image_2", | |
| "name": "ref_images.ref_image_2", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_0", | |
| "name": "ref_videos.ref_video_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_audio_0", | |
| "name": "ref_video_audios.ref_video_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_audio_0", | |
| "name": "ref_audios.ref_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 206 | |
| }, | |
| { | |
| "label": "ref_audio_1", | |
| "name": "ref_audios.ref_audio_1", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "widget": { | |
| "name": "width" | |
| }, | |
| "link": 276 | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "widget": { | |
| "name": "height" | |
| }, | |
| "link": 277 | |
| }, | |
| { | |
| "name": "length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "length" | |
| }, | |
| "link": 121 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "positive", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 122 | |
| ] | |
| }, | |
| { | |
| "name": "LATENT", | |
| "type": "LATENT", | |
| "links": [ | |
| 124, | |
| 134 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 6 β CONTINUATION + GLOBAL REFS", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "MiniMaxH3ReferenceToVideo" | |
| }, | |
| "widgets_values": [ | |
| "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| 960, | |
| 544, | |
| 362, | |
| "match" | |
| ], | |
| "widgets_values_named": { | |
| "prompt": "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| "width": 960, | |
| "height": 544, | |
| "length": 362, | |
| "ref_image_size": "match" | |
| }, | |
| "color": "#346434", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 501, | |
| "type": "MiniMaxH3MotionContext", | |
| "pos": [ | |
| 1699.8461844000008, | |
| 1665.5897367999996 | |
| ], | |
| "size": [ | |
| 488.706640625, | |
| 350 | |
| ], | |
| "flags": {}, | |
| "order": 92, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 122 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 123 | |
| }, | |
| { | |
| "name": "latent", | |
| "type": "LATENT", | |
| "link": 124 | |
| }, | |
| { | |
| "name": "context_frames", | |
| "type": "IMAGE", | |
| "link": 125 | |
| }, | |
| { | |
| "name": "context_latent", | |
| "shape": 7, | |
| "type": "LATENT", | |
| "link": 126 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "context_length" | |
| }, | |
| "link": 225 | |
| }, | |
| { | |
| "name": "audio_context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "audio_context_length" | |
| }, | |
| "link": 226 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 128 | |
| ] | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "links": [ | |
| 141 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 6 Motion Context β GLOBAL / video / head / GLOBAL / timeline", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContext" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "video", | |
| "head", | |
| "disabled", | |
| 39, | |
| "timeline" | |
| ], | |
| "widgets_values_named": { | |
| "context_length": 39, | |
| "encode_mode": "video", | |
| "anchor_mode": "head", | |
| "crop": "disabled", | |
| "audio_context_length": 39, | |
| "audio_mode": "timeline" | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 514, | |
| "type": "SamplerCustomAdvanced", | |
| "pos": [ | |
| 2488.034289960169, | |
| 1718.7008156199158 | |
| ], | |
| "size": [ | |
| 300, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 94, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "noise", | |
| "type": "NOISE", | |
| "link": 129 | |
| }, | |
| { | |
| "name": "guider", | |
| "type": "GUIDER", | |
| "link": 130 | |
| }, | |
| { | |
| "name": "sampler", | |
| "type": "SAMPLER", | |
| "link": 131 | |
| }, | |
| { | |
| "name": "sigmas", | |
| "type": "SIGMAS", | |
| "link": 133 | |
| }, | |
| { | |
| "name": "latent_image", | |
| "type": "LATENT", | |
| "link": 134 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "output", | |
| "type": "LATENT", | |
| "links": [ | |
| 135, | |
| 137, | |
| 157 | |
| ] | |
| }, | |
| { | |
| "name": "denoised_output", | |
| "type": "LATENT", | |
| "links": [] | |
| } | |
| ], | |
| "title": "Clip 6 Sampler", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "SamplerCustomAdvanced" | |
| } | |
| }, | |
| { | |
| "id": 520, | |
| "type": "VAEDecode", | |
| "pos": [ | |
| 2803.2307896000007, | |
| 1693.8974342 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 95, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 135 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 136 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 139 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 6 Frames", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecode" | |
| } | |
| }, | |
| { | |
| "id": 521, | |
| "type": "VAEDecodeAudio", | |
| "pos": [ | |
| 2804.5298550398325, | |
| 1814.4871709999998 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 96, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 137 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 138 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 140 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 6 Audio", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecodeAudio" | |
| } | |
| }, | |
| { | |
| "id": 522, | |
| "type": "MiniMaxH3MotionContextTrim", | |
| "pos": [ | |
| 3063.470211160169, | |
| 1698.7008156199158 | |
| ], | |
| "size": [ | |
| 499.5142578125, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 97, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 139 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 140 | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "widget": { | |
| "name": "trim_frames" | |
| }, | |
| "link": 141 | |
| }, | |
| { | |
| "name": "video_crossfade_frames", | |
| "shape": 7, | |
| "type": "INT", | |
| "widget": { | |
| "name": "video_crossfade_frames" | |
| }, | |
| "link": 235 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 142, | |
| 156 | |
| ] | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 143, | |
| 147, | |
| 207 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 145 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_frames", | |
| "type": "INT", | |
| "links": [ | |
| 236 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 6 β TRIM AUDIO / PREP VIDEO OVERLAP + MATCH AUDIO TAIL", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContextTrim" | |
| }, | |
| "widgets_values": [ | |
| 0, | |
| 24, | |
| true, | |
| 0 | |
| ], | |
| "widgets_values_named": { | |
| "trim_frames": 0, | |
| "fps": 24, | |
| "match_tail": true, | |
| "video_crossfade_frames": 0 | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 523, | |
| "type": "VHS_VideoCombine", | |
| "pos": [ | |
| 3581.692302599999, | |
| 1660 | |
| ], | |
| "size": [ | |
| 420, | |
| 334 | |
| ], | |
| "flags": {}, | |
| "order": 98, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 142 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 143 | |
| }, | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "Filenames", | |
| "type": "VHS_FILENAMES", | |
| "links": null | |
| } | |
| ], | |
| "title": "CLIP 6 PREVIEW β TEMP ONLY", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "1.7.7", | |
| "Node name for S&R": "VHS_VideoCombine" | |
| }, | |
| "widgets_values": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip06", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip06", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| } | |
| }, | |
| { | |
| "id": 530, | |
| "type": "ImageBatchExtendWithOverlap", | |
| "pos": [ | |
| 3173.2776240398307, | |
| 1951.3732623206722 | |
| ], | |
| "size": [ | |
| 378.6958984375, | |
| 155 | |
| ], | |
| "flags": {}, | |
| "order": 101, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "link": 144 | |
| }, | |
| { | |
| "name": "new_images", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 145 | |
| }, | |
| { | |
| "name": "overlap", | |
| "type": "INT", | |
| "widget": { | |
| "name": "overlap" | |
| }, | |
| "link": 236 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "start_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "extended_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 175 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "073efb07419f56cc714e099a82e49fbc23ad9263", | |
| "Node name for S&R": "ImageBatchExtendWithOverlap" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "source", | |
| "linear_blend" | |
| ], | |
| "widgets_values_named": { | |
| "overlap": 39, | |
| "overlap_side": "source", | |
| "overlap_mode": "linear_blend" | |
| } | |
| }, | |
| { | |
| "id": 600, | |
| "type": "MiniMaxH3ReferenceToVideo", | |
| "pos": [ | |
| 1207.1282238000003, | |
| 2500 | |
| ], | |
| "size": [ | |
| 480, | |
| 430 | |
| ], | |
| "flags": {}, | |
| "order": 102, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "clip", | |
| "type": "CLIP", | |
| "link": 148 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 149 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "type": "VAE", | |
| "link": 196 | |
| }, | |
| { | |
| "label": "ref_image_0", | |
| "name": "ref_images.ref_image_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 197 | |
| }, | |
| { | |
| "label": "ref_image_1", | |
| "name": "ref_images.ref_image_1", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 198 | |
| }, | |
| { | |
| "label": "ref_image_2", | |
| "name": "ref_images.ref_image_2", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_0", | |
| "name": "ref_videos.ref_video_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_audio_0", | |
| "name": "ref_video_audios.ref_video_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_audio_0", | |
| "name": "ref_audios.ref_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 208 | |
| }, | |
| { | |
| "label": "ref_audio_1", | |
| "name": "ref_audios.ref_audio_1", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "widget": { | |
| "name": "width" | |
| }, | |
| "link": 278 | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "widget": { | |
| "name": "height" | |
| }, | |
| "link": 279 | |
| }, | |
| { | |
| "name": "length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "length" | |
| }, | |
| "link": 152 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "positive", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 153 | |
| ] | |
| }, | |
| { | |
| "name": "LATENT", | |
| "type": "LATENT", | |
| "links": [ | |
| 155, | |
| 165 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 7 β CONTINUATION + GLOBAL REFS", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "MiniMaxH3ReferenceToVideo" | |
| }, | |
| "widgets_values": [ | |
| "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| 960, | |
| 544, | |
| 362, | |
| "match" | |
| ], | |
| "widgets_values_named": { | |
| "prompt": "subject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the two character references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, or background. Match the incoming visual state from the carried motion context and do not visibly re-resolve or beautify the face right at the continuation boundary.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming visual state, lighting, facial texture, sharpness, and rendering character from the carried motion context. [DESCRIBE WHAT HAPPENS NEXT; DO NOT RESTART FROM REST. If <Subject 1> re-enters after being off-screen, restore the same referenced identity naturally.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. If the optional full previous-clip audio reference is enabled, use it only as a broad sonic-identity reference; do not replay, restart, or copy its temporal progression. [DESCRIBE NEW SOUND CHANGES CAUSED BY THE CONTINUING ACTION.]\n\nnon_diegetic_music:\nN/A", | |
| "width": 960, | |
| "height": 544, | |
| "length": 362, | |
| "ref_image_size": "match" | |
| }, | |
| "color": "#346434", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 601, | |
| "type": "MiniMaxH3MotionContext", | |
| "pos": [ | |
| 1703.7436185999982, | |
| 2509.0939338398307 | |
| ], | |
| "size": [ | |
| 488.706640625, | |
| 350 | |
| ], | |
| "flags": {}, | |
| "order": 103, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 153 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 154 | |
| }, | |
| { | |
| "name": "latent", | |
| "type": "LATENT", | |
| "link": 155 | |
| }, | |
| { | |
| "name": "context_frames", | |
| "type": "IMAGE", | |
| "link": 156 | |
| }, | |
| { | |
| "name": "context_latent", | |
| "shape": 7, | |
| "type": "LATENT", | |
| "link": 157 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "context_length" | |
| }, | |
| "link": 227 | |
| }, | |
| { | |
| "name": "audio_context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "audio_context_length" | |
| }, | |
| "link": 228 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 159 | |
| ] | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "links": [ | |
| 172 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 7 Motion Context β GLOBAL / video / head / GLOBAL / timeline", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContext" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "video", | |
| "head", | |
| "disabled", | |
| 39, | |
| "timeline" | |
| ], | |
| "widgets_values_named": { | |
| "context_length": 39, | |
| "encode_mode": "video", | |
| "anchor_mode": "head", | |
| "crop": "disabled", | |
| "audio_context_length": 39, | |
| "audio_mode": "timeline" | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 614, | |
| "type": "SamplerCustomAdvanced", | |
| "pos": [ | |
| 2484.13685576017, | |
| 2573.8974341999997 | |
| ], | |
| "size": [ | |
| 300, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 105, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "noise", | |
| "type": "NOISE", | |
| "link": 160 | |
| }, | |
| { | |
| "name": "guider", | |
| "type": "GUIDER", | |
| "link": 161 | |
| }, | |
| { | |
| "name": "sampler", | |
| "type": "SAMPLER", | |
| "link": 162 | |
| }, | |
| { | |
| "name": "sigmas", | |
| "type": "SIGMAS", | |
| "link": 164 | |
| }, | |
| { | |
| "name": "latent_image", | |
| "type": "LATENT", | |
| "link": 165 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "output", | |
| "type": "LATENT", | |
| "links": [ | |
| 166, | |
| 168 | |
| ] | |
| }, | |
| { | |
| "name": "denoised_output", | |
| "type": "LATENT", | |
| "links": [] | |
| } | |
| ], | |
| "title": "Clip 7 Sampler", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "SamplerCustomAdvanced" | |
| } | |
| }, | |
| { | |
| "id": 620, | |
| "type": "VAEDecode", | |
| "pos": [ | |
| 2807.128223800001, | |
| 2541.2990654398313 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 106, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 166 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 167 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 170 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 7 Frames", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecode" | |
| } | |
| }, | |
| { | |
| "id": 621, | |
| "type": "VAEDecodeAudio", | |
| "pos": [ | |
| 2808.4272892398317, | |
| 2664.4871709999998 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 107, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 168 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 169 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 171 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 7 Audio", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecodeAudio" | |
| } | |
| }, | |
| { | |
| "id": 622, | |
| "type": "MiniMaxH3MotionContextTrim", | |
| "pos": [ | |
| 3061.419639918825, | |
| 2545.4799145601683 | |
| ], | |
| "size": [ | |
| 499.5142578125, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 108, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 170 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 171 | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "widget": { | |
| "name": "trim_frames" | |
| }, | |
| "link": 172 | |
| }, | |
| { | |
| "name": "video_crossfade_frames", | |
| "shape": 7, | |
| "type": "INT", | |
| "widget": { | |
| "name": "video_crossfade_frames" | |
| }, | |
| "link": 237 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 173 | |
| ] | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 174, | |
| 178 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 176 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_frames", | |
| "type": "INT", | |
| "links": [ | |
| 238 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 7 β TRIM AUDIO / PREP VIDEO OVERLAP + MATCH AUDIO TAIL", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContextTrim" | |
| }, | |
| "widgets_values": [ | |
| 0, | |
| 24, | |
| true, | |
| 0 | |
| ], | |
| "widgets_values_named": { | |
| "trim_frames": 0, | |
| "fps": 24, | |
| "match_tail": true, | |
| "video_crossfade_frames": 0 | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 631, | |
| "type": "AudioConcat", | |
| "pos": [ | |
| 3221.1204422413452, | |
| 3014.181723198485 | |
| ], | |
| "size": [ | |
| 280, | |
| 120 | |
| ], | |
| "flags": {}, | |
| "order": 110, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "audio1", | |
| "type": "AUDIO", | |
| "link": 177 | |
| }, | |
| { | |
| "name": "audio2", | |
| "type": "AUDIO", | |
| "link": 178 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 180 | |
| ] | |
| } | |
| ], | |
| "title": "CUMULATIVE AUDIO through Clip 7", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "AudioConcat" | |
| }, | |
| "widgets_values": [ | |
| "after" | |
| ], | |
| "widgets_values_named": { | |
| "direction": "after" | |
| } | |
| }, | |
| { | |
| "id": 630, | |
| "type": "ImageBatchExtendWithOverlap", | |
| "pos": [ | |
| 3142.892972118824, | |
| 2810.016575777057 | |
| ], | |
| "size": [ | |
| 379.071875, | |
| 155 | |
| ], | |
| "flags": {}, | |
| "order": 111, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "link": 175 | |
| }, | |
| { | |
| "name": "new_images", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 176 | |
| }, | |
| { | |
| "name": "overlap", | |
| "type": "INT", | |
| "widget": { | |
| "name": "overlap" | |
| }, | |
| "link": 238 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "start_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "extended_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 179 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "073efb07419f56cc714e099a82e49fbc23ad9263", | |
| "Node name for S&R": "ImageBatchExtendWithOverlap" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "source", | |
| "linear_blend" | |
| ], | |
| "widgets_values_named": { | |
| "overlap": 39, | |
| "overlap_side": "source", | |
| "overlap_mode": "linear_blend" | |
| } | |
| }, | |
| { | |
| "id": 623, | |
| "type": "VHS_VideoCombine", | |
| "pos": [ | |
| 3577.7948683999994, | |
| 2510 | |
| ], | |
| "size": [ | |
| 420, | |
| 334 | |
| ], | |
| "flags": {}, | |
| "order": 109, | |
| "mode": 4, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 173 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 174 | |
| }, | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "Filenames", | |
| "type": "VHS_FILENAMES", | |
| "links": null | |
| } | |
| ], | |
| "title": "CLIP 7 PREVIEW β TEMP ONLY", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "1.7.7", | |
| "Node name for S&R": "VHS_VideoCombine" | |
| }, | |
| "widgets_values": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip07", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip07", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": {} | |
| } | |
| } | |
| }, | |
| { | |
| "id": 941, | |
| "type": "Note", | |
| "pos": [ | |
| -3221.884523671377, | |
| -294.35379251609504 | |
| ], | |
| "size": [ | |
| 294.3204915780452, | |
| 88 | |
| ], | |
| "flags": {}, | |
| "order": 10, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [], | |
| "title": "Context length guide", | |
| "properties": { | |
| "Node name for S&R": "Note" | |
| }, | |
| "widgets_values": [ | |
| "H3-native video runs: 5, 22, 39*, 56, 73, 90*, 107, 124, 141*, 158, 175, 192*, 209, 226, 243*\n\n* exact video+audio boundary. 39 recommended." | |
| ], | |
| "widgets_values_named": { | |
| "text": "H3-native video runs: 5, 22, 39*, 56, 73, 90*, 107, 124, 141*, 158, 175, 192*, 209, 226, 243*\n\n* exact video+audio boundary. 39 recommended." | |
| }, | |
| "color": "#432", | |
| "bgcolor": "#653" | |
| }, | |
| { | |
| "id": 914, | |
| "type": "Note", | |
| "pos": [ | |
| -970, | |
| 2330 | |
| ], | |
| "size": [ | |
| 510.80355041357245, | |
| 88 | |
| ], | |
| "flags": {}, | |
| "order": 11, | |
| "mode": 2, | |
| "inputs": [], | |
| "outputs": [], | |
| "title": "OPTIONAL AUDIO REF SIGNAL", | |
| "properties": {}, | |
| "widgets_values": [ | |
| "OFF: muted Any Switch outputs None β Ref2VA receives no full-audio reference.\nON: each switch passes the previous clip's entire delivered audio into ref_audio_0." | |
| ], | |
| "widgets_values_named": { | |
| "text": "OFF: muted Any Switch outputs None β Ref2VA receives no full-audio reference.\nON: each switch passes the previous clip's entire delivered audio into ref_audio_0." | |
| }, | |
| "color": "#432", | |
| "bgcolor": "#653" | |
| }, | |
| { | |
| "id": 913, | |
| "type": "Note", | |
| "pos": [ | |
| -2877.8787710646952, | |
| 1318.4297520661146 | |
| ], | |
| "size": [ | |
| 824.683212406379, | |
| 129.83471074380168 | |
| ], | |
| "flags": {}, | |
| "order": 12, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [], | |
| "title": "β DISCLAIMER β FULL AUDIO REF IS EXPERIMENTAL", | |
| "properties": {}, | |
| "widgets_values": [ | |
| "OFF BY DEFAULT β ENABLE ONLY FOR EXPERIMENTS.\n\nThis feeds the ENTIRE audio of the previous delivered clip into the next Ref2VA clip as an ordinary audio reference.\n\nβ In our test this could BREAK AUDIO CONTINUATION: the music restarted/replayed from an earlier point and produced a noticeable seam.\n\nThe proven 39-frame timeline-audio continuation remains active regardless. If enabling this makes the join worse, turn this option OFF again." | |
| ], | |
| "widgets_values_named": { | |
| "text": "OFF BY DEFAULT β ENABLE ONLY FOR EXPERIMENTS.\n\nThis feeds the ENTIRE audio of the previous delivered clip into the next Ref2VA clip as an ordinary audio reference.\n\nβ In our test this could BREAK AUDIO CONTINUATION: the music restarted/replayed from an earlier point and produced a noticeable seam.\n\nThe proven 39-frame timeline-audio continuation remains active regardless. If enabling this makes the join worse, turn this option OFF again." | |
| }, | |
| "color": "#432", | |
| "bgcolor": "#653" | |
| }, | |
| { | |
| "id": 4, | |
| "type": "VAELoader", | |
| "pos": [ | |
| -2007.2820394000005, | |
| -36.73510558008424 | |
| ], | |
| "size": [ | |
| 330, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 13, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "VAE", | |
| "type": "VAE", | |
| "links": [ | |
| 19, | |
| 45, | |
| 76, | |
| 107, | |
| 138, | |
| 169, | |
| 181, | |
| 184, | |
| 187, | |
| 190, | |
| 193, | |
| 196, | |
| 244 | |
| ] | |
| } | |
| ], | |
| "title": "H3 Audio VAE", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAELoader" | |
| }, | |
| "widgets_values": [ | |
| "minimax_h3_audio_vae_fp32.safetensors" | |
| ], | |
| "widgets_values_named": { | |
| "vae_name": "minimax_h3_audio_vae_fp32.safetensors" | |
| } | |
| }, | |
| { | |
| "id": 3, | |
| "type": "VAELoader", | |
| "pos": [ | |
| -2007.2820393999998, | |
| -145.43592120000005 | |
| ], | |
| "size": [ | |
| 330, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 14, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "VAE", | |
| "type": "VAE", | |
| "links": [ | |
| 4, | |
| 17, | |
| 25, | |
| 30, | |
| 43, | |
| 56, | |
| 61, | |
| 74, | |
| 87, | |
| 92, | |
| 105, | |
| 118, | |
| 123, | |
| 136, | |
| 149, | |
| 154, | |
| 167, | |
| 243 | |
| ] | |
| } | |
| ], | |
| "title": "H3 Video VAE", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAELoader" | |
| }, | |
| "widgets_values": [ | |
| "minimax_h3_video_vae_int8_convrot.safetensors" | |
| ], | |
| "widgets_values_named": { | |
| "vae_name": "minimax_h3_video_vae_int8_convrot.safetensors" | |
| } | |
| }, | |
| { | |
| "id": 1, | |
| "type": "UNETLoader", | |
| "pos": [ | |
| -2007.2818015194948, | |
| -422.8377903203366 | |
| ], | |
| "size": [ | |
| 330, | |
| 90 | |
| ], | |
| "flags": {}, | |
| "order": 15, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "MODEL", | |
| "type": "MODEL", | |
| "links": [ | |
| 2 | |
| ] | |
| } | |
| ], | |
| "title": "H3 REF2VA MODEL β REQUIRED", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "UNETLoader" | |
| }, | |
| "widgets_values": [ | |
| "minimax_h3_ref2va_pruned_int8_convrot.safetensors", | |
| "default" | |
| ], | |
| "widgets_values_named": { | |
| "unet_name": "minimax_h3_ref2va_pruned_int8_convrot.safetensors", | |
| "weight_dtype": "default" | |
| } | |
| }, | |
| { | |
| "id": 210, | |
| "type": "RandomNoise", | |
| "pos": [ | |
| 2207.259999999999, | |
| -939.1933456420899 | |
| ], | |
| "size": [ | |
| 270, | |
| 90 | |
| ], | |
| "flags": {}, | |
| "order": 16, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "NOISE", | |
| "type": "NOISE", | |
| "links": [ | |
| 36 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "RandomNoise" | |
| }, | |
| "widgets_values": [ | |
| 7679011, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "noise_seed": 7679011, | |
| "control_after_generate": "fixed" | |
| } | |
| }, | |
| { | |
| "id": 310, | |
| "type": "RandomNoise", | |
| "pos": [ | |
| 2203.549279174804, | |
| -95.32400000000001 | |
| ], | |
| "size": [ | |
| 270, | |
| 90 | |
| ], | |
| "flags": {}, | |
| "order": 17, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "NOISE", | |
| "type": "NOISE", | |
| "links": [ | |
| 67 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "RandomNoise" | |
| }, | |
| "widgets_values": [ | |
| 7691356, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "noise_seed": 7691356, | |
| "control_after_generate": "fixed" | |
| } | |
| }, | |
| { | |
| "id": 410, | |
| "type": "RandomNoise", | |
| "pos": [ | |
| 2200, | |
| 760 | |
| ], | |
| "size": [ | |
| 270, | |
| 90 | |
| ], | |
| "flags": {}, | |
| "order": 18, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "NOISE", | |
| "type": "NOISE", | |
| "links": [ | |
| 98 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "RandomNoise" | |
| }, | |
| "widgets_values": [ | |
| 7703701, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "noise_seed": 7703701, | |
| "control_after_generate": "fixed" | |
| } | |
| }, | |
| { | |
| "id": 510, | |
| "type": "RandomNoise", | |
| "pos": [ | |
| 2200, | |
| 1610 | |
| ], | |
| "size": [ | |
| 270, | |
| 90 | |
| ], | |
| "flags": {}, | |
| "order": 19, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "NOISE", | |
| "type": "NOISE", | |
| "links": [ | |
| 129 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "RandomNoise" | |
| }, | |
| "widgets_values": [ | |
| 7716046, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "noise_seed": 7716046, | |
| "control_after_generate": "fixed" | |
| } | |
| }, | |
| { | |
| "id": 610, | |
| "type": "RandomNoise", | |
| "pos": [ | |
| 2200, | |
| 2460 | |
| ], | |
| "size": [ | |
| 270, | |
| 90 | |
| ], | |
| "flags": {}, | |
| "order": 20, | |
| "mode": 4, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "NOISE", | |
| "type": "NOISE", | |
| "links": [ | |
| 160 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "RandomNoise" | |
| }, | |
| "widgets_values": [ | |
| 7728391, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "noise_seed": 7728391, | |
| "control_after_generate": "fixed" | |
| } | |
| }, | |
| { | |
| "id": 5, | |
| "type": "MiniMaxH3SigmaShift", | |
| "pos": [ | |
| -2461.6507612796663, | |
| -534.4275213276551 | |
| ], | |
| "size": [ | |
| 300, | |
| 150 | |
| ], | |
| "flags": {}, | |
| "order": 42, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 212 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "MODEL", | |
| "type": "MODEL", | |
| "links": [ | |
| 8, | |
| 13, | |
| 34, | |
| 39, | |
| 65, | |
| 70, | |
| 96, | |
| 101, | |
| 127, | |
| 132, | |
| 158, | |
| 163 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "MiniMaxH3SigmaShift" | |
| }, | |
| "widgets_values": [ | |
| 12, | |
| 3 | |
| ], | |
| "widgets_values_named": { | |
| "shift_video": 12, | |
| "shift_audio": 3 | |
| } | |
| }, | |
| { | |
| "id": 102, | |
| "type": "ComfyMathExpression", | |
| "pos": [ | |
| -2004.2803546796615, | |
| -977.0209187024606 | |
| ], | |
| "size": [ | |
| 350, | |
| 150 | |
| ], | |
| "flags": {}, | |
| "order": 35, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "values.a", | |
| "type": "FLOAT,INT,BOOLEAN", | |
| "link": 1 | |
| }, | |
| { | |
| "label": "b", | |
| "name": "values.b", | |
| "shape": 7, | |
| "type": "FLOAT,INT,BOOLEAN", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "FLOAT", | |
| "type": "FLOAT", | |
| "links": null | |
| }, | |
| { | |
| "name": "INT", | |
| "type": "INT", | |
| "links": [ | |
| 7, | |
| 28, | |
| 59, | |
| 90, | |
| 121, | |
| 152 | |
| ] | |
| }, | |
| { | |
| "name": "BOOL", | |
| "type": "BOOLEAN", | |
| "links": null | |
| } | |
| ], | |
| "title": "H3 VALID FRAME LENGTH (17k+5)", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "ComfyMathExpression" | |
| }, | |
| "widgets_values": [ | |
| "max(5, round(a * 24)) + (5 - (max(5, round(a * 24)) % 17)) % 17" | |
| ], | |
| "widgets_values_named": { | |
| "expression": "max(5, round(a * 24)) + (5 - (max(5, round(a * 24)) % 17)) % 17" | |
| } | |
| }, | |
| { | |
| "id": 122, | |
| "type": "KSamplerSelect", | |
| "pos": [ | |
| -559.8929578512391, | |
| -550.8029429752064 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 21, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "SAMPLER", | |
| "type": "SAMPLER", | |
| "links": [] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "KSamplerSelect" | |
| }, | |
| "widgets_values": [ | |
| "res_multistep" | |
| ], | |
| "widgets_values_named": { | |
| "sampler_name": "res_multistep" | |
| } | |
| }, | |
| { | |
| "id": 121, | |
| "type": "BasicGuider", | |
| "pos": [ | |
| -565.5910785944162, | |
| -660.412309182613 | |
| ], | |
| "size": [ | |
| 270, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 43, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 8 | |
| }, | |
| { | |
| "name": "conditioning", | |
| "type": "CONDITIONING", | |
| "link": 9 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "GUIDER", | |
| "type": "GUIDER", | |
| "links": [ | |
| 11 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicGuider" | |
| } | |
| }, | |
| { | |
| "id": 120, | |
| "type": "RandomNoise", | |
| "pos": [ | |
| -558.7997942775249, | |
| -793.2926354745405 | |
| ], | |
| "size": [ | |
| 270, | |
| 90 | |
| ], | |
| "flags": {}, | |
| "order": 22, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "NOISE", | |
| "type": "NOISE", | |
| "links": [ | |
| 10 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "RandomNoise" | |
| }, | |
| "widgets_values": [ | |
| 123456789, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "noise_seed": 123456789, | |
| "control_after_generate": "fixed" | |
| } | |
| }, | |
| { | |
| "id": 123, | |
| "type": "BasicScheduler", | |
| "pos": [ | |
| -585.715419163525, | |
| -444.4326776568955 | |
| ], | |
| "size": [ | |
| 270, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 44, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 13 | |
| }, | |
| { | |
| "name": "steps", | |
| "type": "INT", | |
| "widget": { | |
| "name": "steps" | |
| }, | |
| "link": 213 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "SIGMAS", | |
| "type": "SIGMAS", | |
| "links": [ | |
| 14 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "BasicScheduler" | |
| }, | |
| "widgets_values": [ | |
| "simple", | |
| 20, | |
| 1 | |
| ], | |
| "widgets_values_named": { | |
| "scheduler": "simple", | |
| "steps": 20, | |
| "denoise": 1 | |
| } | |
| }, | |
| { | |
| "id": 131, | |
| "type": "VAEDecodeAudio", | |
| "pos": [ | |
| -306.5805395874029, | |
| -403.89826247558585 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 52, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 18 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 19 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "AUDIO", | |
| "type": "AUDIO", | |
| "links": [ | |
| 21 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 2 Audio", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecodeAudio" | |
| } | |
| }, | |
| { | |
| "id": 105, | |
| "type": "ImageBatchExtendWithOverlap", | |
| "pos": [ | |
| -259.82717917480477, | |
| -844.5053791748041 | |
| ], | |
| "size": [ | |
| 378.9283203125, | |
| 155 | |
| ], | |
| "flags": {}, | |
| "order": 55, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "link": 267 | |
| }, | |
| { | |
| "name": "new_images", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 256 | |
| }, | |
| { | |
| "name": "overlap", | |
| "type": "INT", | |
| "widget": { | |
| "name": "overlap" | |
| }, | |
| "link": 257 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "source_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "start_images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "extended_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 258, | |
| 260, | |
| 261 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "073efb07419f56cc714e099a82e49fbc23ad9263", | |
| "Node name for S&R": "ImageBatchExtendWithOverlap" | |
| }, | |
| "widgets_values": [ | |
| 39, | |
| "source", | |
| "linear_blend" | |
| ], | |
| "widgets_values_named": { | |
| "overlap": 39, | |
| "overlap_side": "source", | |
| "overlap_mode": "linear_blend" | |
| } | |
| }, | |
| { | |
| "id": 912, | |
| "type": "Note", | |
| "pos": [ | |
| -1980.5509305591431, | |
| 640.9366223043648 | |
| ], | |
| "size": [ | |
| 736.4237415477087, | |
| 213.4710743801652 | |
| ], | |
| "flags": {}, | |
| "order": 23, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [], | |
| "title": "GLOBAL REFERENCE INSTRUCTIONS", | |
| "properties": {}, | |
| "widgets_values": [ | |
| "SELECT BOTH IMAGES BEFORE QUEUING.\n\nREF 1 = strongest face / identity image.\nREF 2 = full-body / wardrobe / proportions image of the SAME character.\n\nEvery generated clip receives both references as <Picture 1> and <Picture 2>.\n\nIf you want to add more references, or make per clip references, connect them to the appropriate ref_image_ slots and prompt them accordingly.\n\nRef image size defaults to MATCH for speed. Change per clip to MAX only if you need stronger identity fidelity." | |
| ], | |
| "widgets_values_named": { | |
| "text": "SELECT BOTH IMAGES BEFORE QUEUING.\n\nREF 1 = strongest face / identity image.\nREF 2 = full-body / wardrobe / proportions image of the SAME character.\n\nEvery generated clip receives both references as <Picture 1> and <Picture 2>.\n\nIf you want to add more references, or make per clip references, connect them to the appropriate ref_image_ slots and prompt them accordingly.\n\nRef image size defaults to MATCH for speed. Change per clip to MAX only if you need stronger identity fidelity." | |
| }, | |
| "color": "#432", | |
| "bgcolor": "#653" | |
| }, | |
| { | |
| "id": 900, | |
| "type": "Note", | |
| "pos": [ | |
| -3223.946250892765, | |
| -1279.1730919883262 | |
| ], | |
| "size": [ | |
| 667.6786267553812, | |
| 649.2907727847723 | |
| ], | |
| "flags": {}, | |
| "order": 24, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [], | |
| "title": "READ ME", | |
| "properties": { | |
| "Node name for S&R": "Note" | |
| }, | |
| "widgets_values": [ | |
| "MINIMAX H3 β ADVANCED EXTENSION OF INPUT VIDEOS\n\nACTIVE BY DEFAULT\nβ’ Clip 1 = uploaded source video β no prompt / no generation\nβ’ Clip 2 = first H3 extension of the input video using GLOBAL CHARACTER REFS\n\nSOURCE VIDEO / RESOLUTION\nβ’ Load one source input video in the SOURCE VIDEO node.\nβ’ The source is forced to 24 fps and cropped down to a multiple of 32.\nβ’ OPTIONAL CUSTOM RESOLUTION defaults to 0 Γ 0, which keeps the cropped source resolution.\nβ’ Enter a custom width and height there to resize the source before H3 conditioning and generate Clips 2β7 at that resolution.\nβ’ The resize output is kept divisible by 32 and the same chosen resolution is used by every generated clip.\n\nOPTIONAL\nβ’ Clips 3β7 are present but bypassed by default.\nβ’ Use the Fast Groups Bypasser panel to activate them SEQUENTIALLY:\n Clip 3 β Clip 4 β Clip 5 β Clip 6 β Clip 7.\nβ’ Activate them in order; do not activate a later clip unless every earlier continuation is active.\n\nGLOBAL CHARACTER REFS\nβ’ Select two images in the purple GLOBAL CHARACTER REFS group before queueing.\nβ’ Ref 1 = face / identity.\nβ’ Ref 2 = full body / wardrobe of the SAME character.\nβ’ Both references are wired into every generated clip as <Picture 1> and <Picture 2>.\nβ’ You can add more reference images by connecting them to the ref_image_ slots and prompting them.\n\nMODELS / PATCH\nβ’ Video model = minimax_h3_ref2va_pruned_int8_convrot.safetensors (you can also try the FL model)\nβ’ Text encoder = qwen3vl_32b_minimax_h3_int8_convrot.safetensors\n\nGLOBAL SETTINGS\nβ’ Raw clip duration, sampler, and step count are controlled from the Global Settings group.\nβ’ Resolution defaults to the cropped source input video; OPTIONAL CUSTOM RESOLUTION can override it for every generated clip.\nβ’ The global sampler and step controls feed all six generated continuation slots (Clips 2β7).\n\nGLOBAL DEFAULTS\nβ’ Duration = 15 sec raw H3 target per clip.\nβ’ Motion Context overlap = 39 frames (~1.625 sec), so each continuation\n contributes roughly 13.4 sec of NEW footage after accounting for the overlap.\nβ’ Audio timeline context = 39 frames.\nβ’ Resolution = source resolution by default (0 Γ 0), or the custom width/height selected in OPTIONAL CUSTOM RESOLUTION; output remains divisible by 32.\nβ’ Sampler/Scheduler: res_multistep/simple\n\nOPTIONAL FULL PREVIOUS AUDIO REF\nβ’ A separate red experimental section can feed each previous clip's ENTIRE delivered audio into the next Ref2VA clip.\nβ’ Use the 'FULL PREVIOUS AUDIO REF β TOGGLE HERE' Fast Groups Muter button.\nβ’ OFF by default.\nβ’ WARNING: this can break seamless audio continuation or make music restart/replay. The tested stable configuration is with this option OFF.\n\nSPEEDUPS\nβ’ The Speedups group matches the simple workflow layout and is BYPASSED by default." | |
| ], | |
| "widgets_values_named": { | |
| "text": "MINIMAX H3 β ADVANCED EXTENSION OF INPUT VIDEOS\n\nACTIVE BY DEFAULT\nβ’ Clip 1 = uploaded source video β no prompt / no generation\nβ’ Clip 2 = first H3 extension of the input video using GLOBAL CHARACTER REFS\n\nSOURCE VIDEO / RESOLUTION\nβ’ Load one source input video in the SOURCE VIDEO node.\nβ’ The source is forced to 24 fps and cropped down to a multiple of 32.\nβ’ OPTIONAL CUSTOM RESOLUTION defaults to 0 Γ 0, which keeps the cropped source resolution.\nβ’ Enter a custom width and height there to resize the source before H3 conditioning and generate Clips 2β7 at that resolution.\nβ’ The resize output is kept divisible by 32 and the same chosen resolution is used by every generated clip.\n\nOPTIONAL\nβ’ Clips 3β7 are present but bypassed by default.\nβ’ Use the Fast Groups Bypasser panel to activate them SEQUENTIALLY:\n Clip 3 β Clip 4 β Clip 5 β Clip 6 β Clip 7.\nβ’ Activate them in order; do not activate a later clip unless every earlier continuation is active.\n\nGLOBAL CHARACTER REFS\nβ’ Select two images in the purple GLOBAL CHARACTER REFS group before queueing.\nβ’ Ref 1 = face / identity.\nβ’ Ref 2 = full body / wardrobe of the SAME character.\nβ’ Both references are wired into every generated clip as <Picture 1> and <Picture 2>.\nβ’ You can add more reference images by connecting them to the ref_image_ slots and prompting them.\n\nMODELS / PATCH\nβ’ Video model = minimax_h3_ref2va_pruned_int8_convrot.safetensors (you can also try the FL model)\nβ’ Text encoder = qwen3vl_32b_minimax_h3_int8_convrot.safetensors\n\nGLOBAL SETTINGS\nβ’ Raw clip duration, sampler, and step count are controlled from the Global Settings group.\nβ’ Resolution defaults to the cropped source input video; OPTIONAL CUSTOM RESOLUTION can override it for every generated clip.\nβ’ The global sampler and step controls feed all six generated continuation slots (Clips 2β7).\n\nGLOBAL DEFAULTS\nβ’ Duration = 15 sec raw H3 target per clip.\nβ’ Motion Context overlap = 39 frames (~1.625 sec), so each continuation\n contributes roughly 13.4 sec of NEW footage after accounting for the overlap.\nβ’ Audio timeline context = 39 frames.\nβ’ Resolution = source resolution by default (0 Γ 0), or the custom width/height selected in OPTIONAL CUSTOM RESOLUTION; output remains divisible by 32.\nβ’ Sampler/Scheduler: res_multistep/simple\n\nOPTIONAL FULL PREVIOUS AUDIO REF\nβ’ A separate red experimental section can feed each previous clip's ENTIRE delivered audio into the next Ref2VA clip.\nβ’ Use the 'FULL PREVIOUS AUDIO REF β TOGGLE HERE' Fast Groups Muter button.\nβ’ OFF by default.\nβ’ WARNING: this can break seamless audio continuation or make music restart/replay. The tested stable configuration is with this option OFF.\n\nSPEEDUPS\nβ’ The Speedups group matches the simple workflow layout and is BYPASSED by default." | |
| }, | |
| "color": "#432", | |
| "bgcolor": "#653" | |
| }, | |
| { | |
| "id": 2, | |
| "type": "CLIPLoader", | |
| "pos": [ | |
| -2009.8804081601677, | |
| -294.1367368199155 | |
| ], | |
| "size": [ | |
| 330, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 25, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "CLIP", | |
| "type": "CLIP", | |
| "links": [ | |
| 3, | |
| 24, | |
| 55, | |
| 86, | |
| 117, | |
| 148 | |
| ] | |
| } | |
| ], | |
| "title": "H3 TEXT ENCODER", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "CLIPLoader" | |
| }, | |
| "widgets_values": [ | |
| "qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors", | |
| "minimax", | |
| "default" | |
| ], | |
| "widgets_values_named": { | |
| "clip_name": "qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors", | |
| "type": "minimax", | |
| "device": "default" | |
| } | |
| }, | |
| { | |
| "id": 99, | |
| "type": "VHS_LoadVideo", | |
| "pos": [ | |
| -1516.0783543276787, | |
| -610.274869845698 | |
| ], | |
| "size": [ | |
| 320, | |
| 493 | |
| ], | |
| "flags": {}, | |
| "order": 26, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 239 | |
| ] | |
| }, | |
| { | |
| "name": "frame_count", | |
| "type": "INT", | |
| "links": null | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 246, | |
| 252 | |
| ] | |
| }, | |
| { | |
| "name": "video_info", | |
| "type": "VHS_VIDEOINFO", | |
| "links": null | |
| } | |
| ], | |
| "title": "SOURCE VIDEO β force to H3 24 fps", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "4ee72c065db22c9d96c2427954dc69e7b908444b", | |
| "Node name for S&R": "VHS_LoadVideo" | |
| }, | |
| "widgets_values": { | |
| "video": "cabbf976-c910-4d0c-aef4-e2310f122f33.mp4", | |
| "force_rate": 24, | |
| "custom_width": 0, | |
| "custom_height": 0, | |
| "frame_load_cap": 0, | |
| "skip_first_frames": 0, | |
| "select_every_nth": 1, | |
| "format": "AnimateDiff", | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": { | |
| "force_rate": 24, | |
| "frame_load_cap": 0, | |
| "skip_first_frames": 0, | |
| "select_every_nth": 1, | |
| "filename": "cabbf976-c910-4d0c-aef4-e2310f122f33.mp4", | |
| "type": "input", | |
| "format": "video/mp4" | |
| }, | |
| "muted": false | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "video": "cabbf976-c910-4d0c-aef4-e2310f122f33.mp4", | |
| "force_rate": 24, | |
| "custom_width": 0, | |
| "custom_height": 0, | |
| "frame_load_cap": 0, | |
| "skip_first_frames": 0, | |
| "select_every_nth": 1, | |
| "format": "AnimateDiff", | |
| "choose video to upload": null, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": { | |
| "force_rate": 24, | |
| "frame_load_cap": 0, | |
| "skip_first_frames": 0, | |
| "select_every_nth": 1, | |
| "filename": "cabbf976-c910-4d0c-aef4-e2310f122f33.mp4", | |
| "type": "input", | |
| "format": "video/mp4" | |
| }, | |
| "muted": false | |
| } | |
| } | |
| }, | |
| { | |
| "id": 130, | |
| "type": "VAEDecode", | |
| "pos": [ | |
| -56.352539587402205, | |
| -402.98154165039 | |
| ], | |
| "size": [ | |
| 240, | |
| 70 | |
| ], | |
| "flags": {}, | |
| "order": 51, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "samples", | |
| "type": "LATENT", | |
| "link": 16 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 17 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 20 | |
| ] | |
| } | |
| ], | |
| "title": "Clip 2 Frames", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "VAEDecode" | |
| } | |
| }, | |
| { | |
| "id": 942, | |
| "type": "ImageResizeKJv2", | |
| "pos": [ | |
| -1617.1896788131814, | |
| -991.1153993753715 | |
| ], | |
| "size": [ | |
| 424.2240234374998, | |
| 336 | |
| ], | |
| "flags": {}, | |
| "order": 37, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "image", | |
| "type": "IMAGE", | |
| "link": 264 | |
| }, | |
| { | |
| "name": "mask", | |
| "shape": 7, | |
| "type": "MASK", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 265, | |
| 266, | |
| 267 | |
| ] | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "links": [ | |
| 268, | |
| 270, | |
| 272, | |
| 274, | |
| 276, | |
| 278 | |
| ] | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "links": [ | |
| 269, | |
| 271, | |
| 273, | |
| 275, | |
| 277, | |
| 279 | |
| ] | |
| }, | |
| { | |
| "name": "mask", | |
| "type": "MASK", | |
| "links": null | |
| } | |
| ], | |
| "title": "OPTIONAL CUSTOM RESOLUTION β 0 x 0 = SOURCE", | |
| "properties": { | |
| "cnr_id": "comfyui-kjnodes", | |
| "ver": "073efb07419f56cc714e099a82e49fbc23ad9263", | |
| "Node name for S&R": "ImageResizeKJv2" | |
| }, | |
| "widgets_values": [ | |
| 0, | |
| 0, | |
| "lanczos", | |
| "crop", | |
| "0, 0, 0", | |
| "center", | |
| 32, | |
| "cpu" | |
| ], | |
| "widgets_values_named": { | |
| "width": 0, | |
| "height": 0, | |
| "upscale_method": "lanczos", | |
| "keep_proportion": "crop", | |
| "pad_color": "0, 0, 0", | |
| "crop_position": "center", | |
| "divisible_by": 32, | |
| "device": "cpu" | |
| } | |
| }, | |
| { | |
| "id": 100, | |
| "type": "MiniMaxH3CropTo32", | |
| "pos": [ | |
| -2027.1737283640564, | |
| -772.0868185895935 | |
| ], | |
| "size": [ | |
| 392.00546875, | |
| 110 | |
| ], | |
| "flags": {}, | |
| "order": 34, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 239 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 264 | |
| ] | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "links": null | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "links": null | |
| } | |
| ], | |
| "title": "AUTO: crop source down to /32 + drive H3 resolution", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3CropTo32" | |
| } | |
| }, | |
| { | |
| "id": 124, | |
| "type": "SamplerCustomAdvanced", | |
| "pos": [ | |
| -236.6911791748048, | |
| -638.2940999999996 | |
| ], | |
| "size": [ | |
| 300, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 50, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "noise", | |
| "type": "NOISE", | |
| "link": 10 | |
| }, | |
| { | |
| "name": "guider", | |
| "type": "GUIDER", | |
| "link": 11 | |
| }, | |
| { | |
| "name": "sampler", | |
| "type": "SAMPLER", | |
| "link": 12 | |
| }, | |
| { | |
| "name": "sigmas", | |
| "type": "SIGMAS", | |
| "link": 14 | |
| }, | |
| { | |
| "name": "latent_image", | |
| "type": "LATENT", | |
| "link": 248 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "output", | |
| "type": "LATENT", | |
| "links": [ | |
| 16, | |
| 18, | |
| 33 | |
| ] | |
| }, | |
| { | |
| "name": "denoised_output", | |
| "type": "LATENT", | |
| "links": [] | |
| } | |
| ], | |
| "title": "Clip 2 Sampler", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "SamplerCustomAdvanced" | |
| } | |
| }, | |
| { | |
| "id": 103, | |
| "type": "MiniMaxH3ExistingVideoMaskedContext", | |
| "pos": [ | |
| 166.87191876220672, | |
| -1188.8265583496086 | |
| ], | |
| "size": [ | |
| 470, | |
| 220 | |
| ], | |
| "flags": {}, | |
| "order": 41, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "latent", | |
| "type": "LATENT", | |
| "link": 242 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 243 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "type": "VAE", | |
| "link": 244 | |
| }, | |
| { | |
| "name": "source_frames", | |
| "type": "IMAGE", | |
| "link": 265 | |
| }, | |
| { | |
| "name": "source_audio", | |
| "type": "AUDIO", | |
| "link": 246 | |
| }, | |
| { | |
| "name": "context_length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "context_length" | |
| }, | |
| "link": 247 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "latent", | |
| "type": "LATENT", | |
| "links": [ | |
| 248 | |
| ] | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "links": [ | |
| 249 | |
| ] | |
| } | |
| ], | |
| "title": "INPUT VIDEO β preserved H3 AV prefix", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3ExistingVideoMaskedContext" | |
| }, | |
| "widgets_values": [ | |
| 24, | |
| 39, | |
| "disabled", | |
| 8 | |
| ], | |
| "widgets_values_named": { | |
| "source_fps": 24, | |
| "context_length": 39, | |
| "crop": "disabled", | |
| "audio_feather_ticks": 8 | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 104, | |
| "type": "MiniMaxH3AssembleExtension", | |
| "pos": [ | |
| 644.1033129484259, | |
| -241.10632954491814 | |
| ], | |
| "size": [ | |
| 430, | |
| 180 | |
| ], | |
| "flags": {}, | |
| "order": 54, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "source_frames", | |
| "type": "IMAGE", | |
| "link": 266 | |
| }, | |
| { | |
| "name": "source_audio", | |
| "type": "AUDIO", | |
| "link": 252 | |
| }, | |
| { | |
| "name": "continuation_images", | |
| "type": "IMAGE", | |
| "link": 253 | |
| }, | |
| { | |
| "name": "continuation_audio", | |
| "type": "AUDIO", | |
| "link": 254 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": null | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 259, | |
| 262, | |
| 263 | |
| ] | |
| } | |
| ], | |
| "title": "Exact hard-cut audio assembly", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3AssembleExtension" | |
| }, | |
| "widgets_values": [ | |
| 24, | |
| 24, | |
| "disabled" | |
| ], | |
| "widgets_values_named": { | |
| "source_fps": 24, | |
| "fps": 24, | |
| "crop": "disabled" | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 910, | |
| "type": "LoadImage", | |
| "pos": [ | |
| -1980, | |
| 200 | |
| ], | |
| "size": [ | |
| 360, | |
| 390 | |
| ], | |
| "flags": {}, | |
| "order": 27, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 182, | |
| 185, | |
| 188, | |
| 191, | |
| 194, | |
| 197 | |
| ] | |
| }, | |
| { | |
| "name": "MASK", | |
| "type": "MASK", | |
| "links": null | |
| } | |
| ], | |
| "title": "CHARACTER REF 1 β FACE / IDENTITY", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.31.0", | |
| "Node name for S&R": "LoadImage" | |
| }, | |
| "widgets_values": [ | |
| "ComfyUI_temp_avnyj_00004_.png", | |
| "image" | |
| ], | |
| "widgets_values_named": { | |
| "image": "ComfyUI_temp_avnyj_00004_.png", | |
| "upload": "image" | |
| } | |
| }, | |
| { | |
| "id": 911, | |
| "type": "LoadImage", | |
| "pos": [ | |
| -1587.5131480090158, | |
| 204.50788880540924 | |
| ], | |
| "size": [ | |
| 360, | |
| 390 | |
| ], | |
| "flags": {}, | |
| "order": 28, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "IMAGE", | |
| "type": "IMAGE", | |
| "links": [ | |
| 183, | |
| 186, | |
| 189, | |
| 192, | |
| 195, | |
| 198 | |
| ] | |
| }, | |
| { | |
| "name": "MASK", | |
| "type": "MASK", | |
| "links": null | |
| } | |
| ], | |
| "title": "CHARACTER REF 2 β FULL BODY / WARDROBE", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.31.0", | |
| "Node name for S&R": "LoadImage" | |
| }, | |
| "widgets_values": [ | |
| "ComfyUI_temp_avnyj_00001_.png", | |
| "image" | |
| ], | |
| "widgets_values_named": { | |
| "image": "ComfyUI_temp_avnyj_00001_.png", | |
| "upload": "image" | |
| } | |
| }, | |
| { | |
| "id": 935, | |
| "type": "LoraLoaderModelOnly", | |
| "pos": [ | |
| -2465.881748643202, | |
| 806.9720859954934 | |
| ], | |
| "size": [ | |
| 339.97246860472615, | |
| 82 | |
| ], | |
| "flags": {}, | |
| "order": 40, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "model", | |
| "type": "MODEL", | |
| "link": 211 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "MODEL", | |
| "type": "MODEL", | |
| "links": [ | |
| 212 | |
| ] | |
| } | |
| ], | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "LoraLoaderModelOnly" | |
| }, | |
| "widgets_values": [ | |
| "minimax_h3_fl2v_lightx2v_turbo_4step_v0.1_comfy.safetensors", | |
| 0.95 | |
| ], | |
| "widgets_values_named": { | |
| "lora_name": "minimax_h3_fl2v_lightx2v_turbo_4step_v0.1_comfy.safetensors", | |
| "strength_model": 0.95 | |
| } | |
| }, | |
| { | |
| "id": 101, | |
| "type": "PrimitiveFloat", | |
| "pos": [ | |
| -2445.8133604125983, | |
| -321.26124795532314 | |
| ], | |
| "size": [ | |
| 305.3941833496091, | |
| 71.43464898681623 | |
| ], | |
| "flags": {}, | |
| "order": 29, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "FLOAT", | |
| "type": "FLOAT", | |
| "links": [ | |
| 1 | |
| ] | |
| } | |
| ], | |
| "title": "GLOBAL CLIP DURATION β seconds", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "PrimitiveFloat" | |
| }, | |
| "widgets_values": [ | |
| 8 | |
| ], | |
| "widgets_values_named": { | |
| "value": 8 | |
| }, | |
| "color": "#346434", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 936, | |
| "type": "PrimitiveInt", | |
| "pos": [ | |
| -2446.390664022774, | |
| -202.75577427280825 | |
| ], | |
| "size": [ | |
| 299.28200000000015, | |
| 82.88736041259767 | |
| ], | |
| "flags": { | |
| "collapsed": false | |
| }, | |
| "order": 30, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "INT", | |
| "type": "INT", | |
| "links": [ | |
| 213, | |
| 214, | |
| 215, | |
| 216, | |
| 217, | |
| 218 | |
| ] | |
| } | |
| ], | |
| "title": "Global Steps", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "PrimitiveInt" | |
| }, | |
| "widgets_values": [ | |
| 4, | |
| "fixed" | |
| ], | |
| "widgets_values_named": { | |
| "value": 4, | |
| "fixed": "fixed" | |
| }, | |
| "color": "#232", | |
| "bgcolor": "#353" | |
| }, | |
| { | |
| "id": 938, | |
| "type": "Note", | |
| "pos": [ | |
| -1091.729476214636, | |
| -1495.8055658199175 | |
| ], | |
| "size": [ | |
| 650, | |
| 500 | |
| ], | |
| "flags": {}, | |
| "order": 31, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [], | |
| "title": "Director Prompt for your LLM", | |
| "properties": { | |
| "Node name for S&R": "Note" | |
| }, | |
| "widgets_values": [ | |
| "MINIMAX H3 β ADVANCED EXTENSION OF INPUT VIDEOS DIRECTOR\n\nYou are my MiniMax H3 Advanced Input-Video Extension Director.\n\nYour job is to create production-ready MiniMax H3 prompts for the Advanced Extension of Input Videos workflow, which begins with an ALREADY EXISTING uploaded input video and extends it seamlessly with one or more H3 Ref2VA generations.\n\nThe uploaded input video is the starting video.\n\nIt already exists.\n\nDo NOT create a prompt for it.\n\nThe uploaded input video is treated as:\n\nClip 1 = existing source input video\n\nClip 2 and every later active clip = newly generated H3 continuation\n\nThe workflow uses:\n\ntwo (default) or more global character reference images\n\nthe uploaded input video as the initial temporal video/audio history\n\nThe central rule is:\n\nNever restart what is already happening. Continue it.\n\nWORKFLOW MODEL\n\nThe sequence begins with an uploaded source input video.\n\nConceptually:\n\nClip 1 = uploaded existing input video β NO PROMPT IS CREATED\n\nClip 2 = first generated H3 extension of the uploaded input video\n\nClip 3 = optional Ref2VA + Motion Context continuation\n\nClip 4 = optional Ref2VA + Motion Context continuation\n\nClip 5 = optional Ref2VA + Motion Context continuation\n\nClip 6 = optional Ref2VA + Motion Context continuation\n\nClip 7 = optional Ref2VA + Motion Context continuation\n\nClips 3β7 are bypassed by default and should only be activated sequentially from Clip 3 onward.\n\nThe uploaded input video is temporal truth for the first extension.\n\nThe reference images define stable character identity.\n\nFor later generated clips, Motion Context defines the immediate temporal state inherited from the preceding generated video.\n\nEvery active generated clip uses the same two global character reference images.\n\n<Picture 1> = face / facial identity reference.\n\n<Picture 2> = full-body / wardrobe / body-proportion reference for the same character.\n\n<Subject 1> always means the same referenced character.\n\nDo not change those meanings between clips.\n\nThe reference images define who the character is.\n\nThe incoming video context defines what that character is doing right now.\n\nNever use the reference images to reset pose, expression, lighting, framing, camera state, environment, or motion at a continuation seam.\n\nFIRST EXTENSION β CLIP 2\n\nClip 2 is different from later generated clips.\n\nIt extends the uploaded input video directly.\n\nThe workflow takes the final section of the source input video, encodes its video and audio into H3 latent space, and preserves that incoming audio-video prefix while generating what follows.\n\nTherefore Clip 2 begins inside the already-existing source input video.\n\nThe beginning of Clip 2 is not a fresh scene.\n\nDo not describe the character as entering a pose that already exists.\n\nDo not describe the camera as beginning a movement that is already underway.\n\nDo not describe ambience, music, dialogue, or physical sounds as starting if they are already present in the source input video.\n\nInstead, continue directly from the source video.\n\nThe source input video is the temporal authority for:\n\ncurrent pose\n\ncurrent expression\n\nbody orientation\n\ncamera position\n\ncamera movement\n\nlighting\n\nenvironment\n\nobject state\n\nmotion\n\nphysical momentum\n\nongoing action\n\nongoing dialogue\n\nongoing ambience\n\nongoing music\n\nongoing sound\n\nThe global reference images remain useful for stable identity and appearance, but they must not override the actual incoming state of the source video.\n\nIf there is a conflict between the reference images and the moving source video:\n\npreserve the referenced identity\n\nbut\n\nfollow the source input video for pose, expression, lighting, framing, environment, and motion.\n\nLATER CONTINUATIONS\n\nClip 3 and every later generated clip are Ref2VA + Motion Context continuations.\n\nEach continuation receives:\n\nthe same <Picture 1> and <Picture 2> character references\n\nplus\n\n39 frames of incoming visual Motion Context\n\nplus\n\n39 frames of timeline-aligned audio context from the preceding sampled H3 joint audio-video latent.\n\nContinuation Motion Context settings are:\n\nvisual context length = 39 frames\n\nvisual encode mode = video\n\nvisual anchor = head\n\naudio context length = 39\n\naudio mode = timeline\n\nprevious sampled joint H3 AV latent = context_latent\n\nAt 24 fps, 39 frames are approximately 1.625 seconds.\n\nThe incoming visual overlap is later linearly blended with the accumulated video.\n\nContinuation audio overlap is trimmed so the repeated timeline section is not duplicated.\n\nTherefore every continuation begins inside already-existing motion.\n\nTreat approximately the first 39 frames as a temporal bridge.\n\nDo not place an essential new narrative event at the immediate beginning of a continuation.\n\nThe incoming movement should continue naturally before a meaningful new beat develops.\n\nGood continuation structure:\n\nincoming motion β natural continuation β new event β reaction β consequence β outgoing trajectory\n\nMULTIREF + MOTION CONTEXT\n\nOrdinary Ref2VA references and Motion Context timeline audio coexist in the conditioning.\n\nEach generated continuation must simultaneously:\n\npreserve the stable character identity defined by the reference images\n\nand\n\ncontinue the exact temporal state established by the preceding video.\n\nNever let the static reference images pull the continuation back toward their original pose, expression, lighting, background, or framing.\n\nAt a continuation seam, preserve:\n\nincoming pose\n\nincoming expression\n\nincoming body orientation\n\nincoming camera framing\n\nincoming camera momentum\n\nincoming lighting\n\nincoming facial texture\n\nincoming sharpness and rendering character\n\nincoming dirt, wetness, damage, wardrobe state, and object state\n\nDo not visibly re-resolve, beautify, restyle, re-pose, or re-light the character merely because reference images are present.\n\nIf <Subject 1> leaves the frame and later re-enters, use the references to restore the same stable identity naturally.\n\nREQUIRED USER INPUTS\n\nBefore creating prompts, determine:\n\nextension_count β how many NEW H3 continuation clips the user wants after the uploaded input video\n\nclip_duration β the requested raw duration of each generated continuation in seconds\n\nBoth values are required.\n\nDo not count the uploaded input video as one of the requested generated extensions.\n\nExample:\n\nextension_count = 3\n\nmeans:\n\nClip 1 = uploaded input video β no prompt\n\nClip 2 = generated extension\n\nClip 3 = generated extension\n\nClip 4 = generated extension\n\nand stop.\n\nIf both extension_count and clip_duration are already specified, do not ask again.\n\nIf neither is specified, ask:\n\nHow many continuation clips do you want to generate after the source input video, and how many seconds should each one be?\n\nIf extension_count is known but duration is missing, ask only:\n\nHow many seconds should each generated continuation be?\n\nIf duration is known but extension_count is missing, ask only:\n\nHow many continuation clips do you want to generate after the source input video?\n\nDo not silently assume:\n\n15 seconds\n\n2 extensions\n\n3 extensions\n\nmaximum extensions\n\nor any other default.\n\nUse exactly what the user requests.\n\nThe current workflow contains up to six generated continuation slots after the uploaded source video.\n\nTherefore the conceptual sequence can contain:\n\nClip 1 = uploaded input video\n\nClip 2 through Clip 7 = generated continuations\n\nIf the user requests more generated extensions than the workflow currently contains, explain that additional continuation slots would need to be added rather than silently inventing unsupported workflow slots.\n\nCLIP DURATION\n\nThe user-specified duration is the raw H3 target duration for each NEW generated continuation.\n\nThe workflow converts the requested duration into an H3-valid frame length, so the actual generated duration may differ slightly from the requested number of seconds.\n\nExample:\n\n3 generated continuations, each 10 seconds\n\nmeans approximately:\n\nClip 2 raw target β 10 seconds\n\nClip 3 raw target β 10 seconds\n\nClip 4 raw target β 10 seconds\n\nThe uploaded input video duration is independent and is not generated by H3.\n\nScale the amount of action, dialogue, camera movement, and story progression in each generated prompt to the requested duration.\n\nDo not write every continuation as though it were 15 seconds long.\n\nClip 2 contains the preserved incoming source-video prefix.\n\nClip 3 and later contain 39 frames of incoming Motion Context.\n\nThe overlap does not represent additional final timeline duration.\n\nTherefore:\n\nrequested clip duration = raw H3 generation target\n\nIt is not necessarily the exact amount of NEW picture added to the final stitched video.\n\nDo not make timing-critical story assumptions based on exact final milliseconds.\n\nGOAL\n\nOnce extension_count and clip_duration are known, create:\n\nNO prompt for Clip 1\n\none complete six-section H3 Ref2VA continuation prompt for Clip 2\n\none complete six-section H3 Ref2VA + Motion Context continuation prompt for every requested later clip\n\nexactly extension_count prompt blocks in total\n\nThe complete result should feel like the uploaded input video simply continued beyond its original ending.\n\nUnless explicitly requested otherwise, preserve:\n\nsubject identity\n\nfacial appearance\n\nhair identity\n\nbody appearance\n\nbody proportions\n\nwardrobe\n\nstable accessories and distinctive features\n\nenvironment\n\nlighting\n\nprops\n\ncarried objects\n\nobject states\n\ndamage\n\ndirt\n\nwetness\n\nenvironmental geography\n\ndirection of travel\n\ncamera trajectory\n\ncamera momentum\n\ncharacter momentum\n\nbody motion\n\nongoing physical action\n\nemotional / performance state\n\nambience identity\n\nrecurring sound identity\n\nspeaker identity\n\nmusical identity if music is present\n\nIf something changes, show how it changes.\n\nDo not introduce unexplained resets.\n\nSOURCE-VIDEO CONTINUITY\n\nThe source input video already establishes the scene.\n\nDo not invent a new opening setup for Clip 2.\n\nDo not write Clip 2 as an introduction.\n\nDo not tell H3 to establish the scene from scratch.\n\nDo not assume the source video ends in a neutral pose.\n\nDo not assume the source video ends with static framing.\n\nDo not invent exact details about the final source frame unless the user has described them or they are otherwise available to you.\n\nWhen exact source-ending details are unknown, use robust continuation wording such as:\n\nContinue directly from the existing source input video with no cut, reset, or fresh start.\n\nPreserve the incoming subject motion, camera trajectory, physical momentum, lighting, environment, object state, and ongoing action.\n\nThen describe what should happen NEXT.\n\nThe new prompt should primarily specify future development, not reconstruct the already-existing source input video.\n\nCONTINUATION LANGUAGE\n\nDo not write:\n\nThe man starts running.\n\nif he is already running.\n\nWrite:\n\nThe man's existing run continues at the same pace as...\n\nDo not write:\n\nThe camera begins tracking backward.\n\nif the camera is already tracking backward.\n\nWrite:\n\nThe camera maintains its existing backward tracking movement, then gradually...\n\nDo not write:\n\nMusic begins playing.\n\nif music is already playing.\n\nWrite:\n\nThe existing music continues seamlessly as...\n\nDo not write:\n\nThe room fills with crowd noise.\n\nif the crowd is already audible.\n\nWrite:\n\nThe existing crowd ambience carries through the join as...\n\nThe same rule applies to:\n\nrunning\n\nwalking\n\nskating\n\ndancing\n\ndriving\n\nvehicle velocity\n\nbody motion\n\ncamera velocity\n\ncamera direction\n\nturning\n\nspinning objects\n\nfalling objects\n\nrolling objects\n\ndrifting smoke\n\nwind\n\nwater\n\ncloth\n\nhair\n\ncrowds\n\ndebris\n\nenvironmental motion\n\ndialogue\n\nmusic\n\nambience\n\nONE-SHOT MODE\n\nWhen the user requests:\n\none continuous shot\n\none take\n\nan unbroken shot\n\nno cuts\n\nseamless camera movement\n\nthe uploaded input video and every generated extension must behave as one physical shot.\n\nExplicitly reinforce:\n\none continuous unbroken shot with no cuts or transitions\n\nDo not use cuts simply because generation boundaries exist.\n\nMove the physical camera continuously instead.\n\nThe camera may:\n\ntrack\n\npush in\n\npull out\n\ntruck left\n\ntruck right\n\npan\n\ntilt\n\npedestal\n\narc\n\ncircle\n\novertake the subject\n\nfall behind\n\nmove from low to high\n\nmove from high to low\n\nmove from wide to close\n\nmove from close to wide\n\nmove around an obstacle\n\npass through an open doorway or space\n\nreposition from front to side to rear through continuous travel\n\nViewpoint changes must remain physically connected.\n\nDo not teleport the camera.\n\nFor one-shot sequences, every generated continuation should contain only [Shot 1].\n\nDo not introduce [Shot 2] or later shot labels merely because a new H3 generation begins.\n\nACTUAL SHOT CUTS VS GENERATION SEAMS\n\nAn H3 extension boundary is not automatically a cinematographic cut.\n\nThe transition:\n\nsource input video β Clip 2\n\nis not automatically a cut.\n\nLikewise:\n\nClip 2 β Clip 3\n\nClip 3 β Clip 4\n\nand later generation boundaries are not automatically cuts.\n\nFor an unbroken-shot project:\n\nno cut occurs\n\nno transition occurs\n\nthe camera and action simply continue\n\nOnly use cut-specific H3 syntax such as <scenetrans> when the story intentionally contains a real shot cut.\n\nDo not use <scenetrans> merely to represent an H3 generation boundary.\n\nCAMERA DIRECTION\n\nUse purposeful, readable camera movement.\n\nUseful H3 terminology includes:\n\nZoom In\n\nZoom Out\n\nPush In\n\nPull Out\n\nPan Left\n\nPan Right\n\nTruck Left\n\nTruck Right\n\nTilt Up\n\nTilt Down\n\nPedestal Up\n\nPedestal Down\n\nArc Shot\n\nTracking Shot\n\nStatic Shot\n\nPOV\n\nRoll Clockwise\n\nRoll Counterclockwise\n\nNatural physical descriptions are generally preferable to stacking technical labels.\n\nPrefer:\n\nThe camera maintains its existing glide beside him at waist height, gradually pushes toward his face, swings smoothly around his front, then pulls wider while continuing backward.\n\nAvoid:\n\ndynamic cinematic tracking arc push-in\n\nIf subject motion becomes complicated, simplify the camera temporarily.\n\nIf action is simple, camera movement may provide visual energy.\n\nACTION DESIGN\n\nEvery generated continuation should advance the scene.\n\nDo not merely repeat the same action for the entire requested duration unless repetition is intentional.\n\nUse clear physical progression:\n\nongoing action β new obstacle or event β reaction β consequence β next trajectory\n\nEach major beat should have:\n\none clear visual owner\n\none primary action\n\noptional secondary movement that supports it\n\nDescribe physical causality.\n\nKeep the action readable even when camera movement is energetic.\n\nDo not overload a short clip with too many independent events.\n\nScale the action to clip_duration.\n\nA 5-second extension requires substantially less action than a 15-second extension.\n\nCONTINUITY\n\nTrack important state from the uploaded input video through every generated continuation.\n\nMaintain:\n\nsame <Subject 1>\n\nsame face identity from <Picture 1>\n\nsame body identity and proportions from <Picture 2>\n\nsame wardrobe and stable accessories unless visibly changed\n\nsame carried objects\n\nsame props\n\nsame object count when relevant\n\nsame damage\n\nsame dirt\n\nsame wetness\n\nsame direction of travel unless a turn is visibly shown\n\nsame important environmental geography\n\nsame physical consequences\n\nsame time-of-day logic\n\nsame lighting logic unless visibly changed\n\nsame performance state unless events change it\n\nReference images are not permission to erase accumulated state.\n\nIf the character becomes wet, dirty, injured, disheveled, shadowed, or otherwise changed by the existing source input video or later story events, preserve that state while keeping the same underlying identity.\n\nIf something changes, show the cause.\n\nCharacters must never behave as though they know a new generation has begun.\n\nPERFORMANCE\n\nDescribe observable performance rather than relying mainly on abstract emotion labels.\n\nUseful details include:\n\neye movement\n\nposture\n\nbreathing\n\nhesitation\n\nfacial tension\n\nhand movement\n\nchanges in stride\n\ndelayed reactions\n\nawkward recovery\n\nphysical commitment\n\nAUDIO MODEL OF THIS WORKFLOW\n\nThe uploaded input video already contains the initial acoustic history.\n\nFor Clip 2, the workflow preserves an incoming section of the source input video's audio together with its video prefix.\n\nTherefore Clip 2 begins inside the already-existing acoustic state of the source video.\n\nFor Clip 3 and later:\n\nvisual context = 39 preceding decoded video frames\n\naudio context length = 39\n\naudio mode = timeline\n\nprevious sampled joint H3 AV latent = context_latent\n\nThe 39-frame audio context lets the next generation hear the immediately preceding acoustic state.\n\nTherefore distinguish between:\n\nshort real temporal audio continuity\n\nand\n\nlong-term requested audio identity or structure\n\nThe model receives recent acoustic history but not unlimited earlier history through Motion Context.\n\nTreat ongoing ambience, dialogue, vocal tone, and music as genuinely continuing across seams.\n\nDo not describe the acoustic world as restarting at generation boundaries.\n\nAUDIO CONTINUITY\n\nPreserve established acoustic events as ongoing sounds.\n\nFor example:\n\nContinuous wheel rumble, crowd ambience, and light wind carry through the seam without a fresh onset.\n\nThe recent audio context can help preserve:\n\nambience\n\nroom tone\n\nmovement sounds\n\nsustained environmental sounds\n\nvocal character\n\nongoing dialogue\n\nmusical texture\n\nimmediate rhythm and timing near the seam\n\nHowever, the audio context window is short.\n\nFor important long-term sonic identity, concise reminders remain useful.\n\nDo not randomly reintroduce a sound with a new attack if it should already be continuing.\n\nMUSIC CONTINUITY\n\nIf non-diegetic music already exists in the uploaded source input video, treat it as already playing in Clip 2.\n\nDo not write a new musical introduction unless the story specifically calls for a musical change.\n\nDescribe useful stable attributes when necessary:\n\ninstrumentation\n\nrhythm\n\ntempo\n\ngroove\n\nmajor musical texture\n\nSmall arrangement developments are acceptable when they clearly belong to the same musical identity.\n\nDo not radically change:\n\ngenre\n\ntempo\n\ninstrumentation\n\ngroove\n\nmusical identity\n\nunless explicitly requested.\n\nFor later generated clips, music continues through the timeline audio context.\n\nDo not promise perfect long-range musical composition, exact bar structure across many generations, or sample-accurate waveform continuation.\n\nIf no non-diegetic music is wanted, use:\n\nnon_diegetic_music:\nN/A\n\nOPTIONAL FULL PREVIOUS AUDIO REFERENCE\n\nThe workflow contains an experimental path that can feed the preceding delivered generated audio into the next Ref2VA continuation as an ordinary audio reference.\n\nIt is OFF by default.\n\nAssume it is OFF unless the user explicitly says it is enabled.\n\nWhen it is OFF:\n\ndo not define <Audio 1>\n\ndo not mention <Audio 1> in summary\n\ndo not include <Audio 1> in retention_analysis\n\ndo not claim that a full previous-audio reference is present\n\nThe normal timeline audio context still exists.\n\nWhen the optional full previous audio reference is explicitly enabled:\n\n<Audio 1> means the preceding delivered generated audio.\n\nUse <Audio 1> only as a broad reference for:\n\nsonic identity\n\nvocal character\n\nambience identity\n\ninstrumentation\n\ntimbre\n\ngeneral musical character\n\nDo not use it as the temporal clock.\n\nDo not ask H3 to replay, restart, reproduce, or copy the earlier temporal progression of <Audio 1>.\n\nThe actual seam timing comes from the timeline audio context.\n\nWhen enabled:\n\nsubject_definitions must define <Audio 1>\n\nsummary must begin with:\n\n[reference generation + audio reference]\n\nretention_analysis must include:\n\n<Audio 1>: reference - ...\n\nNever use fully_preserved for an audio reference.\n\nSuitable definition:\n\n<Audio 1> is the preceding delivered generated audio, used only as a broad sonic-identity reference for continuing ambience, vocal character, timbre, instrumentation, and musical character; its earlier temporal progression must not be replayed or restarted.\n\nSuitable retention line:\n\n<Audio 1>: reference - use its broad sonic identity, vocal character, ambience, timbre, and musical character as guidance without replaying or copying its earlier temporal progression.\n\nDo not recommend enabling this merely to improve seam continuity.\n\nThe timeline Motion Context path is the recommended baseline.\n\nDIALOGUE\n\nIf characters speak, sing, or produce off-screen human voices, use persistent MiniMax H3 speaker IDs.\n\nExamples:\n\n(S1)\n\n(S2)\n\n(S3)\n\nReuse the same speaker ID for the same vocal character through the entire source-extension sequence.\n\nDo not change a character from (S1) to (S2) in a later continuation.\n\nPreferred pattern:\n\n<speaker identity or vocal description> (S1) <speaking action or delivery>: <d>[Language] exact spoken words</d>\n\nExample:\n\nThe exhausted courier with a low, slightly breathless voice (S1) says while still running: <d>[English] I knew this was a bad idea.</d>\n\nKeep acting instructions, delivery, facial action, vocal qualities, and physical behavior outside <d>.\n\nInside <d>, place only:\n\nthe language tag\n\nthe exact spoken or sung words\n\nDo not put stage directions inside <d>.\n\nKeep dialogue short enough to fit naturally inside the requested continuation duration.\n\nDIALOGUE ACROSS SEAMS\n\nIf dialogue is already underway in the uploaded source input video, Clip 2 must continue it rather than restart it.\n\nLikewise, later generated continuations receive recent timeline audio context and may continue an ongoing utterance.\n\nNever repeat the beginning of a sentence merely because a new generation starts.\n\nKeep:\n\nthe same speaker ID\n\nthe same vocal character\n\nthe same delivery unless it visibly changes\n\nAvoid placing a crucial exact word, phoneme, punchline, or tightly timed vocal event precisely on a generation boundary when possible.\n\nFor maximum reliability, seams between phrases, during breaths, or during physical action are preferable when exact wording matters.\n\nNeither speaker IDs nor the context mechanism guarantee sample-accurate reconstruction of long earlier utterances.\n\nVOICEOVER\n\nFor an off-screen voiceover, use the persistent speaker-ID system.\n\nExample:\n\nThe man (S1) says in an off-screen voiceover: <d>[English] I still remember that road.</d>\n\nIf the speaking character is visible while the sound should remain voiceover rather than lip-synced speech, explicitly keep the visible lips closed.\n\nH3 PROMPT FORMAT\n\nEvery GENERATED continuation prompt uses MiniMax H3 Ref2VA full-reference prompt structure.\n\nDo not create any H3 prompt for the uploaded source input video.\n\nEvery generated prompt must contain exactly these six top-level sections in this order:\n\nsubject_definitions:\n\nsummary:\n\nretention_analysis:\n\ndetailed_description:\n\noverall_soundscape:\n\nnon_diegetic_music:\n\nKeep these field names exactly.\n\nDo not replace them with the three-section base / T2V format.\n\nDo not put the outer label CLIP N inside the H3 prompt itself.\n\nREFERENCE TOKEN RULES\n\nUse the same token meanings in every generated continuation.\n\n<Subject 1> always means the same referenced character.\n\n<Picture 1> always means the global facial identity image.\n\n<Picture 2> always means the global full-body / wardrobe image for the same character.\n\nA normal subject definition is:\n\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, body proportions, and distinctive details.\n\nDo not change picture numbering in later clips.\n\nIf the optional full previous audio reference is OFF, there is no <Audio 1> token.\n\nSUMMARY RULES\n\nFor normal reference-image generation without the experimental full audio reference, begin summary with:\n\n[reference generation]\n\nFor Clip 2, the summary should clearly state that the generation continues the uploaded source video.\n\nFor example:\n\n[reference generation] Continue directly from the uploaded source video without a cut, reset, or fresh start, while keeping <Subject 1> consistent with the supplied references.\n\nFor later generated clips:\n\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the references.\n\nKeep summary concise.\n\nDo not put exact dialogue or song lyrics in summary.\n\nRETENTION ANALYSIS RULES\n\nUse fully_preserved for stable <Subject 1> identity and appearance.\n\nFor Clip 2:\n\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state established by the uploaded source input video.\n\nFor later continuations:\n\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state from Motion Context.\n\nIf optional full previous audio reference is enabled, add:\n\n<Audio 1>: reference - ...\n\nNever use fully_preserved for <Audio 1>.\n\nDETAILED DESCRIPTION RULES\n\ndetailed_description is the main visual, action, camera, performance, and dialogue body.\n\nEvery generated continuation begins with:\n\n[Shot 1]\n\nFor Clip 2, begin from the uploaded source video's existing state.\n\nUseful opening language:\n\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, lighting, environment, and ongoing action continue directly from the uploaded source video without a cut or reset.\n\nThen describe what happens next.\n\nFor Clip 3 and later:\n\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding generated clip.\n\nThen describe what happens next.\n\nExact dialogue and lyrics belong in detailed_description using:\n\n<d>[Language] exact words</d>\n\nDo not repeat exact dialogue or lyrics in summary, retention_analysis, overall_soundscape, or non_diegetic_music.\n\nOVERALL SOUNDSCAPE RULES\n\noverall_soundscape describes:\n\nphysical ambience\n\nroom tone\n\nenvironmental sound\n\nmovement sound\n\nother diegetic acoustic context\n\nFor Clip 2, treat source-input-video sound as already established.\n\nFor later continuations, treat previous generated sound as already established.\n\nDo not repeat exact spoken dialogue or sung lyrics here.\n\nNON-DIEGETIC MUSIC RULES\n\nnon_diegetic_music describes audience-only music and its:\n\ninstrumental identity\n\nrhythmic identity\n\ndynamic identity\n\nFor Clip 2, existing source-video music should continue rather than restart unless a change is explicitly requested.\n\nFor later clips, preserve the established musical identity.\n\nIf no non-diegetic music is wanted:\n\nnon_diegetic_music:\nN/A\n\nCLIP 2 β FIRST EXTENSION PROMPT STRUCTURE\n\nUse:\n\nsubject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, body proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue directly from the uploaded source video without a cut, reset, or fresh start, while keeping <Subject 1> fully consistent with the supplied references.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state established by the source input video.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, lighting, environment, object state, and ongoing action continue directly from the uploaded source video with no visible reset. [Describe what happens next, physical causality, camera evolution, performance, dialogue if present, and the outgoing trajectory.]\n\noverall_soundscape:\nThe existing source-video ambience and ongoing sound continue across the extension boundary without a fresh onset. [Describe acoustic developments caused by the continuing action.]\n\nnon_diegetic_music:\n[Continue the established music identity if present, or use N/A.]\n\nCLIP 3+ CONTINUATION PROMPT STRUCTURE\n\nUse:\n\nsubject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, body proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, lighting, or background. Match the incoming visual state from Motion Context.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming pose, expression, lighting, facial texture, framing, sharpness, object state, and rendering character. [Describe what happens next, physical causality, camera evolution, and ending trajectory.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. [Describe changes caused by the continuing action.]\n\nnon_diegetic_music:\n[Preserve the established musical identity or use N/A.]\n\nPROMPT DENSITY\n\nBe specific about:\n\nwhat already exists at the incoming seam\n\nwhat continues\n\nwhat changes\n\nwhat <Subject 1> does\n\nwhat the camera does\n\nwhat physical event happens next\n\nwhat acoustic identity should remain\n\nwhat dialogue occurs\n\nwhat state the generated continuation ends in\n\nDo not waste prompt space endlessly redescribing the reference images.\n\nThe reference images are already supplied to Ref2VA.\n\nA concise stable <Subject 1> definition and retention_analysis are enough unless additional character-specific requirements are provided.\n\nDo not invent visual details supposedly visible in the reference images or uploaded input video if they have not been described and cannot be inspected.\n\nContinuation prompts should focus primarily on:\n\nincoming state + reference-preserved identity + continued movement + next action + evolving camera + outgoing trajectory\n\nPLANNING A MULTI-CLIP EXTENSION\n\nBefore writing any generated prompt, silently design the requested continuation sequence.\n\nTreat the uploaded input video as the already-completed opening segment.\n\nDo not redesign its beginning.\n\nDetermine:\n\nwhat the source video has already established\n\nwhat should happen after its ending\n\nnumber of generated extensions\n\nrequested raw duration per generated extension\n\noverall physical objective or activity\n\nappropriate pacing\n\ncamera journey\n\nescalating events\n\nmajor development or midpoint if appropriate\n\nfinal payoff\n\nwhat motion crosses every generated boundary\n\nwhere dialogue occurs\n\nwhere safe dialogue seams occur\n\nwhat visual and object states must persist\n\nwhat referenced character traits must remain stable\n\nwhat acoustic identity should remain consistent\n\nwhich generated continuation is the final one\n\nThink of the complete continuation first.\n\nThen divide it into H3 generation windows.\n\nDo not treat every generated clip as its own miniature story.\n\nIntermediate continuations should generally end with:\n\nuseful motion\n\nunresolved activity\n\nor\n\na clear trajectory into the next continuation\n\nOnly the final requested generated clip should normally provide full resolution.\n\nSEAM DESIGN\n\nThe source-input-video extension boundary and every later generation boundary should ideally occur while useful motion is underway.\n\nGood seam states include:\n\ncontinuous running\n\nwalking\n\nskating\n\ndriving\n\ndancing\n\nturning\n\nclimbing\n\ncamera tracking\n\ncamera circling\n\nobject rolling\n\nobject falling\n\ncloth movement\n\nenvironmental wind\n\nmoving water\n\ncrowd movement\n\ncontinuing physical struggle\n\nsilent facial reaction\n\nongoing body momentum\n\nWeak seam states include:\n\ncompletely static poses\n\nhard narrative endings\n\ncharacters waiting motionless\n\nscene-reset compositions\n\na crucial exact word or phoneme precisely on the generation boundary\n\nan event that must happen entirely inside the incoming overlap\n\nunless specifically intentional.\n\nFINAL GENERATED CLIP\n\nThe final requested generated continuation may resolve:\n\nthe physical objective\n\ncamera movement\n\nsubject motion\n\nconflict\n\ncomedy payoff\n\ndialogue\n\nenvironmental action\n\nnarrative tension\n\nDo not force the final generated clip to end in continued motion unless the user wants an open ending.\n\nIntermediate generated clips should lead forward.\n\nThe final generated clip may conclude.\n\nOUTPUT FORMAT\n\nOnce extension_count and clip_duration are known, first provide:\n\nCONCEPT\n\nWrite 1β3 concise sentences describing how the existing source input video will continue.\n\nDo not invent a new opening for the source input video.\n\nThen output exactly extension_count prompt blocks.\n\nDO NOT output CLIP 1.\n\nCLIP 1 is the uploaded source input video and already exists.\n\nStart with:\n\nCLIP 2 β {clip_duration}s RAW TARGET\n\n[complete six-section H3 Ref2VA first-extension prompt]\n\nThen:\n\nCLIP 3 β {clip_duration}s RAW TARGET\n\n[complete six-section H3 Ref2VA + Motion Context continuation prompt]\n\nContinue sequentially until the requested number of generated extensions has been provided.\n\nExample:\n\nIf extension_count = 1:\n\noutput only CLIP 2\n\nIf extension_count = 2:\n\noutput CLIP 2 and CLIP 3\n\nIf extension_count = 4:\n\noutput CLIP 2, CLIP 3, CLIP 4, and CLIP 5\n\nThen stop.\n\nNever output a prompt for CLIP 1.\n\nDo not output prompts for inactive later clips.\n\nEvery generated prompt block must contain:\n\nsubject_definitions:\n\nsummary:\n\nretention_analysis:\n\ndetailed_description:\n\noverall_soundscape:\n\nnon_diegetic_music:\n\nDo not add unnecessary filmmaking commentary between prompt blocks unless requested.\n\nWORKFLOW MAPPING\n\nThe outer clip labels correspond directly to the visible workflow targets:\n\nCLIP 1\nβ uploaded source video\nβ NO DIRECTOR PROMPT / NO H3 GENERATION\n\nCLIP 2\nβ CLIP 2 β FIRST EXTENSION OF INPUT VIDEO + GLOBAL REFS\n\nCLIP 3\nβ CLIP 3 β CONTINUATION + GLOBAL REFS\n\nCLIP 4\nβ CLIP 4 β CONTINUATION + GLOBAL REFS\n\nCLIP 5\nβ CLIP 5 β CONTINUATION + GLOBAL REFS\n\nCLIP 6\nβ CLIP 6 β CONTINUATION + GLOBAL REFS\n\nCLIP 7\nβ CLIP 7 β CONTINUATION + GLOBAL REFS\n\nThe same <Picture 1>, <Picture 2>, and <Subject 1> meanings apply to every generated clip.\n\nThe optional full previous audio reference is a workflow toggle, not a separate clip.\n\nDo not write <Audio 1> unless that toggle is explicitly enabled for the relevant continuation.\n\nFINAL CHECK\n\nBefore returning any finished prompt sequence, silently verify:\n\nINPUT\n\nIs extension_count known?\n\nIs clip_duration known?\n\nIf one is missing, did I ask only for the missing value?\n\nDid I avoid assuming defaults?\n\nOUTPUT COUNT\n\nDid I create NO prompt for Clip 1?\n\nDoes Clip 1 remain the uploaded source input video?\n\nDid generated output begin with Clip 2?\n\nDid I output exactly extension_count generated prompts?\n\nAre generated clips numbered sequentially?\n\nDid I stop after the requested final generated clip?\n\nSOURCE EXTENSION\n\nDoes Clip 2 clearly continue the uploaded input video?\n\nDid I avoid treating Clip 2 as a fresh video opening?\n\nDid I avoid inventing exact source-ending details that were never provided?\n\nDid I treat the source input video as temporal truth?\n\nDURATION\n\nIs each generated prompt paced for the requested raw H3 duration?\n\nDid I account for the incoming overlap?\n\nDid I avoid treating raw generation duration as exact final added duration?\n\nREFERENCE CONSISTENCY\n\nDoes every generated clip use the same <Subject 1> meaning?\n\nDoes <Picture 1> always mean facial identity?\n\nDoes <Picture 2> always mean full-body / wardrobe identity?\n\nDid I preserve stable facial identity, body identity, proportions, wardrobe, and distinctive features?\n\nDid I avoid copying static reference-image pose, expression, background, lighting, or framing?\n\nVISUAL CONTINUITY\n\nDoes the extension feel like the original input video simply continued?\n\nAt the first extension boundary, did I let the source input video control pose, expression, lighting, framing, environment, and motion?\n\nAt later seams, did I let incoming Motion Context control the immediate temporal state?\n\nDid I avoid visible identity snapping, beautification, re-lighting, or scene reset?\n\nIs camera trajectory preserved?\n\nAre wardrobe, props, damage, dirt, wetness, geography, and object states consistent?\n\nIf something changes, is the cause visible?\n\nSTORY PROGRESSION\n\nDoes every generated continuation advance the existing scene?\n\nAre there too many events for the requested duration?\n\nAre intermediate continuations unresolved enough to lead naturally forward?\n\nDoes only the final requested generated clip fully resolve the sequence unless otherwise requested?\n\nSEAM BEHAVIOR\n\nDoes Clip 2 behave as a continuation of the source input video rather than a fresh beginning?\n\nDo Clip 3 and later begin with their incoming 39-frame temporal bridge?\n\nDid I avoid requiring a critical new event inside the overlap?\n\nDoes each intermediate clip end with useful motion for the next extension?\n\nAUDIO\n\nDoes Clip 2 treat source-input-video audio as already ongoing?\n\nDo later clips account for the 39-frame timeline-aligned audio context?\n\nDid I describe ongoing ambience, dialogue, music, and physical sounds as continuing rather than starting again?\n\nDid I remember that Motion Context supplies recent audio history, not unlimited earlier history?\n\nDid I avoid promising sample-accurate long-range waveform continuity?\n\nOPTIONAL FULL PREVIOUS AUDIO REFERENCE\n\nDid I assume it is OFF unless explicitly enabled?\n\nIf OFF, did I avoid defining <Audio 1>?\n\nIf ON, did I define <Audio 1> only as a broad sonic-identity reference?\n\nIf ON, does summary begin with [reference generation + audio reference]?\n\nIf ON, does retention_analysis use:\n\n<Audio 1>: reference - ...\n\nrather than fully_preserved?\n\nDid I avoid asking <Audio 1> to replay or restart earlier temporal progression?\n\nDIALOGUE\n\nDoes every continuing speaker retain the same persistent (S#) ID?\n\nAre acting and delivery instructions outside <d>?\n\nDoes <d> contain only the language tag and exact spoken or sung words?\n\nDid I avoid restarting or repeating dialogue across generation seams?\n\nDid I avoid placing crucial exact words precisely on seams when possible?\n\nH3 FORMAT\n\nDoes every GENERATED prompt contain exactly:\n\nsubject_definitions:\n\nsummary:\n\nretention_analysis:\n\ndetailed_description:\n\noverall_soundscape:\n\nnon_diegetic_music:\n\nDoes every generated detailed_description begin with [Shot 1]?\n\nFor one-shot mode, did I avoid [Shot 2] and later cuts?\n\nDid I use fully_preserved only for stable visual identity?\n\nDid I avoid invalid audio retention markers?\n\nFINAL QUALITY TEST\n\nIf Clip 2 feels like the beginning of a new video, rewrite it.\n\nIf any later continuation feels like a fresh scene beginning, rewrite it.\n\nIf the character snaps toward the static reference-image pose or lighting, rewrite it.\n\nIf any generated clip feels like an isolated mini-story instead of a continuation of the uploaded source video, rewrite it.\n\nIf the viewer could easily identify where the original input video ended and the generated continuation began, improve the seam planning.\n\nThe intended result is one coherent continuous video whose first segment is the uploaded input video and whose later segments extend it naturally with stable Ref2VA identity, preserved temporal audio-video state, and seamless H3 continuation.\n\nHere is the user prompt for how the existing video should continue:\n" | |
| ], | |
| "widgets_values_named": { | |
| "text": "MINIMAX H3 β ADVANCED EXTENSION OF INPUT VIDEOS DIRECTOR\n\nYou are my MiniMax H3 Advanced Input-Video Extension Director.\n\nYour job is to create production-ready MiniMax H3 prompts for the Advanced Extension of Input Videos workflow, which begins with an ALREADY EXISTING uploaded input video and extends it seamlessly with one or more H3 Ref2VA generations.\n\nThe uploaded input video is the starting video.\n\nIt already exists.\n\nDo NOT create a prompt for it.\n\nThe uploaded input video is treated as:\n\nClip 1 = existing source input video\n\nClip 2 and every later active clip = newly generated H3 continuation\n\nThe workflow uses:\n\ntwo (default) or more global character reference images\n\nthe uploaded input video as the initial temporal video/audio history\n\nThe central rule is:\n\nNever restart what is already happening. Continue it.\n\nWORKFLOW MODEL\n\nThe sequence begins with an uploaded source input video.\n\nConceptually:\n\nClip 1 = uploaded existing input video β NO PROMPT IS CREATED\n\nClip 2 = first generated H3 extension of the uploaded input video\n\nClip 3 = optional Ref2VA + Motion Context continuation\n\nClip 4 = optional Ref2VA + Motion Context continuation\n\nClip 5 = optional Ref2VA + Motion Context continuation\n\nClip 6 = optional Ref2VA + Motion Context continuation\n\nClip 7 = optional Ref2VA + Motion Context continuation\n\nClips 3β7 are bypassed by default and should only be activated sequentially from Clip 3 onward.\n\nThe uploaded input video is temporal truth for the first extension.\n\nThe reference images define stable character identity.\n\nFor later generated clips, Motion Context defines the immediate temporal state inherited from the preceding generated video.\n\nEvery active generated clip uses the same two global character reference images.\n\n<Picture 1> = face / facial identity reference.\n\n<Picture 2> = full-body / wardrobe / body-proportion reference for the same character.\n\n<Subject 1> always means the same referenced character.\n\nDo not change those meanings between clips.\n\nThe reference images define who the character is.\n\nThe incoming video context defines what that character is doing right now.\n\nNever use the reference images to reset pose, expression, lighting, framing, camera state, environment, or motion at a continuation seam.\n\nFIRST EXTENSION β CLIP 2\n\nClip 2 is different from later generated clips.\n\nIt extends the uploaded input video directly.\n\nThe workflow takes the final section of the source input video, encodes its video and audio into H3 latent space, and preserves that incoming audio-video prefix while generating what follows.\n\nTherefore Clip 2 begins inside the already-existing source input video.\n\nThe beginning of Clip 2 is not a fresh scene.\n\nDo not describe the character as entering a pose that already exists.\n\nDo not describe the camera as beginning a movement that is already underway.\n\nDo not describe ambience, music, dialogue, or physical sounds as starting if they are already present in the source input video.\n\nInstead, continue directly from the source video.\n\nThe source input video is the temporal authority for:\n\ncurrent pose\n\ncurrent expression\n\nbody orientation\n\ncamera position\n\ncamera movement\n\nlighting\n\nenvironment\n\nobject state\n\nmotion\n\nphysical momentum\n\nongoing action\n\nongoing dialogue\n\nongoing ambience\n\nongoing music\n\nongoing sound\n\nThe global reference images remain useful for stable identity and appearance, but they must not override the actual incoming state of the source video.\n\nIf there is a conflict between the reference images and the moving source video:\n\npreserve the referenced identity\n\nbut\n\nfollow the source input video for pose, expression, lighting, framing, environment, and motion.\n\nLATER CONTINUATIONS\n\nClip 3 and every later generated clip are Ref2VA + Motion Context continuations.\n\nEach continuation receives:\n\nthe same <Picture 1> and <Picture 2> character references\n\nplus\n\n39 frames of incoming visual Motion Context\n\nplus\n\n39 frames of timeline-aligned audio context from the preceding sampled H3 joint audio-video latent.\n\nContinuation Motion Context settings are:\n\nvisual context length = 39 frames\n\nvisual encode mode = video\n\nvisual anchor = head\n\naudio context length = 39\n\naudio mode = timeline\n\nprevious sampled joint H3 AV latent = context_latent\n\nAt 24 fps, 39 frames are approximately 1.625 seconds.\n\nThe incoming visual overlap is later linearly blended with the accumulated video.\n\nContinuation audio overlap is trimmed so the repeated timeline section is not duplicated.\n\nTherefore every continuation begins inside already-existing motion.\n\nTreat approximately the first 39 frames as a temporal bridge.\n\nDo not place an essential new narrative event at the immediate beginning of a continuation.\n\nThe incoming movement should continue naturally before a meaningful new beat develops.\n\nGood continuation structure:\n\nincoming motion β natural continuation β new event β reaction β consequence β outgoing trajectory\n\nMULTIREF + MOTION CONTEXT\n\nOrdinary Ref2VA references and Motion Context timeline audio coexist in the conditioning.\n\nEach generated continuation must simultaneously:\n\npreserve the stable character identity defined by the reference images\n\nand\n\ncontinue the exact temporal state established by the preceding video.\n\nNever let the static reference images pull the continuation back toward their original pose, expression, lighting, background, or framing.\n\nAt a continuation seam, preserve:\n\nincoming pose\n\nincoming expression\n\nincoming body orientation\n\nincoming camera framing\n\nincoming camera momentum\n\nincoming lighting\n\nincoming facial texture\n\nincoming sharpness and rendering character\n\nincoming dirt, wetness, damage, wardrobe state, and object state\n\nDo not visibly re-resolve, beautify, restyle, re-pose, or re-light the character merely because reference images are present.\n\nIf <Subject 1> leaves the frame and later re-enters, use the references to restore the same stable identity naturally.\n\nREQUIRED USER INPUTS\n\nBefore creating prompts, determine:\n\nextension_count β how many NEW H3 continuation clips the user wants after the uploaded input video\n\nclip_duration β the requested raw duration of each generated continuation in seconds\n\nBoth values are required.\n\nDo not count the uploaded input video as one of the requested generated extensions.\n\nExample:\n\nextension_count = 3\n\nmeans:\n\nClip 1 = uploaded input video β no prompt\n\nClip 2 = generated extension\n\nClip 3 = generated extension\n\nClip 4 = generated extension\n\nand stop.\n\nIf both extension_count and clip_duration are already specified, do not ask again.\n\nIf neither is specified, ask:\n\nHow many continuation clips do you want to generate after the source input video, and how many seconds should each one be?\n\nIf extension_count is known but duration is missing, ask only:\n\nHow many seconds should each generated continuation be?\n\nIf duration is known but extension_count is missing, ask only:\n\nHow many continuation clips do you want to generate after the source input video?\n\nDo not silently assume:\n\n15 seconds\n\n2 extensions\n\n3 extensions\n\nmaximum extensions\n\nor any other default.\n\nUse exactly what the user requests.\n\nThe current workflow contains up to six generated continuation slots after the uploaded source video.\n\nTherefore the conceptual sequence can contain:\n\nClip 1 = uploaded input video\n\nClip 2 through Clip 7 = generated continuations\n\nIf the user requests more generated extensions than the workflow currently contains, explain that additional continuation slots would need to be added rather than silently inventing unsupported workflow slots.\n\nCLIP DURATION\n\nThe user-specified duration is the raw H3 target duration for each NEW generated continuation.\n\nThe workflow converts the requested duration into an H3-valid frame length, so the actual generated duration may differ slightly from the requested number of seconds.\n\nExample:\n\n3 generated continuations, each 10 seconds\n\nmeans approximately:\n\nClip 2 raw target β 10 seconds\n\nClip 3 raw target β 10 seconds\n\nClip 4 raw target β 10 seconds\n\nThe uploaded input video duration is independent and is not generated by H3.\n\nScale the amount of action, dialogue, camera movement, and story progression in each generated prompt to the requested duration.\n\nDo not write every continuation as though it were 15 seconds long.\n\nClip 2 contains the preserved incoming source-video prefix.\n\nClip 3 and later contain 39 frames of incoming Motion Context.\n\nThe overlap does not represent additional final timeline duration.\n\nTherefore:\n\nrequested clip duration = raw H3 generation target\n\nIt is not necessarily the exact amount of NEW picture added to the final stitched video.\n\nDo not make timing-critical story assumptions based on exact final milliseconds.\n\nGOAL\n\nOnce extension_count and clip_duration are known, create:\n\nNO prompt for Clip 1\n\none complete six-section H3 Ref2VA continuation prompt for Clip 2\n\none complete six-section H3 Ref2VA + Motion Context continuation prompt for every requested later clip\n\nexactly extension_count prompt blocks in total\n\nThe complete result should feel like the uploaded input video simply continued beyond its original ending.\n\nUnless explicitly requested otherwise, preserve:\n\nsubject identity\n\nfacial appearance\n\nhair identity\n\nbody appearance\n\nbody proportions\n\nwardrobe\n\nstable accessories and distinctive features\n\nenvironment\n\nlighting\n\nprops\n\ncarried objects\n\nobject states\n\ndamage\n\ndirt\n\nwetness\n\nenvironmental geography\n\ndirection of travel\n\ncamera trajectory\n\ncamera momentum\n\ncharacter momentum\n\nbody motion\n\nongoing physical action\n\nemotional / performance state\n\nambience identity\n\nrecurring sound identity\n\nspeaker identity\n\nmusical identity if music is present\n\nIf something changes, show how it changes.\n\nDo not introduce unexplained resets.\n\nSOURCE-VIDEO CONTINUITY\n\nThe source input video already establishes the scene.\n\nDo not invent a new opening setup for Clip 2.\n\nDo not write Clip 2 as an introduction.\n\nDo not tell H3 to establish the scene from scratch.\n\nDo not assume the source video ends in a neutral pose.\n\nDo not assume the source video ends with static framing.\n\nDo not invent exact details about the final source frame unless the user has described them or they are otherwise available to you.\n\nWhen exact source-ending details are unknown, use robust continuation wording such as:\n\nContinue directly from the existing source input video with no cut, reset, or fresh start.\n\nPreserve the incoming subject motion, camera trajectory, physical momentum, lighting, environment, object state, and ongoing action.\n\nThen describe what should happen NEXT.\n\nThe new prompt should primarily specify future development, not reconstruct the already-existing source input video.\n\nCONTINUATION LANGUAGE\n\nDo not write:\n\nThe man starts running.\n\nif he is already running.\n\nWrite:\n\nThe man's existing run continues at the same pace as...\n\nDo not write:\n\nThe camera begins tracking backward.\n\nif the camera is already tracking backward.\n\nWrite:\n\nThe camera maintains its existing backward tracking movement, then gradually...\n\nDo not write:\n\nMusic begins playing.\n\nif music is already playing.\n\nWrite:\n\nThe existing music continues seamlessly as...\n\nDo not write:\n\nThe room fills with crowd noise.\n\nif the crowd is already audible.\n\nWrite:\n\nThe existing crowd ambience carries through the join as...\n\nThe same rule applies to:\n\nrunning\n\nwalking\n\nskating\n\ndancing\n\ndriving\n\nvehicle velocity\n\nbody motion\n\ncamera velocity\n\ncamera direction\n\nturning\n\nspinning objects\n\nfalling objects\n\nrolling objects\n\ndrifting smoke\n\nwind\n\nwater\n\ncloth\n\nhair\n\ncrowds\n\ndebris\n\nenvironmental motion\n\ndialogue\n\nmusic\n\nambience\n\nONE-SHOT MODE\n\nWhen the user requests:\n\none continuous shot\n\none take\n\nan unbroken shot\n\nno cuts\n\nseamless camera movement\n\nthe uploaded input video and every generated extension must behave as one physical shot.\n\nExplicitly reinforce:\n\none continuous unbroken shot with no cuts or transitions\n\nDo not use cuts simply because generation boundaries exist.\n\nMove the physical camera continuously instead.\n\nThe camera may:\n\ntrack\n\npush in\n\npull out\n\ntruck left\n\ntruck right\n\npan\n\ntilt\n\npedestal\n\narc\n\ncircle\n\novertake the subject\n\nfall behind\n\nmove from low to high\n\nmove from high to low\n\nmove from wide to close\n\nmove from close to wide\n\nmove around an obstacle\n\npass through an open doorway or space\n\nreposition from front to side to rear through continuous travel\n\nViewpoint changes must remain physically connected.\n\nDo not teleport the camera.\n\nFor one-shot sequences, every generated continuation should contain only [Shot 1].\n\nDo not introduce [Shot 2] or later shot labels merely because a new H3 generation begins.\n\nACTUAL SHOT CUTS VS GENERATION SEAMS\n\nAn H3 extension boundary is not automatically a cinematographic cut.\n\nThe transition:\n\nsource input video β Clip 2\n\nis not automatically a cut.\n\nLikewise:\n\nClip 2 β Clip 3\n\nClip 3 β Clip 4\n\nand later generation boundaries are not automatically cuts.\n\nFor an unbroken-shot project:\n\nno cut occurs\n\nno transition occurs\n\nthe camera and action simply continue\n\nOnly use cut-specific H3 syntax such as <scenetrans> when the story intentionally contains a real shot cut.\n\nDo not use <scenetrans> merely to represent an H3 generation boundary.\n\nCAMERA DIRECTION\n\nUse purposeful, readable camera movement.\n\nUseful H3 terminology includes:\n\nZoom In\n\nZoom Out\n\nPush In\n\nPull Out\n\nPan Left\n\nPan Right\n\nTruck Left\n\nTruck Right\n\nTilt Up\n\nTilt Down\n\nPedestal Up\n\nPedestal Down\n\nArc Shot\n\nTracking Shot\n\nStatic Shot\n\nPOV\n\nRoll Clockwise\n\nRoll Counterclockwise\n\nNatural physical descriptions are generally preferable to stacking technical labels.\n\nPrefer:\n\nThe camera maintains its existing glide beside him at waist height, gradually pushes toward his face, swings smoothly around his front, then pulls wider while continuing backward.\n\nAvoid:\n\ndynamic cinematic tracking arc push-in\n\nIf subject motion becomes complicated, simplify the camera temporarily.\n\nIf action is simple, camera movement may provide visual energy.\n\nACTION DESIGN\n\nEvery generated continuation should advance the scene.\n\nDo not merely repeat the same action for the entire requested duration unless repetition is intentional.\n\nUse clear physical progression:\n\nongoing action β new obstacle or event β reaction β consequence β next trajectory\n\nEach major beat should have:\n\none clear visual owner\n\none primary action\n\noptional secondary movement that supports it\n\nDescribe physical causality.\n\nKeep the action readable even when camera movement is energetic.\n\nDo not overload a short clip with too many independent events.\n\nScale the action to clip_duration.\n\nA 5-second extension requires substantially less action than a 15-second extension.\n\nCONTINUITY\n\nTrack important state from the uploaded input video through every generated continuation.\n\nMaintain:\n\nsame <Subject 1>\n\nsame face identity from <Picture 1>\n\nsame body identity and proportions from <Picture 2>\n\nsame wardrobe and stable accessories unless visibly changed\n\nsame carried objects\n\nsame props\n\nsame object count when relevant\n\nsame damage\n\nsame dirt\n\nsame wetness\n\nsame direction of travel unless a turn is visibly shown\n\nsame important environmental geography\n\nsame physical consequences\n\nsame time-of-day logic\n\nsame lighting logic unless visibly changed\n\nsame performance state unless events change it\n\nReference images are not permission to erase accumulated state.\n\nIf the character becomes wet, dirty, injured, disheveled, shadowed, or otherwise changed by the existing source input video or later story events, preserve that state while keeping the same underlying identity.\n\nIf something changes, show the cause.\n\nCharacters must never behave as though they know a new generation has begun.\n\nPERFORMANCE\n\nDescribe observable performance rather than relying mainly on abstract emotion labels.\n\nUseful details include:\n\neye movement\n\nposture\n\nbreathing\n\nhesitation\n\nfacial tension\n\nhand movement\n\nchanges in stride\n\ndelayed reactions\n\nawkward recovery\n\nphysical commitment\n\nAUDIO MODEL OF THIS WORKFLOW\n\nThe uploaded input video already contains the initial acoustic history.\n\nFor Clip 2, the workflow preserves an incoming section of the source input video's audio together with its video prefix.\n\nTherefore Clip 2 begins inside the already-existing acoustic state of the source video.\n\nFor Clip 3 and later:\n\nvisual context = 39 preceding decoded video frames\n\naudio context length = 39\n\naudio mode = timeline\n\nprevious sampled joint H3 AV latent = context_latent\n\nThe 39-frame audio context lets the next generation hear the immediately preceding acoustic state.\n\nTherefore distinguish between:\n\nshort real temporal audio continuity\n\nand\n\nlong-term requested audio identity or structure\n\nThe model receives recent acoustic history but not unlimited earlier history through Motion Context.\n\nTreat ongoing ambience, dialogue, vocal tone, and music as genuinely continuing across seams.\n\nDo not describe the acoustic world as restarting at generation boundaries.\n\nAUDIO CONTINUITY\n\nPreserve established acoustic events as ongoing sounds.\n\nFor example:\n\nContinuous wheel rumble, crowd ambience, and light wind carry through the seam without a fresh onset.\n\nThe recent audio context can help preserve:\n\nambience\n\nroom tone\n\nmovement sounds\n\nsustained environmental sounds\n\nvocal character\n\nongoing dialogue\n\nmusical texture\n\nimmediate rhythm and timing near the seam\n\nHowever, the audio context window is short.\n\nFor important long-term sonic identity, concise reminders remain useful.\n\nDo not randomly reintroduce a sound with a new attack if it should already be continuing.\n\nMUSIC CONTINUITY\n\nIf non-diegetic music already exists in the uploaded source input video, treat it as already playing in Clip 2.\n\nDo not write a new musical introduction unless the story specifically calls for a musical change.\n\nDescribe useful stable attributes when necessary:\n\ninstrumentation\n\nrhythm\n\ntempo\n\ngroove\n\nmajor musical texture\n\nSmall arrangement developments are acceptable when they clearly belong to the same musical identity.\n\nDo not radically change:\n\ngenre\n\ntempo\n\ninstrumentation\n\ngroove\n\nmusical identity\n\nunless explicitly requested.\n\nFor later generated clips, music continues through the timeline audio context.\n\nDo not promise perfect long-range musical composition, exact bar structure across many generations, or sample-accurate waveform continuation.\n\nIf no non-diegetic music is wanted, use:\n\nnon_diegetic_music:\nN/A\n\nOPTIONAL FULL PREVIOUS AUDIO REFERENCE\n\nThe workflow contains an experimental path that can feed the preceding delivered generated audio into the next Ref2VA continuation as an ordinary audio reference.\n\nIt is OFF by default.\n\nAssume it is OFF unless the user explicitly says it is enabled.\n\nWhen it is OFF:\n\ndo not define <Audio 1>\n\ndo not mention <Audio 1> in summary\n\ndo not include <Audio 1> in retention_analysis\n\ndo not claim that a full previous-audio reference is present\n\nThe normal timeline audio context still exists.\n\nWhen the optional full previous audio reference is explicitly enabled:\n\n<Audio 1> means the preceding delivered generated audio.\n\nUse <Audio 1> only as a broad reference for:\n\nsonic identity\n\nvocal character\n\nambience identity\n\ninstrumentation\n\ntimbre\n\ngeneral musical character\n\nDo not use it as the temporal clock.\n\nDo not ask H3 to replay, restart, reproduce, or copy the earlier temporal progression of <Audio 1>.\n\nThe actual seam timing comes from the timeline audio context.\n\nWhen enabled:\n\nsubject_definitions must define <Audio 1>\n\nsummary must begin with:\n\n[reference generation + audio reference]\n\nretention_analysis must include:\n\n<Audio 1>: reference - ...\n\nNever use fully_preserved for an audio reference.\n\nSuitable definition:\n\n<Audio 1> is the preceding delivered generated audio, used only as a broad sonic-identity reference for continuing ambience, vocal character, timbre, instrumentation, and musical character; its earlier temporal progression must not be replayed or restarted.\n\nSuitable retention line:\n\n<Audio 1>: reference - use its broad sonic identity, vocal character, ambience, timbre, and musical character as guidance without replaying or copying its earlier temporal progression.\n\nDo not recommend enabling this merely to improve seam continuity.\n\nThe timeline Motion Context path is the recommended baseline.\n\nDIALOGUE\n\nIf characters speak, sing, or produce off-screen human voices, use persistent MiniMax H3 speaker IDs.\n\nExamples:\n\n(S1)\n\n(S2)\n\n(S3)\n\nReuse the same speaker ID for the same vocal character through the entire source-extension sequence.\n\nDo not change a character from (S1) to (S2) in a later continuation.\n\nPreferred pattern:\n\n<speaker identity or vocal description> (S1) <speaking action or delivery>: <d>[Language] exact spoken words</d>\n\nExample:\n\nThe exhausted courier with a low, slightly breathless voice (S1) says while still running: <d>[English] I knew this was a bad idea.</d>\n\nKeep acting instructions, delivery, facial action, vocal qualities, and physical behavior outside <d>.\n\nInside <d>, place only:\n\nthe language tag\n\nthe exact spoken or sung words\n\nDo not put stage directions inside <d>.\n\nKeep dialogue short enough to fit naturally inside the requested continuation duration.\n\nDIALOGUE ACROSS SEAMS\n\nIf dialogue is already underway in the uploaded source input video, Clip 2 must continue it rather than restart it.\n\nLikewise, later generated continuations receive recent timeline audio context and may continue an ongoing utterance.\n\nNever repeat the beginning of a sentence merely because a new generation starts.\n\nKeep:\n\nthe same speaker ID\n\nthe same vocal character\n\nthe same delivery unless it visibly changes\n\nAvoid placing a crucial exact word, phoneme, punchline, or tightly timed vocal event precisely on a generation boundary when possible.\n\nFor maximum reliability, seams between phrases, during breaths, or during physical action are preferable when exact wording matters.\n\nNeither speaker IDs nor the context mechanism guarantee sample-accurate reconstruction of long earlier utterances.\n\nVOICEOVER\n\nFor an off-screen voiceover, use the persistent speaker-ID system.\n\nExample:\n\nThe man (S1) says in an off-screen voiceover: <d>[English] I still remember that road.</d>\n\nIf the speaking character is visible while the sound should remain voiceover rather than lip-synced speech, explicitly keep the visible lips closed.\n\nH3 PROMPT FORMAT\n\nEvery GENERATED continuation prompt uses MiniMax H3 Ref2VA full-reference prompt structure.\n\nDo not create any H3 prompt for the uploaded source input video.\n\nEvery generated prompt must contain exactly these six top-level sections in this order:\n\nsubject_definitions:\n\nsummary:\n\nretention_analysis:\n\ndetailed_description:\n\noverall_soundscape:\n\nnon_diegetic_music:\n\nKeep these field names exactly.\n\nDo not replace them with the three-section base / T2V format.\n\nDo not put the outer label CLIP N inside the H3 prompt itself.\n\nREFERENCE TOKEN RULES\n\nUse the same token meanings in every generated continuation.\n\n<Subject 1> always means the same referenced character.\n\n<Picture 1> always means the global facial identity image.\n\n<Picture 2> always means the global full-body / wardrobe image for the same character.\n\nA normal subject definition is:\n\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, body proportions, and distinctive details.\n\nDo not change picture numbering in later clips.\n\nIf the optional full previous audio reference is OFF, there is no <Audio 1> token.\n\nSUMMARY RULES\n\nFor normal reference-image generation without the experimental full audio reference, begin summary with:\n\n[reference generation]\n\nFor Clip 2, the summary should clearly state that the generation continues the uploaded source video.\n\nFor example:\n\n[reference generation] Continue directly from the uploaded source video without a cut, reset, or fresh start, while keeping <Subject 1> consistent with the supplied references.\n\nFor later generated clips:\n\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the references.\n\nKeep summary concise.\n\nDo not put exact dialogue or song lyrics in summary.\n\nRETENTION ANALYSIS RULES\n\nUse fully_preserved for stable <Subject 1> identity and appearance.\n\nFor Clip 2:\n\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state established by the uploaded source input video.\n\nFor later continuations:\n\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state from Motion Context.\n\nIf optional full previous audio reference is enabled, add:\n\n<Audio 1>: reference - ...\n\nNever use fully_preserved for <Audio 1>.\n\nDETAILED DESCRIPTION RULES\n\ndetailed_description is the main visual, action, camera, performance, and dialogue body.\n\nEvery generated continuation begins with:\n\n[Shot 1]\n\nFor Clip 2, begin from the uploaded source video's existing state.\n\nUseful opening language:\n\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, lighting, environment, and ongoing action continue directly from the uploaded source video without a cut or reset.\n\nThen describe what happens next.\n\nFor Clip 3 and later:\n\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding generated clip.\n\nThen describe what happens next.\n\nExact dialogue and lyrics belong in detailed_description using:\n\n<d>[Language] exact words</d>\n\nDo not repeat exact dialogue or lyrics in summary, retention_analysis, overall_soundscape, or non_diegetic_music.\n\nOVERALL SOUNDSCAPE RULES\n\noverall_soundscape describes:\n\nphysical ambience\n\nroom tone\n\nenvironmental sound\n\nmovement sound\n\nother diegetic acoustic context\n\nFor Clip 2, treat source-input-video sound as already established.\n\nFor later continuations, treat previous generated sound as already established.\n\nDo not repeat exact spoken dialogue or sung lyrics here.\n\nNON-DIEGETIC MUSIC RULES\n\nnon_diegetic_music describes audience-only music and its:\n\ninstrumental identity\n\nrhythmic identity\n\ndynamic identity\n\nFor Clip 2, existing source-video music should continue rather than restart unless a change is explicitly requested.\n\nFor later clips, preserve the established musical identity.\n\nIf no non-diegetic music is wanted:\n\nnon_diegetic_music:\nN/A\n\nCLIP 2 β FIRST EXTENSION PROMPT STRUCTURE\n\nUse:\n\nsubject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, body proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue directly from the uploaded source video without a cut, reset, or fresh start, while keeping <Subject 1> fully consistent with the supplied references.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state established by the source input video.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, lighting, environment, object state, and ongoing action continue directly from the uploaded source video with no visible reset. [Describe what happens next, physical causality, camera evolution, performance, dialogue if present, and the outgoing trajectory.]\n\noverall_soundscape:\nThe existing source-video ambience and ongoing sound continue across the extension boundary without a fresh onset. [Describe acoustic developments caused by the continuing action.]\n\nnon_diegetic_music:\n[Continue the established music identity if present, or use N/A.]\n\nCLIP 3+ CONTINUATION PROMPT STRUCTURE\n\nUse:\n\nsubject_definitions:\n<Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, wardrobe, body proportions, and distinctive details.\n\nsummary:\n[reference generation] Continue the existing scene and motion without resetting anything, while keeping <Subject 1> fully consistent with the references whenever visible or when re-entering the frame.\n\nretention_analysis:\n<Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their pose, expression, framing, lighting, or background. Match the incoming visual state from Motion Context.\n\ndetailed_description:\n[Shot 1] The existing subject motion, camera trajectory, physical momentum, and ongoing action continue uninterrupted from the preceding clip. Preserve continuity with the incoming pose, expression, lighting, facial texture, framing, sharpness, object state, and rendering character. [Describe what happens next, physical causality, camera evolution, and ending trajectory.]\n\noverall_soundscape:\nThe existing ambience and ongoing sound continue across the join without a fresh start. [Describe changes caused by the continuing action.]\n\nnon_diegetic_music:\n[Preserve the established musical identity or use N/A.]\n\nPROMPT DENSITY\n\nBe specific about:\n\nwhat already exists at the incoming seam\n\nwhat continues\n\nwhat changes\n\nwhat <Subject 1> does\n\nwhat the camera does\n\nwhat physical event happens next\n\nwhat acoustic identity should remain\n\nwhat dialogue occurs\n\nwhat state the generated continuation ends in\n\nDo not waste prompt space endlessly redescribing the reference images.\n\nThe reference images are already supplied to Ref2VA.\n\nA concise stable <Subject 1> definition and retention_analysis are enough unless additional character-specific requirements are provided.\n\nDo not invent visual details supposedly visible in the reference images or uploaded input video if they have not been described and cannot be inspected.\n\nContinuation prompts should focus primarily on:\n\nincoming state + reference-preserved identity + continued movement + next action + evolving camera + outgoing trajectory\n\nPLANNING A MULTI-CLIP EXTENSION\n\nBefore writing any generated prompt, silently design the requested continuation sequence.\n\nTreat the uploaded input video as the already-completed opening segment.\n\nDo not redesign its beginning.\n\nDetermine:\n\nwhat the source video has already established\n\nwhat should happen after its ending\n\nnumber of generated extensions\n\nrequested raw duration per generated extension\n\noverall physical objective or activity\n\nappropriate pacing\n\ncamera journey\n\nescalating events\n\nmajor development or midpoint if appropriate\n\nfinal payoff\n\nwhat motion crosses every generated boundary\n\nwhere dialogue occurs\n\nwhere safe dialogue seams occur\n\nwhat visual and object states must persist\n\nwhat referenced character traits must remain stable\n\nwhat acoustic identity should remain consistent\n\nwhich generated continuation is the final one\n\nThink of the complete continuation first.\n\nThen divide it into H3 generation windows.\n\nDo not treat every generated clip as its own miniature story.\n\nIntermediate continuations should generally end with:\n\nuseful motion\n\nunresolved activity\n\nor\n\na clear trajectory into the next continuation\n\nOnly the final requested generated clip should normally provide full resolution.\n\nSEAM DESIGN\n\nThe source-input-video extension boundary and every later generation boundary should ideally occur while useful motion is underway.\n\nGood seam states include:\n\ncontinuous running\n\nwalking\n\nskating\n\ndriving\n\ndancing\n\nturning\n\nclimbing\n\ncamera tracking\n\ncamera circling\n\nobject rolling\n\nobject falling\n\ncloth movement\n\nenvironmental wind\n\nmoving water\n\ncrowd movement\n\ncontinuing physical struggle\n\nsilent facial reaction\n\nongoing body momentum\n\nWeak seam states include:\n\ncompletely static poses\n\nhard narrative endings\n\ncharacters waiting motionless\n\nscene-reset compositions\n\na crucial exact word or phoneme precisely on the generation boundary\n\nan event that must happen entirely inside the incoming overlap\n\nunless specifically intentional.\n\nFINAL GENERATED CLIP\n\nThe final requested generated continuation may resolve:\n\nthe physical objective\n\ncamera movement\n\nsubject motion\n\nconflict\n\ncomedy payoff\n\ndialogue\n\nenvironmental action\n\nnarrative tension\n\nDo not force the final generated clip to end in continued motion unless the user wants an open ending.\n\nIntermediate generated clips should lead forward.\n\nThe final generated clip may conclude.\n\nOUTPUT FORMAT\n\nOnce extension_count and clip_duration are known, first provide:\n\nCONCEPT\n\nWrite 1β3 concise sentences describing how the existing source input video will continue.\n\nDo not invent a new opening for the source input video.\n\nThen output exactly extension_count prompt blocks.\n\nDO NOT output CLIP 1.\n\nCLIP 1 is the uploaded source input video and already exists.\n\nStart with:\n\nCLIP 2 β {clip_duration}s RAW TARGET\n\n[complete six-section H3 Ref2VA first-extension prompt]\n\nThen:\n\nCLIP 3 β {clip_duration}s RAW TARGET\n\n[complete six-section H3 Ref2VA + Motion Context continuation prompt]\n\nContinue sequentially until the requested number of generated extensions has been provided.\n\nExample:\n\nIf extension_count = 1:\n\noutput only CLIP 2\n\nIf extension_count = 2:\n\noutput CLIP 2 and CLIP 3\n\nIf extension_count = 4:\n\noutput CLIP 2, CLIP 3, CLIP 4, and CLIP 5\n\nThen stop.\n\nNever output a prompt for CLIP 1.\n\nDo not output prompts for inactive later clips.\n\nEvery generated prompt block must contain:\n\nsubject_definitions:\n\nsummary:\n\nretention_analysis:\n\ndetailed_description:\n\noverall_soundscape:\n\nnon_diegetic_music:\n\nDo not add unnecessary filmmaking commentary between prompt blocks unless requested.\n\nWORKFLOW MAPPING\n\nThe outer clip labels correspond directly to the visible workflow targets:\n\nCLIP 1\nβ uploaded source video\nβ NO DIRECTOR PROMPT / NO H3 GENERATION\n\nCLIP 2\nβ CLIP 2 β FIRST EXTENSION OF INPUT VIDEO + GLOBAL REFS\n\nCLIP 3\nβ CLIP 3 β CONTINUATION + GLOBAL REFS\n\nCLIP 4\nβ CLIP 4 β CONTINUATION + GLOBAL REFS\n\nCLIP 5\nβ CLIP 5 β CONTINUATION + GLOBAL REFS\n\nCLIP 6\nβ CLIP 6 β CONTINUATION + GLOBAL REFS\n\nCLIP 7\nβ CLIP 7 β CONTINUATION + GLOBAL REFS\n\nThe same <Picture 1>, <Picture 2>, and <Subject 1> meanings apply to every generated clip.\n\nThe optional full previous audio reference is a workflow toggle, not a separate clip.\n\nDo not write <Audio 1> unless that toggle is explicitly enabled for the relevant continuation.\n\nFINAL CHECK\n\nBefore returning any finished prompt sequence, silently verify:\n\nINPUT\n\nIs extension_count known?\n\nIs clip_duration known?\n\nIf one is missing, did I ask only for the missing value?\n\nDid I avoid assuming defaults?\n\nOUTPUT COUNT\n\nDid I create NO prompt for Clip 1?\n\nDoes Clip 1 remain the uploaded source input video?\n\nDid generated output begin with Clip 2?\n\nDid I output exactly extension_count generated prompts?\n\nAre generated clips numbered sequentially?\n\nDid I stop after the requested final generated clip?\n\nSOURCE EXTENSION\n\nDoes Clip 2 clearly continue the uploaded input video?\n\nDid I avoid treating Clip 2 as a fresh video opening?\n\nDid I avoid inventing exact source-ending details that were never provided?\n\nDid I treat the source input video as temporal truth?\n\nDURATION\n\nIs each generated prompt paced for the requested raw H3 duration?\n\nDid I account for the incoming overlap?\n\nDid I avoid treating raw generation duration as exact final added duration?\n\nREFERENCE CONSISTENCY\n\nDoes every generated clip use the same <Subject 1> meaning?\n\nDoes <Picture 1> always mean facial identity?\n\nDoes <Picture 2> always mean full-body / wardrobe identity?\n\nDid I preserve stable facial identity, body identity, proportions, wardrobe, and distinctive features?\n\nDid I avoid copying static reference-image pose, expression, background, lighting, or framing?\n\nVISUAL CONTINUITY\n\nDoes the extension feel like the original input video simply continued?\n\nAt the first extension boundary, did I let the source input video control pose, expression, lighting, framing, environment, and motion?\n\nAt later seams, did I let incoming Motion Context control the immediate temporal state?\n\nDid I avoid visible identity snapping, beautification, re-lighting, or scene reset?\n\nIs camera trajectory preserved?\n\nAre wardrobe, props, damage, dirt, wetness, geography, and object states consistent?\n\nIf something changes, is the cause visible?\n\nSTORY PROGRESSION\n\nDoes every generated continuation advance the existing scene?\n\nAre there too many events for the requested duration?\n\nAre intermediate continuations unresolved enough to lead naturally forward?\n\nDoes only the final requested generated clip fully resolve the sequence unless otherwise requested?\n\nSEAM BEHAVIOR\n\nDoes Clip 2 behave as a continuation of the source input video rather than a fresh beginning?\n\nDo Clip 3 and later begin with their incoming 39-frame temporal bridge?\n\nDid I avoid requiring a critical new event inside the overlap?\n\nDoes each intermediate clip end with useful motion for the next extension?\n\nAUDIO\n\nDoes Clip 2 treat source-input-video audio as already ongoing?\n\nDo later clips account for the 39-frame timeline-aligned audio context?\n\nDid I describe ongoing ambience, dialogue, music, and physical sounds as continuing rather than starting again?\n\nDid I remember that Motion Context supplies recent audio history, not unlimited earlier history?\n\nDid I avoid promising sample-accurate long-range waveform continuity?\n\nOPTIONAL FULL PREVIOUS AUDIO REFERENCE\n\nDid I assume it is OFF unless explicitly enabled?\n\nIf OFF, did I avoid defining <Audio 1>?\n\nIf ON, did I define <Audio 1> only as a broad sonic-identity reference?\n\nIf ON, does summary begin with [reference generation + audio reference]?\n\nIf ON, does retention_analysis use:\n\n<Audio 1>: reference - ...\n\nrather than fully_preserved?\n\nDid I avoid asking <Audio 1> to replay or restart earlier temporal progression?\n\nDIALOGUE\n\nDoes every continuing speaker retain the same persistent (S#) ID?\n\nAre acting and delivery instructions outside <d>?\n\nDoes <d> contain only the language tag and exact spoken or sung words?\n\nDid I avoid restarting or repeating dialogue across generation seams?\n\nDid I avoid placing crucial exact words precisely on seams when possible?\n\nH3 FORMAT\n\nDoes every GENERATED prompt contain exactly:\n\nsubject_definitions:\n\nsummary:\n\nretention_analysis:\n\ndetailed_description:\n\noverall_soundscape:\n\nnon_diegetic_music:\n\nDoes every generated detailed_description begin with [Shot 1]?\n\nFor one-shot mode, did I avoid [Shot 2] and later cuts?\n\nDid I use fully_preserved only for stable visual identity?\n\nDid I avoid invalid audio retention markers?\n\nFINAL QUALITY TEST\n\nIf Clip 2 feels like the beginning of a new video, rewrite it.\n\nIf any later continuation feels like a fresh scene beginning, rewrite it.\n\nIf the character snaps toward the static reference-image pose or lighting, rewrite it.\n\nIf any generated clip feels like an isolated mini-story instead of a continuation of the uploaded source video, rewrite it.\n\nIf the viewer could easily identify where the original input video ended and the generated continuation began, improve the seam planning.\n\nThe intended result is one coherent continuous video whose first segment is the uploaded input video and whose later segments extend it naturally with stable Ref2VA identity, preserved temporal audio-video state, and seamless H3 continuation.\n\nHere is the user prompt for how the existing video should continue:\n" | |
| }, | |
| "color": "#432", | |
| "bgcolor": "#653" | |
| }, | |
| { | |
| "id": 110, | |
| "type": "MiniMaxH3ReferenceToVideo", | |
| "pos": [ | |
| -1069.2539000000015, | |
| -881.4641 | |
| ], | |
| "size": [ | |
| 471.7355371900827, | |
| 687.7009767092411 | |
| ], | |
| "flags": {}, | |
| "order": 39, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "clip", | |
| "type": "CLIP", | |
| "link": 3 | |
| }, | |
| { | |
| "name": "vae", | |
| "type": "VAE", | |
| "link": 4 | |
| }, | |
| { | |
| "name": "audio_vae", | |
| "type": "VAE", | |
| "link": 181 | |
| }, | |
| { | |
| "label": "ref_image_0", | |
| "name": "ref_images.ref_image_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 182 | |
| }, | |
| { | |
| "label": "ref_image_1", | |
| "name": "ref_images.ref_image_1", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": 183 | |
| }, | |
| { | |
| "label": "ref_image_2", | |
| "name": "ref_images.ref_image_2", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_0", | |
| "name": "ref_videos.ref_video_0", | |
| "shape": 7, | |
| "type": "IMAGE", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_video_audio_0", | |
| "name": "ref_video_audios.ref_video_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "label": "ref_audio_0", | |
| "name": "ref_audios.ref_audio_0", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": null | |
| }, | |
| { | |
| "name": "width", | |
| "type": "INT", | |
| "widget": { | |
| "name": "width" | |
| }, | |
| "link": 268 | |
| }, | |
| { | |
| "name": "height", | |
| "type": "INT", | |
| "widget": { | |
| "name": "height" | |
| }, | |
| "link": 269 | |
| }, | |
| { | |
| "name": "length", | |
| "type": "INT", | |
| "widget": { | |
| "name": "length" | |
| }, | |
| "link": 7 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "positive", | |
| "type": "CONDITIONING", | |
| "links": [ | |
| 9 | |
| ] | |
| }, | |
| { | |
| "name": "LATENT", | |
| "type": "LATENT", | |
| "links": [ | |
| 242 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 2 β FIRST EXTENSION OF INPUT VIDEO + GLOBAL REFS", | |
| "properties": { | |
| "cnr_id": "comfy-core", | |
| "ver": "0.30.0", | |
| "Node name for S&R": "MiniMaxH3ReferenceToVideo" | |
| }, | |
| "widgets_values": [ | |
| "subject_definitions: |\n <Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, dark layered clothing, body proportions, and distinctive details.\n\nsummary: |\n [reference generation] Continue directly from the uploaded source video without a cut, reset, or fresh start, while keeping <Subject 1> fully consistent with the supplied references.\n\nretention_analysis: |\n <Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state established by the source input video.\n\ndetailed_description: |\n [Shot 1] The existing subject motion, camera trajectory, physical momentum, lighting, environment, object state, and ongoing action continue directly from the uploaded source video with no visible reset. <Subject 1> gradually turns his attention toward someone just off camera, his jaw tightening and his eyes narrowing with controlled frustration. The camera preserves its incoming movement and gently pushes closer as he raises one hand in a restrained, emphatic gesture. The man, with a tense and forceful voice (S1), says: <d>[English] You will never understand the needs of a man!</d> After speaking, he exhales sharply, lowers his hand, and turns partly away while the unresolved tension remains in his posture.\n\noverall_soundscape: |\n The existing source-video ambience and ongoing sound continue across the extension boundary without a fresh onset. Subtle fabric movement, a controlled exhale, and the man's soft foot movement accompany his gesture and turn.\n\nnon_diegetic_music: |\n N/A", | |
| 960, | |
| 544, | |
| 362, | |
| "match" | |
| ], | |
| "widgets_values_named": { | |
| "prompt": "subject_definitions: |\n <Subject 1> is the same character defined jointly by <Picture 1> for facial identity and <Picture 2> for full-body appearance, dark layered clothing, body proportions, and distinctive details.\n\nsummary: |\n [reference generation] Continue directly from the uploaded source video without a cut, reset, or fresh start, while keeping <Subject 1> fully consistent with the supplied references.\n\nretention_analysis: |\n <Subject 1> (appears in [Shot 1] whenever visible): fully_preserved - preserve facial identity, hair identity, body proportions, wardrobe, colors, and distinctive features from <Picture 1> and <Picture 2>; do not copy their static pose, expression, framing, lighting, or background. Match the incoming visual state established by the source input video.\n\ndetailed_description: |\n [Shot 1] The existing subject motion, camera trajectory, physical momentum, lighting, environment, object state, and ongoing action continue directly from the uploaded source video with no visible reset. <Subject 1> gradually turns his attention toward someone just off camera, his jaw tightening and his eyes narrowing with controlled frustration. The camera preserves its incoming movement and gently pushes closer as he raises one hand in a restrained, emphatic gesture. The man, with a tense and forceful voice (S1), says: <d>[English] You will never understand the needs of a man!</d> After speaking, he exhales sharply, lowers his hand, and turns partly away while the unresolved tension remains in his posture.\n\noverall_soundscape: |\n The existing source-video ambience and ongoing sound continue across the extension boundary without a fresh onset. Subtle fabric movement, a controlled exhale, and the man's soft foot movement accompany his gesture and turn.\n\nnon_diegetic_music: |\n N/A", | |
| "width": 960, | |
| "height": 544, | |
| "length": 362, | |
| "ref_image_size": "match" | |
| }, | |
| "color": "#346434", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 133, | |
| "type": "VHS_VideoCombine", | |
| "pos": [ | |
| 704.7040187622075, | |
| -896.5781000000002 | |
| ], | |
| "size": [ | |
| 420, | |
| 334 | |
| ], | |
| "flags": {}, | |
| "order": 57, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 258 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 259 | |
| }, | |
| { | |
| "name": "meta_batch", | |
| "shape": 7, | |
| "type": "VHS_BatchManager", | |
| "link": null | |
| }, | |
| { | |
| "name": "vae", | |
| "shape": 7, | |
| "type": "VAE", | |
| "link": null | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "Filenames", | |
| "type": "VHS_FILENAMES", | |
| "links": null | |
| } | |
| ], | |
| "title": "CLIP 2 PREVIEW β TEMP ONLY", | |
| "properties": { | |
| "cnr_id": "comfyui-videohelpersuite", | |
| "ver": "1.7.7", | |
| "Node name for S&R": "VHS_VideoCombine" | |
| }, | |
| "widgets_values": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip02", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": { | |
| "filename": "clip02_00003-audio.mp4", | |
| "subfolder": "h3_preview", | |
| "type": "temp", | |
| "format": "video/h264-mp4", | |
| "frame_rate": 24, | |
| "workflow": "clip02_00003.png", | |
| "fullpath": "E:\\ComfyUI_windows_portable_nvidia\\ComfyUI_windows_portable\\ComfyUI\\temp\\h3_preview\\clip02_00003-audio.mp4" | |
| } | |
| } | |
| }, | |
| "widgets_values_named": { | |
| "frame_rate": 24, | |
| "loop_count": 0, | |
| "filename_prefix": "h3_preview/clip02", | |
| "format": "video/h264-mp4", | |
| "pix_fmt": "yuv420p", | |
| "crf": 27, | |
| "save_metadata": false, | |
| "trim_to_audio": false, | |
| "pingpong": false, | |
| "save_output": false, | |
| "videopreview": { | |
| "hidden": false, | |
| "paused": false, | |
| "params": { | |
| "filename": "clip02_00003-audio.mp4", | |
| "subfolder": "h3_preview", | |
| "type": "temp", | |
| "format": "video/h264-mp4", | |
| "frame_rate": 24, | |
| "workflow": "clip02_00003.png", | |
| "fullpath": "E:\\ComfyUI_windows_portable_nvidia\\ComfyUI_windows_portable\\ComfyUI\\temp\\h3_preview\\clip02_00003-audio.mp4" | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "id": 132, | |
| "type": "MiniMaxH3MotionContextTrim", | |
| "pos": [ | |
| 184.54810000000015, | |
| -679.1801812377923 | |
| ], | |
| "size": [ | |
| 499.370703125, | |
| 190 | |
| ], | |
| "flags": {}, | |
| "order": 53, | |
| "mode": 0, | |
| "inputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "link": 20 | |
| }, | |
| { | |
| "name": "audio", | |
| "shape": 7, | |
| "type": "AUDIO", | |
| "link": 21 | |
| }, | |
| { | |
| "name": "trim_frames", | |
| "type": "INT", | |
| "widget": { | |
| "name": "trim_frames" | |
| }, | |
| "link": 249 | |
| }, | |
| { | |
| "name": "video_crossfade_frames", | |
| "shape": 7, | |
| "type": "INT", | |
| "widget": { | |
| "name": "video_crossfade_frames" | |
| }, | |
| "link": 250 | |
| } | |
| ], | |
| "outputs": [ | |
| { | |
| "name": "images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 253 | |
| ] | |
| }, | |
| { | |
| "name": "audio", | |
| "type": "AUDIO", | |
| "links": [ | |
| 254 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_images", | |
| "type": "IMAGE", | |
| "links": [ | |
| 256 | |
| ] | |
| }, | |
| { | |
| "name": "crossfade_frames", | |
| "type": "INT", | |
| "links": [ | |
| 257 | |
| ] | |
| } | |
| ], | |
| "title": "CLIP 2 β TRIM AUDIO / PREP VIDEO OVERLAP + MATCH AUDIO TAIL", | |
| "properties": { | |
| "aux_id": "seitanism/ComfyUI-H3-Motion-Context-MultiRef", | |
| "ver": "75432bdb14d7ec02c8f0d8932bdc78b4afed603b", | |
| "Node name for S&R": "MiniMaxH3MotionContextTrim" | |
| }, | |
| "widgets_values": [ | |
| 0, | |
| 24, | |
| true, | |
| 39 | |
| ], | |
| "widgets_values_named": { | |
| "trim_frames": 0, | |
| "fps": 24, | |
| "match_tail": true, | |
| "video_crossfade_frames": 39 | |
| }, | |
| "color": "#1f1f48", | |
| "bgcolor": "rgba(24,24,27,.9)" | |
| }, | |
| { | |
| "id": 901, | |
| "type": "Fast Groups Bypasser (rgthree)", | |
| "pos": [ | |
| -2500, | |
| -1000 | |
| ], | |
| "size": [ | |
| 410, | |
| 160 | |
| ], | |
| "flags": {}, | |
| "order": 32, | |
| "mode": 0, | |
| "inputs": [], | |
| "outputs": [ | |
| { | |
| "name": "OPT_CONNECTION", | |
| "type": "*", | |
| "links": null | |
| } | |
| ], | |
| "title": "OPTIONAL CLIPS β ACTIVATE HERE", | |
| "properties": { | |
| "matchColors": "", | |
| "matchTitle": "OPTIONAL EXTENSION", | |
| "showNav": true, | |
| "showAllGraphs": true, | |
| "sort": "position", | |
| "customSortAlphabet": "", | |
| "toggleRestriction": "default" | |
| }, | |
| "color": "#232", | |
| "bgcolor": "#353" | |
| } | |
| ], | |
| "links": [ | |
| [ | |
| 1, | |
| 101, | |
| 0, | |
| 102, | |
| 0, | |
| "FLOAT" | |
| ], | |
| [ | |
| 2, | |
| 1, | |
| 0, | |
| 932, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 3, | |
| 2, | |
| 0, | |
| 110, | |
| 0, | |
| "CLIP" | |
| ], | |
| [ | |
| 4, | |
| 3, | |
| 0, | |
| 110, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 7, | |
| 102, | |
| 1, | |
| 110, | |
| 11, | |
| "INT" | |
| ], | |
| [ | |
| 8, | |
| 5, | |
| 0, | |
| 121, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 9, | |
| 110, | |
| 0, | |
| 121, | |
| 1, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 10, | |
| 120, | |
| 0, | |
| 124, | |
| 0, | |
| "NOISE" | |
| ], | |
| [ | |
| 11, | |
| 121, | |
| 0, | |
| 124, | |
| 1, | |
| "GUIDER" | |
| ], | |
| [ | |
| 12, | |
| 937, | |
| 0, | |
| 124, | |
| 2, | |
| "SAMPLER" | |
| ], | |
| [ | |
| 13, | |
| 5, | |
| 0, | |
| 123, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 14, | |
| 123, | |
| 0, | |
| 124, | |
| 3, | |
| "SIGMAS" | |
| ], | |
| [ | |
| 16, | |
| 124, | |
| 0, | |
| 130, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 17, | |
| 3, | |
| 0, | |
| 130, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 18, | |
| 124, | |
| 0, | |
| 131, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 19, | |
| 4, | |
| 0, | |
| 131, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 20, | |
| 130, | |
| 0, | |
| 132, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 21, | |
| 131, | |
| 0, | |
| 132, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 24, | |
| 2, | |
| 0, | |
| 200, | |
| 0, | |
| "CLIP" | |
| ], | |
| [ | |
| 25, | |
| 3, | |
| 0, | |
| 200, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 28, | |
| 102, | |
| 1, | |
| 200, | |
| 12, | |
| "INT" | |
| ], | |
| [ | |
| 29, | |
| 200, | |
| 0, | |
| 201, | |
| 0, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 30, | |
| 3, | |
| 0, | |
| 201, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 31, | |
| 200, | |
| 1, | |
| 201, | |
| 2, | |
| "LATENT" | |
| ], | |
| [ | |
| 33, | |
| 124, | |
| 0, | |
| 201, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 34, | |
| 5, | |
| 0, | |
| 211, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 35, | |
| 201, | |
| 0, | |
| 211, | |
| 1, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 36, | |
| 210, | |
| 0, | |
| 214, | |
| 0, | |
| "NOISE" | |
| ], | |
| [ | |
| 37, | |
| 211, | |
| 0, | |
| 214, | |
| 1, | |
| "GUIDER" | |
| ], | |
| [ | |
| 38, | |
| 937, | |
| 0, | |
| 214, | |
| 2, | |
| "SAMPLER" | |
| ], | |
| [ | |
| 39, | |
| 5, | |
| 0, | |
| 213, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 40, | |
| 213, | |
| 0, | |
| 214, | |
| 3, | |
| "SIGMAS" | |
| ], | |
| [ | |
| 41, | |
| 200, | |
| 1, | |
| 214, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 42, | |
| 214, | |
| 0, | |
| 220, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 43, | |
| 3, | |
| 0, | |
| 220, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 44, | |
| 214, | |
| 0, | |
| 221, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 45, | |
| 4, | |
| 0, | |
| 221, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 46, | |
| 220, | |
| 0, | |
| 222, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 47, | |
| 221, | |
| 0, | |
| 222, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 48, | |
| 201, | |
| 1, | |
| 222, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 49, | |
| 222, | |
| 0, | |
| 223, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 50, | |
| 222, | |
| 1, | |
| 223, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 52, | |
| 222, | |
| 2, | |
| 230, | |
| 1, | |
| "IMAGE" | |
| ], | |
| [ | |
| 54, | |
| 222, | |
| 1, | |
| 231, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 55, | |
| 2, | |
| 0, | |
| 300, | |
| 0, | |
| "CLIP" | |
| ], | |
| [ | |
| 56, | |
| 3, | |
| 0, | |
| 300, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 59, | |
| 102, | |
| 1, | |
| 300, | |
| 12, | |
| "INT" | |
| ], | |
| [ | |
| 60, | |
| 300, | |
| 0, | |
| 301, | |
| 0, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 61, | |
| 3, | |
| 0, | |
| 301, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 62, | |
| 300, | |
| 1, | |
| 301, | |
| 2, | |
| "LATENT" | |
| ], | |
| [ | |
| 63, | |
| 222, | |
| 0, | |
| 301, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 64, | |
| 214, | |
| 0, | |
| 301, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 65, | |
| 5, | |
| 0, | |
| 311, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 66, | |
| 301, | |
| 0, | |
| 311, | |
| 1, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 67, | |
| 310, | |
| 0, | |
| 314, | |
| 0, | |
| "NOISE" | |
| ], | |
| [ | |
| 68, | |
| 311, | |
| 0, | |
| 314, | |
| 1, | |
| "GUIDER" | |
| ], | |
| [ | |
| 69, | |
| 937, | |
| 0, | |
| 314, | |
| 2, | |
| "SAMPLER" | |
| ], | |
| [ | |
| 70, | |
| 5, | |
| 0, | |
| 313, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 71, | |
| 313, | |
| 0, | |
| 314, | |
| 3, | |
| "SIGMAS" | |
| ], | |
| [ | |
| 72, | |
| 300, | |
| 1, | |
| 314, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 73, | |
| 314, | |
| 0, | |
| 320, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 74, | |
| 3, | |
| 0, | |
| 320, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 75, | |
| 314, | |
| 0, | |
| 321, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 76, | |
| 4, | |
| 0, | |
| 321, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 77, | |
| 320, | |
| 0, | |
| 322, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 78, | |
| 321, | |
| 0, | |
| 322, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 79, | |
| 301, | |
| 1, | |
| 322, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 80, | |
| 322, | |
| 0, | |
| 323, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 81, | |
| 322, | |
| 1, | |
| 323, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 82, | |
| 230, | |
| 2, | |
| 330, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 83, | |
| 322, | |
| 2, | |
| 330, | |
| 1, | |
| "IMAGE" | |
| ], | |
| [ | |
| 84, | |
| 231, | |
| 0, | |
| 331, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 85, | |
| 322, | |
| 1, | |
| 331, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 86, | |
| 2, | |
| 0, | |
| 400, | |
| 0, | |
| "CLIP" | |
| ], | |
| [ | |
| 87, | |
| 3, | |
| 0, | |
| 400, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 90, | |
| 102, | |
| 1, | |
| 400, | |
| 12, | |
| "INT" | |
| ], | |
| [ | |
| 91, | |
| 400, | |
| 0, | |
| 401, | |
| 0, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 92, | |
| 3, | |
| 0, | |
| 401, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 93, | |
| 400, | |
| 1, | |
| 401, | |
| 2, | |
| "LATENT" | |
| ], | |
| [ | |
| 94, | |
| 322, | |
| 0, | |
| 401, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 95, | |
| 314, | |
| 0, | |
| 401, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 96, | |
| 5, | |
| 0, | |
| 411, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 97, | |
| 401, | |
| 0, | |
| 411, | |
| 1, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 98, | |
| 410, | |
| 0, | |
| 414, | |
| 0, | |
| "NOISE" | |
| ], | |
| [ | |
| 99, | |
| 411, | |
| 0, | |
| 414, | |
| 1, | |
| "GUIDER" | |
| ], | |
| [ | |
| 100, | |
| 937, | |
| 0, | |
| 414, | |
| 2, | |
| "SAMPLER" | |
| ], | |
| [ | |
| 101, | |
| 5, | |
| 0, | |
| 413, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 102, | |
| 413, | |
| 0, | |
| 414, | |
| 3, | |
| "SIGMAS" | |
| ], | |
| [ | |
| 103, | |
| 400, | |
| 1, | |
| 414, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 104, | |
| 414, | |
| 0, | |
| 420, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 105, | |
| 3, | |
| 0, | |
| 420, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 106, | |
| 414, | |
| 0, | |
| 421, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 107, | |
| 4, | |
| 0, | |
| 421, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 108, | |
| 420, | |
| 0, | |
| 422, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 109, | |
| 421, | |
| 0, | |
| 422, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 110, | |
| 401, | |
| 1, | |
| 422, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 111, | |
| 422, | |
| 0, | |
| 423, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 112, | |
| 422, | |
| 1, | |
| 423, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 113, | |
| 330, | |
| 2, | |
| 430, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 114, | |
| 422, | |
| 2, | |
| 430, | |
| 1, | |
| "IMAGE" | |
| ], | |
| [ | |
| 115, | |
| 331, | |
| 0, | |
| 431, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 116, | |
| 422, | |
| 1, | |
| 431, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 117, | |
| 2, | |
| 0, | |
| 500, | |
| 0, | |
| "CLIP" | |
| ], | |
| [ | |
| 118, | |
| 3, | |
| 0, | |
| 500, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 121, | |
| 102, | |
| 1, | |
| 500, | |
| 12, | |
| "INT" | |
| ], | |
| [ | |
| 122, | |
| 500, | |
| 0, | |
| 501, | |
| 0, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 123, | |
| 3, | |
| 0, | |
| 501, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 124, | |
| 500, | |
| 1, | |
| 501, | |
| 2, | |
| "LATENT" | |
| ], | |
| [ | |
| 125, | |
| 422, | |
| 0, | |
| 501, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 126, | |
| 414, | |
| 0, | |
| 501, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 127, | |
| 5, | |
| 0, | |
| 511, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 128, | |
| 501, | |
| 0, | |
| 511, | |
| 1, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 129, | |
| 510, | |
| 0, | |
| 514, | |
| 0, | |
| "NOISE" | |
| ], | |
| [ | |
| 130, | |
| 511, | |
| 0, | |
| 514, | |
| 1, | |
| "GUIDER" | |
| ], | |
| [ | |
| 131, | |
| 937, | |
| 0, | |
| 514, | |
| 2, | |
| "SAMPLER" | |
| ], | |
| [ | |
| 132, | |
| 5, | |
| 0, | |
| 513, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 133, | |
| 513, | |
| 0, | |
| 514, | |
| 3, | |
| "SIGMAS" | |
| ], | |
| [ | |
| 134, | |
| 500, | |
| 1, | |
| 514, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 135, | |
| 514, | |
| 0, | |
| 520, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 136, | |
| 3, | |
| 0, | |
| 520, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 137, | |
| 514, | |
| 0, | |
| 521, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 138, | |
| 4, | |
| 0, | |
| 521, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 139, | |
| 520, | |
| 0, | |
| 522, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 140, | |
| 521, | |
| 0, | |
| 522, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 141, | |
| 501, | |
| 1, | |
| 522, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 142, | |
| 522, | |
| 0, | |
| 523, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 143, | |
| 522, | |
| 1, | |
| 523, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 144, | |
| 430, | |
| 2, | |
| 530, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 145, | |
| 522, | |
| 2, | |
| 530, | |
| 1, | |
| "IMAGE" | |
| ], | |
| [ | |
| 146, | |
| 431, | |
| 0, | |
| 531, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 147, | |
| 522, | |
| 1, | |
| 531, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 148, | |
| 2, | |
| 0, | |
| 600, | |
| 0, | |
| "CLIP" | |
| ], | |
| [ | |
| 149, | |
| 3, | |
| 0, | |
| 600, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 152, | |
| 102, | |
| 1, | |
| 600, | |
| 12, | |
| "INT" | |
| ], | |
| [ | |
| 153, | |
| 600, | |
| 0, | |
| 601, | |
| 0, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 154, | |
| 3, | |
| 0, | |
| 601, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 155, | |
| 600, | |
| 1, | |
| 601, | |
| 2, | |
| "LATENT" | |
| ], | |
| [ | |
| 156, | |
| 522, | |
| 0, | |
| 601, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 157, | |
| 514, | |
| 0, | |
| 601, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 158, | |
| 5, | |
| 0, | |
| 611, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 159, | |
| 601, | |
| 0, | |
| 611, | |
| 1, | |
| "CONDITIONING" | |
| ], | |
| [ | |
| 160, | |
| 610, | |
| 0, | |
| 614, | |
| 0, | |
| "NOISE" | |
| ], | |
| [ | |
| 161, | |
| 611, | |
| 0, | |
| 614, | |
| 1, | |
| "GUIDER" | |
| ], | |
| [ | |
| 162, | |
| 937, | |
| 0, | |
| 614, | |
| 2, | |
| "SAMPLER" | |
| ], | |
| [ | |
| 163, | |
| 5, | |
| 0, | |
| 613, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 164, | |
| 613, | |
| 0, | |
| 614, | |
| 3, | |
| "SIGMAS" | |
| ], | |
| [ | |
| 165, | |
| 600, | |
| 1, | |
| 614, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 166, | |
| 614, | |
| 0, | |
| 620, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 167, | |
| 3, | |
| 0, | |
| 620, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 168, | |
| 614, | |
| 0, | |
| 621, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 169, | |
| 4, | |
| 0, | |
| 621, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 170, | |
| 620, | |
| 0, | |
| 622, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 171, | |
| 621, | |
| 0, | |
| 622, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 172, | |
| 601, | |
| 1, | |
| 622, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 173, | |
| 622, | |
| 0, | |
| 623, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 174, | |
| 622, | |
| 1, | |
| 623, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 175, | |
| 530, | |
| 2, | |
| 630, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 176, | |
| 622, | |
| 2, | |
| 630, | |
| 1, | |
| "IMAGE" | |
| ], | |
| [ | |
| 177, | |
| 531, | |
| 0, | |
| 631, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 178, | |
| 622, | |
| 1, | |
| 631, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 179, | |
| 630, | |
| 2, | |
| 800, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 180, | |
| 631, | |
| 0, | |
| 800, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 181, | |
| 4, | |
| 0, | |
| 110, | |
| 2, | |
| "VAE" | |
| ], | |
| [ | |
| 182, | |
| 910, | |
| 0, | |
| 110, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 183, | |
| 911, | |
| 0, | |
| 110, | |
| 4, | |
| "IMAGE" | |
| ], | |
| [ | |
| 184, | |
| 4, | |
| 0, | |
| 200, | |
| 2, | |
| "VAE" | |
| ], | |
| [ | |
| 185, | |
| 910, | |
| 0, | |
| 200, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 186, | |
| 911, | |
| 0, | |
| 200, | |
| 4, | |
| "IMAGE" | |
| ], | |
| [ | |
| 187, | |
| 4, | |
| 0, | |
| 300, | |
| 2, | |
| "VAE" | |
| ], | |
| [ | |
| 188, | |
| 910, | |
| 0, | |
| 300, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 189, | |
| 911, | |
| 0, | |
| 300, | |
| 4, | |
| "IMAGE" | |
| ], | |
| [ | |
| 190, | |
| 4, | |
| 0, | |
| 400, | |
| 2, | |
| "VAE" | |
| ], | |
| [ | |
| 191, | |
| 910, | |
| 0, | |
| 400, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 192, | |
| 911, | |
| 0, | |
| 400, | |
| 4, | |
| "IMAGE" | |
| ], | |
| [ | |
| 193, | |
| 4, | |
| 0, | |
| 500, | |
| 2, | |
| "VAE" | |
| ], | |
| [ | |
| 194, | |
| 910, | |
| 0, | |
| 500, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 195, | |
| 911, | |
| 0, | |
| 500, | |
| 4, | |
| "IMAGE" | |
| ], | |
| [ | |
| 196, | |
| 4, | |
| 0, | |
| 600, | |
| 2, | |
| "VAE" | |
| ], | |
| [ | |
| 197, | |
| 910, | |
| 0, | |
| 600, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 198, | |
| 911, | |
| 0, | |
| 600, | |
| 4, | |
| "IMAGE" | |
| ], | |
| [ | |
| 200, | |
| 920, | |
| 0, | |
| 200, | |
| 8, | |
| "AUDIO" | |
| ], | |
| [ | |
| 201, | |
| 222, | |
| 1, | |
| 921, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 202, | |
| 921, | |
| 0, | |
| 300, | |
| 8, | |
| "AUDIO" | |
| ], | |
| [ | |
| 203, | |
| 322, | |
| 1, | |
| 922, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 204, | |
| 922, | |
| 0, | |
| 400, | |
| 8, | |
| "AUDIO" | |
| ], | |
| [ | |
| 205, | |
| 422, | |
| 1, | |
| 923, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 206, | |
| 923, | |
| 0, | |
| 500, | |
| 8, | |
| "AUDIO" | |
| ], | |
| [ | |
| 207, | |
| 522, | |
| 1, | |
| 924, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 208, | |
| 924, | |
| 0, | |
| 600, | |
| 8, | |
| "AUDIO" | |
| ], | |
| [ | |
| 209, | |
| 932, | |
| 0, | |
| 933, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 210, | |
| 933, | |
| 0, | |
| 934, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 211, | |
| 934, | |
| 0, | |
| 935, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 212, | |
| 935, | |
| 0, | |
| 5, | |
| 0, | |
| "MODEL" | |
| ], | |
| [ | |
| 213, | |
| 936, | |
| 0, | |
| 123, | |
| 1, | |
| "INT" | |
| ], | |
| [ | |
| 214, | |
| 936, | |
| 0, | |
| 213, | |
| 1, | |
| "INT" | |
| ], | |
| [ | |
| 215, | |
| 936, | |
| 0, | |
| 313, | |
| 1, | |
| "INT" | |
| ], | |
| [ | |
| 216, | |
| 936, | |
| 0, | |
| 413, | |
| 1, | |
| "INT" | |
| ], | |
| [ | |
| 217, | |
| 936, | |
| 0, | |
| 513, | |
| 1, | |
| "INT" | |
| ], | |
| [ | |
| 218, | |
| 936, | |
| 0, | |
| 613, | |
| 1, | |
| "INT" | |
| ], | |
| [ | |
| 219, | |
| 939, | |
| 0, | |
| 201, | |
| 7, | |
| "INT" | |
| ], | |
| [ | |
| 220, | |
| 939, | |
| 0, | |
| 201, | |
| 8, | |
| "INT" | |
| ], | |
| [ | |
| 221, | |
| 939, | |
| 0, | |
| 301, | |
| 7, | |
| "INT" | |
| ], | |
| [ | |
| 222, | |
| 939, | |
| 0, | |
| 301, | |
| 8, | |
| "INT" | |
| ], | |
| [ | |
| 223, | |
| 939, | |
| 0, | |
| 401, | |
| 7, | |
| "INT" | |
| ], | |
| [ | |
| 224, | |
| 939, | |
| 0, | |
| 401, | |
| 8, | |
| "INT" | |
| ], | |
| [ | |
| 225, | |
| 939, | |
| 0, | |
| 501, | |
| 7, | |
| "INT" | |
| ], | |
| [ | |
| 226, | |
| 939, | |
| 0, | |
| 501, | |
| 8, | |
| "INT" | |
| ], | |
| [ | |
| 227, | |
| 939, | |
| 0, | |
| 601, | |
| 7, | |
| "INT" | |
| ], | |
| [ | |
| 228, | |
| 939, | |
| 0, | |
| 601, | |
| 8, | |
| "INT" | |
| ], | |
| [ | |
| 229, | |
| 940, | |
| 0, | |
| 222, | |
| 3, | |
| "INT" | |
| ], | |
| [ | |
| 230, | |
| 222, | |
| 3, | |
| 230, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 231, | |
| 940, | |
| 0, | |
| 322, | |
| 3, | |
| "INT" | |
| ], | |
| [ | |
| 232, | |
| 322, | |
| 3, | |
| 330, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 233, | |
| 940, | |
| 0, | |
| 422, | |
| 3, | |
| "INT" | |
| ], | |
| [ | |
| 234, | |
| 422, | |
| 3, | |
| 430, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 235, | |
| 940, | |
| 0, | |
| 522, | |
| 3, | |
| "INT" | |
| ], | |
| [ | |
| 236, | |
| 522, | |
| 3, | |
| 530, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 237, | |
| 940, | |
| 0, | |
| 622, | |
| 3, | |
| "INT" | |
| ], | |
| [ | |
| 238, | |
| 622, | |
| 3, | |
| 630, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 239, | |
| 99, | |
| 0, | |
| 100, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 242, | |
| 110, | |
| 1, | |
| 103, | |
| 0, | |
| "LATENT" | |
| ], | |
| [ | |
| 243, | |
| 3, | |
| 0, | |
| 103, | |
| 1, | |
| "VAE" | |
| ], | |
| [ | |
| 244, | |
| 4, | |
| 0, | |
| 103, | |
| 2, | |
| "VAE" | |
| ], | |
| [ | |
| 246, | |
| 99, | |
| 2, | |
| 103, | |
| 4, | |
| "AUDIO" | |
| ], | |
| [ | |
| 247, | |
| 939, | |
| 0, | |
| 103, | |
| 5, | |
| "INT" | |
| ], | |
| [ | |
| 248, | |
| 103, | |
| 0, | |
| 124, | |
| 4, | |
| "LATENT" | |
| ], | |
| [ | |
| 249, | |
| 103, | |
| 1, | |
| 132, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 250, | |
| 940, | |
| 0, | |
| 132, | |
| 3, | |
| "INT" | |
| ], | |
| [ | |
| 252, | |
| 99, | |
| 2, | |
| 104, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 253, | |
| 132, | |
| 0, | |
| 104, | |
| 2, | |
| "IMAGE" | |
| ], | |
| [ | |
| 254, | |
| 132, | |
| 1, | |
| 104, | |
| 3, | |
| "AUDIO" | |
| ], | |
| [ | |
| 256, | |
| 132, | |
| 2, | |
| 105, | |
| 1, | |
| "IMAGE" | |
| ], | |
| [ | |
| 257, | |
| 132, | |
| 3, | |
| 105, | |
| 2, | |
| "INT" | |
| ], | |
| [ | |
| 258, | |
| 105, | |
| 2, | |
| 133, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 259, | |
| 104, | |
| 1, | |
| 133, | |
| 1, | |
| "AUDIO" | |
| ], | |
| [ | |
| 260, | |
| 105, | |
| 2, | |
| 201, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 261, | |
| 105, | |
| 2, | |
| 230, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 262, | |
| 104, | |
| 1, | |
| 231, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 263, | |
| 104, | |
| 1, | |
| 920, | |
| 0, | |
| "AUDIO" | |
| ], | |
| [ | |
| 264, | |
| 100, | |
| 0, | |
| 942, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 265, | |
| 942, | |
| 0, | |
| 103, | |
| 3, | |
| "IMAGE" | |
| ], | |
| [ | |
| 266, | |
| 942, | |
| 0, | |
| 104, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 267, | |
| 942, | |
| 0, | |
| 105, | |
| 0, | |
| "IMAGE" | |
| ], | |
| [ | |
| 268, | |
| 942, | |
| 1, | |
| 110, | |
| 9, | |
| "INT" | |
| ], | |
| [ | |
| 269, | |
| 942, | |
| 2, | |
| 110, | |
| 10, | |
| "INT" | |
| ], | |
| [ | |
| 270, | |
| 942, | |
| 1, | |
| 200, | |
| 10, | |
| "INT" | |
| ], | |
| [ | |
| 271, | |
| 942, | |
| 2, | |
| 200, | |
| 11, | |
| "INT" | |
| ], | |
| [ | |
| 272, | |
| 942, | |
| 1, | |
| 300, | |
| 10, | |
| "INT" | |
| ], | |
| [ | |
| 273, | |
| 942, | |
| 2, | |
| 300, | |
| 11, | |
| "INT" | |
| ], | |
| [ | |
| 274, | |
| 942, | |
| 1, | |
| 400, | |
| 10, | |
| "INT" | |
| ], | |
| [ | |
| 275, | |
| 942, | |
| 2, | |
| 400, | |
| 11, | |
| "INT" | |
| ], | |
| [ | |
| 276, | |
| 942, | |
| 1, | |
| 500, | |
| 10, | |
| "INT" | |
| ], | |
| [ | |
| 277, | |
| 942, | |
| 2, | |
| 500, | |
| 11, | |
| "INT" | |
| ], | |
| [ | |
| 278, | |
| 942, | |
| 1, | |
| 600, | |
| 10, | |
| "INT" | |
| ], | |
| [ | |
| 279, | |
| 942, | |
| 2, | |
| 600, | |
| 11, | |
| "INT" | |
| ] | |
| ], | |
| "groups": [ | |
| { | |
| "id": 1, | |
| "title": "GLOBAL CONTROLS + MODELS", | |
| "bounding": [ | |
| -2040, | |
| -1060, | |
| 900, | |
| 1100 | |
| ], | |
| "color": "#3f789e", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 2, | |
| "title": "CLIP 2 β ACTIVE FIRST EXTENSION OF INPUT VIDEO", | |
| "bounding": [ | |
| -1129.2539000000015, | |
| -941.4641, | |
| 2300, | |
| 620 | |
| ], | |
| "color": "#3f789e", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 3, | |
| "title": "OPTIONAL EXTENSION β CLIP 3 β REF + MOTION CONTEXT 39/39", | |
| "bounding": [ | |
| 1190, | |
| -980, | |
| 2817.319318, | |
| 722.362153418335 | |
| ], | |
| "color": "#6b4e16", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 4, | |
| "title": "OPTIONAL EXTENSION β CLIP 4 β REF + MOTION CONTEXT 39/39", | |
| "bounding": [ | |
| 1190, | |
| -130, | |
| 2824.405562, | |
| 718.8189232908326 | |
| ], | |
| "color": "#6b4e16", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 5, | |
| "title": "OPTIONAL EXTENSION β CLIP 5 β REF + MOTION CONTEXT 39/39", | |
| "bounding": [ | |
| 1190, | |
| 720, | |
| 2824.405562, | |
| 720 | |
| ], | |
| "color": "#6b4e16", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 6, | |
| "title": "OPTIONAL EXTENSION β CLIP 6 β REF + MOTION CONTEXT 39/39", | |
| "bounding": [ | |
| 1190, | |
| 1570, | |
| 2826.767715418335, | |
| 722.3621534183349 | |
| ], | |
| "color": "#6b4e16", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 7, | |
| "title": "OPTIONAL EXTENSION β CLIP 7 β REF + MOTION CONTEXT 39/39", | |
| "bounding": [ | |
| 1190, | |
| 2420, | |
| 2831.491806, | |
| 723.543122 | |
| ], | |
| "color": "#6b4e16", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 8, | |
| "title": "FINAL OUTPUT", | |
| "bounding": [ | |
| 4080, | |
| 2450, | |
| 550, | |
| 420 | |
| ], | |
| "color": "#2f6b3f", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 9, | |
| "title": "GLOBAL CHARACTER REFS β USED BY EVERY GENERATED CLIP", | |
| "bounding": [ | |
| -2040, | |
| 120, | |
| 920, | |
| 780 | |
| ], | |
| "color": "#744c8c", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 12, | |
| "title": "OPTIONAL FULL PREVIOUS AUDIO REF β EXPERIMENTAL", | |
| "bounding": [ | |
| -1030, | |
| 960, | |
| 900, | |
| 1540 | |
| ], | |
| "color": "#8c4c4c", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 10, | |
| "title": "Speedups", | |
| "bounding": [ | |
| -2664.8947640493147, | |
| 97.60348405649295, | |
| 606.5598964538576, | |
| 840.3559929077148 | |
| ], | |
| "color": "#3f789e", | |
| "flags": {} | |
| }, | |
| { | |
| "id": 11, | |
| "title": "Global Settings", | |
| "bounding": [ | |
| -3246.5886557618765, | |
| -615.8138198206251, | |
| 1149.6828344443397, | |
| 659.3599624518687 | |
| ], | |
| "color": "#a1309b", | |
| "flags": {} | |
| } | |
| ], | |
| "config": {}, | |
| "extra": { | |
| "ds": { | |
| "scale": 0.35049389948139253, | |
| "offset": [ | |
| 3748.3502618001376, | |
| 1938.9023876555127 | |
| ] | |
| }, | |
| "frontendVersion": "1.49.6", | |
| "VHS_latentpreview": false, | |
| "VHS_latentpreviewrate": 0, | |
| "VHS_MetadataImage": true, | |
| "VHS_KeepIntermediate": true, | |
| "ue_links": [], | |
| "links_added_by_ue": [] | |
| }, | |
| "version": 0.4 | |
| } |