File size: 11,647 Bytes
4f4aa26
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
"""Build LTX-2.5 first-and-last-frame graphs in ComfyUI API format.

WHY THIS EXISTS
LTX-2.5 has a stock "First & Last Frame to Video" workflow, but it is a UI
subgraph (awkward to queue programmatically) and it *generates* audio from an
empty latent. The image+audio path that exists for 2.3 was rebuilt for 2.5 as a
hybrid (first frame via LTXVImgToVideoInplace, last via LTXVAddGuide). This
unifies both into one API-format generator using the stock guide pattern β€”
BOTH frames through chained LTXVAddGuide at frame_idx 0 and -1 β€” and adds the
two things neither had: a LoRA slot and a separate audio save.

TWO MODES
  gen    LTXVEmptyLatentAudio β€” the model generates the audio along with the
         video. This is the mode for "write the line in the prompt, let the
         base model speak it", whose output you then feed to the AKUSPACE a2a
         pass. The base model's own reverb is inconsistent, which is exactly
         the before/after story.
  drive  Existing audio is VAE-encoded and pinned by a zero SolidMask, so the
         video denoises around fixed audio. Use when the audio is already final
         (e.g. an a2a pass has already roomed it).

TWO PRESETS β€” PICK THE ONE THAT MATCHES THE WEIGHTS
LTX-2.5 ships two transformers that need opposite sampling regimes, and mixing
them produces soft, washed-out frames that look like a broken model but are just
the wrong preset:

  distilled (default)  int8 distilled transformer + int8 gemma, ~8 steps with
                       explicit ManualSigmas, LTXVDualCFGGuider at 1/1,
                       SamplerEulerAncestral. This mirrors the shipped ComfyUI
                       preset and is the regime verified to produce clean output.
  dev                  bf16 dev transformer + bf16 gemma, ~24 steps at CFG 4
                       through LTXVScheduler and plain CFGGuider.

An earlier revision of this file argued for CFGGuider over LTXVDualCFGGuider on
the grounds that the latter was absent from a schema dump. That was wrong: the
dump had been taken before the node pack finished loading. Both are present on a
correctly installed pack.

Every node type and input name below is validated against a live object_info
dump before the graph is written β€” see validate(). Never ship a graph that has
not been through it; wrong input names are the trap that costs the most time.
"""

import argparse
import json
import sys
from pathlib import Path

W, H = 960, 544
TRANSFORMER = "ltx-2.5-22b-dev-transformer-bf16.safetensors"
TEXT_ENCODER = "gemma4-12b-with-proj-ltx-2.5-bf16.safetensors"
# The distilled pair the shipped preset uses. Different sampling regime β€” see
# WORKING_CONFIG.md. Mixing the two is what produces "distorted" output.
TRANSFORMER_DISTILLED = "ltx-2.5-22b-distilled-transformer-comfy-int8-convrot.safetensors"
TEXT_ENCODER_INT8 = "gemma4-12b-with-proj-ltx-2.5-comfy-int8-convrot.safetensors"
SIGMAS_DISTILLED = "1.0, 0.99375, 0.9875, 0.98125, 0.975, 0.909375, 0.725, 0.421875, 0.0"
VIDEO_VAE = "ltx-2.5-video-vae-bf16.safetensors"
AUDIO_VAE = "ltx-2.5-audio-vae-bf16.safetensors"
NEGATIVE = ("distorted, clipped, metallic, unintelligible speech, changed words, "
            "changed rhythm, deformed hands, extra fingers, watermark, text")


def build(a) -> dict:
    g: dict = {}

    def n(nid, cls, **inputs):
        g[str(nid)] = {"class_type": cls, "inputs": inputs}

    unet = TRANSFORMER_DISTILLED if a.preset == "distilled" else TRANSFORMER
    clip = TEXT_ENCODER_INT8 if a.preset == "distilled" else TEXT_ENCODER
    n(1, "UNETLoader", unet_name=unet, weight_dtype="default")
    model = ["1", 0]
    if a.lora:
        n(2, "LoraLoaderModelOnly", model=["1", 0], lora_name=a.lora, strength_model=a.lora_strength)
        model = ["2", 0]

    n(3, "CLIPLoader", clip_name=clip, type="ltxv")
    n(4, "CLIPTextEncode", clip=["3", 0], text=a.prompt)
    n(5, "CLIPTextEncode", clip=["3", 0], text=a.negative)
    n(6, "LTXVConditioning", positive=["4", 0], negative=["5", 0], frame_rate=float(a.fps))

    n(7, "VAELoader", vae_name=VIDEO_VAE)
    n(8, "VAELoader", vae_name=AUDIO_VAE)

    # first frame
    n(9, "LoadImage", image=a.first_image)
    n(10, "ImageScale", image=["9", 0], upscale_method="lanczos", width=a.width, height=a.height, crop="center")
    n(11, "LTXVPreprocess", image=["10", 0], img_compression=18)
    # last frame
    n(12, "LoadImage", image=a.last_image)
    n(13, "ImageScale", image=["12", 0], upscale_method="lanczos", width=a.width, height=a.height, crop="center")
    n(14, "LTXVPreprocess", image=["13", 0], img_compression=18)

    n(15, "EmptyLTXVLatentVideo", width=a.width, height=a.height, length=a.frames, batch_size=1)

    # Chained guides: the second consumes the first's conditioning AND latent.
    # frame_idx -1 is the stock idiom for "last"; it is what the shipped 2.5
    # flf2v workflow uses, so it is verified usage rather than an assumption.
    n(16, "LTXVAddGuide", positive=["6", 0], negative=["6", 1], vae=["7", 0],
      latent=["15", 0], image=["11", 0], frame_idx=0, strength=a.guide_strength)
    n(17, "LTXVAddGuide", positive=["16", 0], negative=["16", 1], vae=["7", 0],
      latent=["16", 2], image=["14", 0], frame_idx=-1, strength=a.guide_strength)

    if a.mode == "gen":
        n(18, "LTXVEmptyLatentAudio", frames_number=a.frames, frame_rate=float(a.fps),
          batch_size=1, audio_vae=["8", 0])
        audio_latent = ["18", 0]
    else:
        n(18, "LoadAudio", audio=a.audio)
        n(19, "TrimAudioDuration", audio=["18", 0], start_index=0.0,
          duration=round(a.frames / a.fps, 3))
        n(20, "LTXVAudioVAEEncode", audio=["19", 0], audio_vae=["8", 0])
        n(21, "SolidMask", value=0.0, width=1024, height=1024)
        n(22, "SetLatentNoiseMask", samples=["20", 0], mask=["21", 0])
        audio_latent = ["22", 0]

    n(30, "LTXVConcatAVLatent", video_latent=["17", 2], audio_latent=audio_latent)
    n(31, "RandomNoise", noise_seed=a.seed)

    if a.preset == "distilled":
        # The regime the shipped preset uses, and the only one confirmed to
        # produce clean frames. Distilled wants ~8 steps at CFG 1 with explicit
        # sigmas; feeding it dev-style 24 steps at CFG 4 yields soft, washed-out
        # output that looks like a broken model but is just the wrong preset.
        n(32, "LTXVDualCFGGuider", model=model, positive=["17", 0], negative=["17", 1],
          video_cfg=a.video_cfg, audio_cfg=a.audio_cfg)
        n(33, "SamplerEulerAncestral", eta=a.eta, s_noise=a.s_noise)
        n(34, "ManualSigmas", sigmas=a.sigmas)
    else:
        n(32, "CFGGuider", model=model, positive=["17", 0], negative=["17", 1], cfg=a.cfg)
        n(33, "KSamplerSelect", sampler_name="euler")
        n(34, "LTXVScheduler", latent=["30", 0], steps=a.steps, max_shift=2.05,
          base_shift=0.95, stretch=True, terminal=0.1)

    n(35, "SamplerCustomAdvanced", noise=["31", 0], guider=["32", 0], sampler=["33", 0],
      sigmas=["34", 0], latent_image=["30", 0])

    n(36, "LTXVSeparateAVLatent", av_latent=["35", 0])
    # Crop the guide frames out of the video latent before decoding.
    n(37, "LTXVCropGuides", positive=["17", 0], negative=["17", 1], latent=["36", 0])
    n(38, "VAEDecodeTiled", samples=["37", 2], vae=["7", 0], tile_size=512,
      overlap=64, temporal_size=64, temporal_overlap=8)
    n(39, "LTXVAudioVAEDecode", samples=["36", 1], audio_vae=["8", 0])

    n(40, "CreateVideo", images=["38", 0], audio=["39", 0], fps=float(a.fps))
    n(41, "SaveVideo", video=["40", 0], filename_prefix=a.out_prefix, format="auto", codec="auto")
    # Audio saved separately so it can be round-tripped through the AKUSPACE
    # a2a pass and recombined β€” that pair IS the before/after example.
    n(42, "SaveAudio", audio=["39", 0], filename_prefix=a.out_prefix + "_audio")
    return g


def validate(graph: dict, object_info: Path) -> list[str]:
    """Check every class and input name against a live object_info dump."""
    oi = json.loads(object_info.read_text())
    errs = []
    for nid, node in graph.items():
        cls = node["class_type"]
        if cls not in oi:
            errs.append(f"node {nid}: class '{cls}' not in object_info")
            continue
        spec = oi[cls]["input"]
        allowed = set(spec.get("required") or {}) | set(spec.get("optional") or {})
        for name in node["inputs"]:
            if name not in allowed:
                errs.append(f"node {nid} ({cls}): unknown input '{name}' β€” has {sorted(allowed)}")
        for name in (spec.get("required") or {}):
            if name not in node["inputs"]:
                errs.append(f"node {nid} ({cls}): MISSING required input '{name}'")
    # every reference must point at a node that exists
    for nid, node in graph.items():
        for name, val in node["inputs"].items():
            if isinstance(val, list) and len(val) == 2 and isinstance(val[0], str):
                if val[0] not in graph:
                    errs.append(f"node {nid}.{name} -> dangling reference to '{val[0]}'")
    return errs


def main() -> int:
    p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    p.add_argument("--mode", choices=["gen", "drive"], default="gen")
    p.add_argument("--out", required=True)
    p.add_argument("--prompt", default="PLACEHOLDER scene description")
    p.add_argument("--negative", default=NEGATIVE)
    p.add_argument("--first-image", default="frame_first.png")
    p.add_argument("--last-image", default="frame_last.png")
    p.add_argument("--audio", default="track.wav", help="drive mode only")
    p.add_argument("--frames", type=int, default=241, help="241 @ 24fps = 10.04 s")
    p.add_argument("--fps", type=int, default=24)
    p.add_argument("--width", type=int, default=W)
    p.add_argument("--height", type=int, default=H)
    p.add_argument("--steps", type=int, default=24)
    p.add_argument("--cfg", type=float, default=4.0)
    p.add_argument("--seed", type=int, default=42)
    p.add_argument("--guide-strength", type=float, default=0.7,
                   help="0.7 is the stock flf2v value")
    p.add_argument("--lora", default=None, help="e.g. akuspace-v5/lora_weights_step_11500.safetensors")
    p.add_argument("--lora-strength", type=float, default=1.0)
    p.add_argument("--out-prefix", default="v5video/flf25")
    p.add_argument("--preset", choices=["distilled", "dev"], default="distilled",
                   help="distilled = the shipped, verified-good regime (default)")
    p.add_argument("--video-cfg", type=float, default=1.0)
    p.add_argument("--audio-cfg", type=float, default=1.0)
    p.add_argument("--eta", type=float, default=0.0)
    p.add_argument("--s-noise", type=float, default=1.0)
    p.add_argument("--sigmas", default=SIGMAS_DISTILLED)
    p.add_argument("--object-info", default=None, help="live object_info.json to validate against")
    a = p.parse_args()

    graph = build(a)

    if a.object_info:
        errs = validate(graph, Path(a.object_info))
        if errs:
            print(f"VALIDATION FAILED ({len(errs)}):", file=sys.stderr)
            for e in errs:
                print("  " + e, file=sys.stderr)
            return 1
        print(f"validated OK against {a.object_info}")
    else:
        print("WARNING: not validated β€” pass --object-info to check against a live install")

    Path(a.out).write_text(json.dumps(graph, indent=2))
    print(f"{a.out}: {len(graph)} nodes, mode={a.mode}, "
          f"{a.frames}f @ {a.fps}fps = {a.frames/a.fps:.2f}s, lora={a.lora or 'none'}")
    return 0


if __name__ == "__main__":
    sys.exit(main())