Text-to-Video
VideoX Fun
MiniMax-H3-Acc-LoRAs / predict_t2v.py
hkunzhe's picture
Upload MiniMax-H3 acceleration LoRAs and examples
d41cfc8 verified
Raw
History Blame Contribute Delete
2.67 kB
import sys
from pathlib import Path
import torch
from diffusers import ComponentsManager, ModularPipeline
from diffusers.utils.export_utils import encode_video
sys.path.insert(0, str(Path(__file__).resolve().parent))
from minimax_h3_pdd import apply_pdd_lora
model_path = "MiniMaxAI/MiniMax-H3"
pdd_lora_path = str(Path(__file__).resolve().parent / "MiniMax-H3-FL2VA-Acc-8Step.safetensors")
prompt = (
"[Shot 1] Cinematic wide shot, low angle, camera pushing in fast. A knight in dented steel plate armour "
"charges across a muddy tournament field toward a shaggy brown warhorse-mounted rival, his tattered blue "
"surcoat streaming behind him. He raises a notched longsword in both hands and brings it down; the rival "
"parries with a round wooden shield that splinters, chips of wood spinning into the air. Clods of wet earth "
"fly up from their boots and hooves. Grey banners snap in a strong crosswind under an overcast sky, and a "
"crowd of blurred spectators presses against a wooden barricade in the background. The camera swings around "
"the clash as the knight shoulders into the shield and both men stagger sideways."
"Ringing steel-on-steel impacts with long metallic decay, the dry crack of splitting wood, heavy plate "
"rattling with every stride, boots sucking out of deep mud, a horse snorting and stamping, wind buffeting the "
"microphone and a roaring crowd swelling behind it."
"Full orchestral battle cue, fast ostinato strings under blaring low brass, timpani accenting each sword "
"impact."
)
sample_size = [704, 1280]
video_length = 124
manager = ComponentsManager()
manager.enable_auto_cpu_offload(device="cuda", memory_reserve_margin="12GB")
pipeline = ModularPipeline.from_pretrained(model_path, workflow="t2va", components_manager=manager)
pipeline.load_components(dtype=torch.bfloat16, pretrained_model_name_or_path=model_path)
nfe = apply_pdd_lora(
pipeline.transformer, pdd_lora_path, pipeline.scheduler.shift, pipeline.audio_scheduler.shift
)
generator = torch.Generator().manual_seed(42)
result = pipeline(
prompt=prompt,
height=sample_size[0],
width=sample_size[1],
num_frames=video_length,
num_inference_steps=nfe + 1,
generator=generator,
output_type="np",
output=["videos", "audio", "sampling_rate"],
)
output_path = Path(__file__).resolve().parent / "samples" / "t2v.mp4"
output_path.parent.mkdir(parents=True, exist_ok=True)
encode_video(
result["videos"][0],
fps=24,
output_path=str(output_path),
audio=result["audio"][0],
audio_sample_rate=int(result["sampling_rate"]),
)
print(f"Saved {output_path}", flush=True)