Text Generation
PEFT
Safetensors
English
lora
sft
trl
script-generation
minimax-h3
video-generation
conversational
Instructions to use woodfireind/H3-ScriptGen with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use woodfireind/H3-ScriptGen with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen3.5-0.8B") model = PeftModel.from_pretrained(base_model, "woodfireind/H3-ScriptGen") - Notebooks
- Google Colab
- Kaggle
H3-ScriptGen: MiniMax-H3 FL2VA scriptwriting LoRA (Qwen3.5-0.8B base, continue-trained from final adapter)
7dbeac1 verified | """MiniMax H3 video prompt structure for script → FL2V. | |
| Guides (cached under optimization/h3-shrink/docs/): | |
| - VIDEO_PROMPT_base-en.txt — T2VA / I2VA / FL2VA / L2VA | |
| - VIDEO_PROMPT_ref-en.txt — full-reference Ref2VA | |
| Script / scene generators should emit fields that map cleanly into these | |
| structures so storyboard panels + H3 student FL2V get usable prompts. | |
| """ | |
| from __future__ import annotations | |
| import re | |
| from typing import Optional | |
| # Local H3 clip length at 24 fps / 124 frames ≈ 5.17 s (length ≡ 5 mod 17). | |
| H3_DEFAULT_CLIP_SECONDS = 5.0 | |
| FL2VA_ALIGN_LINE = ( | |
| "How the reference pictures align with the target video — " | |
| "Picture 1 (from Shot 1) aligns with the 0.00-second mark of the target video; " | |
| "Picture 2 (from Shot {shot_n}) aligns with the {end_s:.2f}-second mark of the target video." | |
| ) | |
| I2VA_ALIGN_LINE = ( | |
| "For the target video, at 0.00 seconds into the target video, " | |
| "<Picture 1> (from [Shot 1]) is fully referenced." | |
| ) | |
| def format_fl2va_prompt( | |
| *, | |
| multimodal_body: str, | |
| overall_soundscape: str = "N/A", | |
| non_diegetic_music: str = "N/A", | |
| duration_s: float = H3_DEFAULT_CLIP_SECONDS, | |
| final_shot: int = 1, | |
| ) -> str: | |
| """Assemble a complete FL2VA prompt (first+last storyboard panels).""" | |
| body = multimodal_body.strip() | |
| if not body.startswith("[Shot"): | |
| body = f"[Shot 1] {body}" | |
| align = FL2VA_ALIGN_LINE.format(shot_n=final_shot, end_s=duration_s) | |
| return ( | |
| f"{align}\n\n" | |
| f"integrated_multimodal_description: {body}\n\n" | |
| f"overall_soundscape: {overall_soundscape.strip() or 'N/A'}\n\n" | |
| f"non_diegetic_music: {non_diegetic_music.strip() or 'N/A'}" | |
| ) | |
| def format_t2va_prompt( | |
| *, | |
| multimodal_body: str, | |
| overall_soundscape: str = "N/A", | |
| non_diegetic_music: str = "N/A", | |
| ) -> str: | |
| body = multimodal_body.strip() | |
| if not body.startswith("[Shot"): | |
| body = f"[Shot 1] {body}" | |
| return ( | |
| f"integrated_multimodal_description: {body}\n\n" | |
| f"overall_soundscape: {overall_soundscape.strip() or 'N/A'}\n\n" | |
| f"non_diegetic_music: {non_diegetic_music.strip() or 'N/A'}" | |
| ) | |
| def parse_scene_h3_fields(scene_text: str) -> dict: | |
| """Extract H3-related fields from a genAI scene block.""" | |
| out: dict = { | |
| "h3_mode": None, | |
| "h3_video_prompt": None, | |
| "storyboard_prompt": None, | |
| "action": None, | |
| "dialogue": [], | |
| "shot": None, | |
| "duration": None, | |
| "soundscape": None, | |
| "music": None, | |
| } | |
| text = scene_text.strip() | |
| m = re.search(r"^H3_MODE:\s*(.+)$", text, re.M | re.I) | |
| if m: | |
| out["h3_mode"] = m.group(1).strip().upper() | |
| m = re.search(r"^STORYBOARD_PROMPT:\s*(.+)$", text, re.M | re.I) | |
| if m: | |
| out["storyboard_prompt"] = m.group(1).strip() | |
| else: | |
| m = re.search(r"^PROMPT:\s*(.+)$", text, re.M | re.I) | |
| if m: | |
| out["storyboard_prompt"] = m.group(1).strip() | |
| m = re.search(r"^ACTION:\s*(.+)$", text, re.M | re.I) | |
| if m: | |
| out["action"] = m.group(1).strip() | |
| m = re.search(r"^SHOT(?:\s*1)?:\s*(.+)$", text, re.M | re.I) | |
| if m: | |
| out["shot"] = m.group(1).strip() | |
| m = re.search(r"^DURATION:\s*([\d.]+)", text, re.M | re.I) | |
| if m: | |
| out["duration"] = float(m.group(1)) | |
| m = re.search(r"^OVERALL_SOUNDSCAPE:\s*(.+)$", text, re.M | re.I) | |
| if m: | |
| out["soundscape"] = m.group(1).strip() | |
| m = re.search(r"^NON_DIEGETIC_MUSIC:\s*(.+)$", text, re.M | re.I) | |
| if m: | |
| out["music"] = m.group(1).strip() | |
| # Multi-line H3_VIDEO_PROMPT until next machine field or next scene. | |
| m = re.search( | |
| r"^H3_VIDEO_PROMPT:\s*\n?(.*?)(?=^\s*(?:LORA|AUDIO|DURATION|STORYBOARD_PROMPT|H3_MODE|ACTION|SHOT|PROMPT|DIALOGUE)\s*:|^\s*##\s*SCENE|\Z)", | |
| text, | |
| re.M | re.S | re.I, | |
| ) | |
| if m: | |
| out["h3_video_prompt"] = m.group(1).strip() | |
| for dm in re.finditer( | |
| r"^DIALOGUE\s*[—\-]\s*(.+?):\s*(.+)$", text, re.M | re.I | |
| ): | |
| out["dialogue"].append((dm.group(1).strip(), dm.group(2).strip())) | |
| return out | |
| def scene_to_h3_video_prompt(scene_text: str, *, mode: str = "FL2VA") -> str: | |
| """Build an H3 video prompt from a scene block (uses H3_VIDEO_PROMPT if present).""" | |
| fields = parse_scene_h3_fields(scene_text) | |
| if fields.get("h3_video_prompt"): | |
| return fields["h3_video_prompt"] | |
| mode = (fields.get("h3_mode") or mode or "FL2VA").upper() | |
| duration = float(fields.get("duration") or H3_DEFAULT_CLIP_SECONDS) | |
| # Prefer 5 s for local H3; clamp to API-friendly range. | |
| duration = max(4.0, min(15.0, duration)) | |
| action = fields.get("action") or "The scene plays out continuously." | |
| shot = fields.get("shot") or "medium shot, slow push in" | |
| style = "Live-action, cinematic" | |
| body_parts = [ | |
| f"{style}, {shot}. {action}", | |
| ] | |
| for name, line in fields.get("dialogue") or []: | |
| # Preserve dialogue language; default English tag. | |
| body_parts.append( | |
| f'The character {name} (S1) says: <d>[English] {line}</d>' | |
| ) | |
| # FL2VA path language | |
| if mode in ("FL2VA", "FL2V"): | |
| body_parts.append( | |
| "The framing begins on the composition of Picture 1 and continuously " | |
| "evolves until it settles into the composition of Picture 2 at the end of the shot." | |
| ) | |
| body = " ".join(body_parts) | |
| return format_fl2va_prompt( | |
| multimodal_body=body, | |
| overall_soundscape=fields.get("soundscape") or "N/A", | |
| non_diegetic_music=fields.get("music") or "N/A", | |
| duration_s=duration, | |
| final_shot=1, | |
| ) | |
| body = " ".join(body_parts) | |
| if mode in ("I2VA", "I2V"): | |
| return ( | |
| f"{I2VA_ALIGN_LINE}\n\n" | |
| + format_t2va_prompt( | |
| multimodal_body=body, | |
| overall_soundscape=fields.get("soundscape") or "N/A", | |
| non_diegetic_music=fields.get("music") or "N/A", | |
| ) | |
| ) | |
| return format_t2va_prompt( | |
| multimodal_body=body, | |
| overall_soundscape=fields.get("soundscape") or "N/A", | |
| non_diegetic_music=fields.get("music") or "N/A", | |
| ) | |
| # --------------------------------------------------------------------------- | |
| # System prompts for LLM / script-LoRA SFT targets | |
| # --------------------------------------------------------------------------- | |
| H3_SCENE_SYSTEM_PROMPT = """You are a scriptwriter for a MiniMax-H3 generative-AI video pipeline (Backlot). | |
| You write ONE scene beat per request. Each scene maps 1:1 to: | |
| (1) a storyboard still (T2I / image-edit), and | |
| (2) one MiniMax H3 video clip via FL2VA (first storyboard panel → last / next panel). | |
| Local H3 clips are ~5 seconds (124 frames @ 24 fps). Prefer DURATION 5 (allowed 4–8 for this local stack). | |
| Output EXACTLY this structure (field labels verbatim, no markdown fences, no commentary): | |
| ## SCENE {N} — {SLUGLINE} | |
| ACTION: <one or two sentences of visual action that can play in ~5 seconds> | |
| DIALOGUE — {NAME}: <line> (at most 1 short line; omit entirely if silent) | |
| SHOT: <camera: type + optional amplitude + speed, natural English e.g. "medium shot, push in with small amplitude at slow speed"> | |
| STORYBOARD_PROMPT: <self-contained image prompt for the key frame: subject, environment, lighting, style; NO camera timeline> | |
| H3_MODE: FL2VA | |
| H3_VIDEO_PROMPT: | |
| How the reference pictures align with the target video — Picture 1 (from Shot 1) aligns with the 0.00-second mark of the target video; Picture 2 (from Shot 1) aligns with the 5.00-second mark of the target video. | |
| integrated_multimodal_description: [Shot 1] Live-action, cinematic, <opening composition matching the storyboard>. <continuous motion path from Picture 1 state to Picture 2 state; include camera motion as natural English>. <optional dialogue as: the young woman (S1) says: <d>[English] line here</d>> | |
| overall_soundscape: <1–3 sentences of ambience / physical sounds, or N/A> | |
| non_diegetic_music: <1–2 sentences of audience-only score instrumentation/tempo, or N/A> | |
| LORA: <image-lora-name:strength, or "none"> | |
| AUDIO: <post-process sfx/music note for OpenCode, or "none"> | |
| DURATION: 5 | |
| MiniMax H3 prompt rules (critical): | |
| - Prefer a SINGLE shot for FL2VA so the model interpolates continuously first→last. | |
| - Camera motion uses type + optional amplitude ("with small amplitude" / "with large amplitude") + speed ("at slow speed" / "at fast speed"). | |
| - Motion types: Zoom In/Out, Push In/Pull Out, Pan Left/Right, Truck Left/Right, Tilt Up/Down, Pedestal Up/Down, Arc Shot, Tracking Shot, Static Shot, Shake Slightly/Strongly, POV, Roll Clockwise/Counterclockwise. | |
| - Cuts (only if essential): later shots use "[Shot N] At MM:SS.mmm, the camera cuts to...". First shot has NO timestamp. | |
| - Speakers: stable (S1), (S2). Dialogue ONLY inside <d>[Language] ...</d>; preserve original words. | |
| - On-screen text in English double quotes, verbatim. | |
| - overall_soundscape: no dialogue/singing; those stay in integrated_multimodal_description. | |
| - non_diegetic_music: audience-only score; N/A if none. | |
| - Write H3_VIDEO_PROMPT body in English; keep dialogue language original. | |
| - Never contradict CANON. Do not repeat AVOID content. | |
| - STORYBOARD_PROMPT is for still image models (Z-Image / Qwen Edit), not the video timeline. | |
| """ | |
| H3_SCRIPT_SYSTEM_PROMPT = """You are a screenwriter for a MiniMax-H3 generative video pipeline. | |
| Write a short multi-beat episode outline as sequential scene blocks (not classic screenplay INT./EXT. pages). | |
| Each scene block uses the same field structure as the single-scene generator (ACTION, SHOT, STORYBOARD_PROMPT, H3_MODE, H3_VIDEO_PROMPT, overall_soundscape, non_diegetic_music, DURATION). | |
| Default H3_MODE is FL2VA: each scene's storyboard is the first frame; the next scene's storyboard is the last frame of the previous clip. | |
| DURATION defaults to 5 (local H3 ~5s clips). | |
| Rules: | |
| - Output ONLY scene blocks, no markdown fences, no commentary. | |
| - H3_VIDEO_PROMPT must follow MiniMax structure: FL2VA alignment line, then integrated_multimodal_description / overall_soundscape / non_diegetic_music. | |
| - Prefer single-shot FL2VA paths; continuous motion from Picture 1 to Picture 2. | |
| - Camera motion in natural English (type + amplitude + speed when meaningful). | |
| - Dialogue inside <d>[Language] ...</d> with speaker (S1) IDs. | |
| - Do not repeat AVOID content. Keep each beat specific and filmable. | |
| """ | |