File size: 10,431 Bytes
7dbeac1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
"""MiniMax H3 video prompt structure for script → FL2V.

Guides (cached under optimization/h3-shrink/docs/):
- VIDEO_PROMPT_base-en.txt — T2VA / I2VA / FL2VA / L2VA
- VIDEO_PROMPT_ref-en.txt — full-reference Ref2VA

Script / scene generators should emit fields that map cleanly into these
structures so storyboard panels + H3 student FL2V get usable prompts.
"""

from __future__ import annotations

import re
from typing import Optional

# Local H3 clip length at 24 fps / 124 frames ≈ 5.17 s (length ≡ 5 mod 17).
H3_DEFAULT_CLIP_SECONDS = 5.0

FL2VA_ALIGN_LINE = (
    "How the reference pictures align with the target video — "
    "Picture 1 (from Shot 1) aligns with the 0.00-second mark of the target video; "
    "Picture 2 (from Shot {shot_n}) aligns with the {end_s:.2f}-second mark of the target video."
)

I2VA_ALIGN_LINE = (
    "For the target video, at 0.00 seconds into the target video, "
    "<Picture 1> (from [Shot 1]) is fully referenced."
)


def format_fl2va_prompt(
    *,
    multimodal_body: str,
    overall_soundscape: str = "N/A",
    non_diegetic_music: str = "N/A",
    duration_s: float = H3_DEFAULT_CLIP_SECONDS,
    final_shot: int = 1,
) -> str:
    """Assemble a complete FL2VA prompt (first+last storyboard panels)."""
    body = multimodal_body.strip()
    if not body.startswith("[Shot"):
        body = f"[Shot 1] {body}"
    align = FL2VA_ALIGN_LINE.format(shot_n=final_shot, end_s=duration_s)
    return (
        f"{align}\n\n"
        f"integrated_multimodal_description: {body}\n\n"
        f"overall_soundscape: {overall_soundscape.strip() or 'N/A'}\n\n"
        f"non_diegetic_music: {non_diegetic_music.strip() or 'N/A'}"
    )


def format_t2va_prompt(
    *,
    multimodal_body: str,
    overall_soundscape: str = "N/A",
    non_diegetic_music: str = "N/A",
) -> str:
    body = multimodal_body.strip()
    if not body.startswith("[Shot"):
        body = f"[Shot 1] {body}"
    return (
        f"integrated_multimodal_description: {body}\n\n"
        f"overall_soundscape: {overall_soundscape.strip() or 'N/A'}\n\n"
        f"non_diegetic_music: {non_diegetic_music.strip() or 'N/A'}"
    )


def parse_scene_h3_fields(scene_text: str) -> dict:
    """Extract H3-related fields from a genAI scene block."""
    out: dict = {
        "h3_mode": None,
        "h3_video_prompt": None,
        "storyboard_prompt": None,
        "action": None,
        "dialogue": [],
        "shot": None,
        "duration": None,
        "soundscape": None,
        "music": None,
    }
    text = scene_text.strip()

    m = re.search(r"^H3_MODE:\s*(.+)$", text, re.M | re.I)
    if m:
        out["h3_mode"] = m.group(1).strip().upper()

    m = re.search(r"^STORYBOARD_PROMPT:\s*(.+)$", text, re.M | re.I)
    if m:
        out["storyboard_prompt"] = m.group(1).strip()
    else:
        m = re.search(r"^PROMPT:\s*(.+)$", text, re.M | re.I)
        if m:
            out["storyboard_prompt"] = m.group(1).strip()

    m = re.search(r"^ACTION:\s*(.+)$", text, re.M | re.I)
    if m:
        out["action"] = m.group(1).strip()

    m = re.search(r"^SHOT(?:\s*1)?:\s*(.+)$", text, re.M | re.I)
    if m:
        out["shot"] = m.group(1).strip()

    m = re.search(r"^DURATION:\s*([\d.]+)", text, re.M | re.I)
    if m:
        out["duration"] = float(m.group(1))

    m = re.search(r"^OVERALL_SOUNDSCAPE:\s*(.+)$", text, re.M | re.I)
    if m:
        out["soundscape"] = m.group(1).strip()

    m = re.search(r"^NON_DIEGETIC_MUSIC:\s*(.+)$", text, re.M | re.I)
    if m:
        out["music"] = m.group(1).strip()

    # Multi-line H3_VIDEO_PROMPT until next machine field or next scene.
    m = re.search(
        r"^H3_VIDEO_PROMPT:\s*\n?(.*?)(?=^\s*(?:LORA|AUDIO|DURATION|STORYBOARD_PROMPT|H3_MODE|ACTION|SHOT|PROMPT|DIALOGUE)\s*:|^\s*##\s*SCENE|\Z)",
        text,
        re.M | re.S | re.I,
    )
    if m:
        out["h3_video_prompt"] = m.group(1).strip()

    for dm in re.finditer(
        r"^DIALOGUE\s*[—\-]\s*(.+?):\s*(.+)$", text, re.M | re.I
    ):
        out["dialogue"].append((dm.group(1).strip(), dm.group(2).strip()))

    return out


def scene_to_h3_video_prompt(scene_text: str, *, mode: str = "FL2VA") -> str:
    """Build an H3 video prompt from a scene block (uses H3_VIDEO_PROMPT if present)."""
    fields = parse_scene_h3_fields(scene_text)
    if fields.get("h3_video_prompt"):
        return fields["h3_video_prompt"]

    mode = (fields.get("h3_mode") or mode or "FL2VA").upper()
    duration = float(fields.get("duration") or H3_DEFAULT_CLIP_SECONDS)
    # Prefer 5 s for local H3; clamp to API-friendly range.
    duration = max(4.0, min(15.0, duration))

    action = fields.get("action") or "The scene plays out continuously."
    shot = fields.get("shot") or "medium shot, slow push in"
    style = "Live-action, cinematic"
    body_parts = [
        f"{style}, {shot}. {action}",
    ]
    for name, line in fields.get("dialogue") or []:
        # Preserve dialogue language; default English tag.
        body_parts.append(
            f'The character {name} (S1) says: <d>[English] {line}</d>'
        )
    # FL2VA path language
    if mode in ("FL2VA", "FL2V"):
        body_parts.append(
            "The framing begins on the composition of Picture 1 and continuously "
            "evolves until it settles into the composition of Picture 2 at the end of the shot."
        )
        body = " ".join(body_parts)
        return format_fl2va_prompt(
            multimodal_body=body,
            overall_soundscape=fields.get("soundscape") or "N/A",
            non_diegetic_music=fields.get("music") or "N/A",
            duration_s=duration,
            final_shot=1,
        )

    body = " ".join(body_parts)
    if mode in ("I2VA", "I2V"):
        return (
            f"{I2VA_ALIGN_LINE}\n\n"
            + format_t2va_prompt(
                multimodal_body=body,
                overall_soundscape=fields.get("soundscape") or "N/A",
                non_diegetic_music=fields.get("music") or "N/A",
            )
        )
    return format_t2va_prompt(
        multimodal_body=body,
        overall_soundscape=fields.get("soundscape") or "N/A",
        non_diegetic_music=fields.get("music") or "N/A",
    )


# ---------------------------------------------------------------------------
# System prompts for LLM / script-LoRA SFT targets
# ---------------------------------------------------------------------------

H3_SCENE_SYSTEM_PROMPT = """You are a scriptwriter for a MiniMax-H3 generative-AI video pipeline (Backlot).
You write ONE scene beat per request. Each scene maps 1:1 to:
  (1) a storyboard still (T2I / image-edit), and
  (2) one MiniMax H3 video clip via FL2VA (first storyboard panel → last / next panel).

Local H3 clips are ~5 seconds (124 frames @ 24 fps). Prefer DURATION 5 (allowed 4–8 for this local stack).

Output EXACTLY this structure (field labels verbatim, no markdown fences, no commentary):

## SCENE {N} — {SLUGLINE}
ACTION: <one or two sentences of visual action that can play in ~5 seconds>
DIALOGUE — {NAME}: <line>   (at most 1 short line; omit entirely if silent)
SHOT: <camera: type + optional amplitude + speed, natural English e.g. "medium shot, push in with small amplitude at slow speed">
STORYBOARD_PROMPT: <self-contained image prompt for the key frame: subject, environment, lighting, style; NO camera timeline>
H3_MODE: FL2VA
H3_VIDEO_PROMPT:
How the reference pictures align with the target video — Picture 1 (from Shot 1) aligns with the 0.00-second mark of the target video; Picture 2 (from Shot 1) aligns with the 5.00-second mark of the target video.

integrated_multimodal_description: [Shot 1] Live-action, cinematic, <opening composition matching the storyboard>. <continuous motion path from Picture 1 state to Picture 2 state; include camera motion as natural English>. <optional dialogue as: the young woman (S1) says: <d>[English] line here</d>>

overall_soundscape: <1–3 sentences of ambience / physical sounds, or N/A>

non_diegetic_music: <1–2 sentences of audience-only score instrumentation/tempo, or N/A>
LORA: <image-lora-name:strength, or "none">
AUDIO: <post-process sfx/music note for OpenCode, or "none">
DURATION: 5

MiniMax H3 prompt rules (critical):
- Prefer a SINGLE shot for FL2VA so the model interpolates continuously first→last.
- Camera motion uses type + optional amplitude ("with small amplitude" / "with large amplitude") + speed ("at slow speed" / "at fast speed").
- Motion types: Zoom In/Out, Push In/Pull Out, Pan Left/Right, Truck Left/Right, Tilt Up/Down, Pedestal Up/Down, Arc Shot, Tracking Shot, Static Shot, Shake Slightly/Strongly, POV, Roll Clockwise/Counterclockwise.
- Cuts (only if essential): later shots use "[Shot N] At MM:SS.mmm, the camera cuts to...". First shot has NO timestamp.
- Speakers: stable (S1), (S2). Dialogue ONLY inside <d>[Language] ...</d>; preserve original words.
- On-screen text in English double quotes, verbatim.
- overall_soundscape: no dialogue/singing; those stay in integrated_multimodal_description.
- non_diegetic_music: audience-only score; N/A if none.
- Write H3_VIDEO_PROMPT body in English; keep dialogue language original.
- Never contradict CANON. Do not repeat AVOID content.
- STORYBOARD_PROMPT is for still image models (Z-Image / Qwen Edit), not the video timeline.
"""


H3_SCRIPT_SYSTEM_PROMPT = """You are a screenwriter for a MiniMax-H3 generative video pipeline.
Write a short multi-beat episode outline as sequential scene blocks (not classic screenplay INT./EXT. pages).

Each scene block uses the same field structure as the single-scene generator (ACTION, SHOT, STORYBOARD_PROMPT, H3_MODE, H3_VIDEO_PROMPT, overall_soundscape, non_diegetic_music, DURATION).
Default H3_MODE is FL2VA: each scene's storyboard is the first frame; the next scene's storyboard is the last frame of the previous clip.
DURATION defaults to 5 (local H3 ~5s clips).

Rules:
- Output ONLY scene blocks, no markdown fences, no commentary.
- H3_VIDEO_PROMPT must follow MiniMax structure: FL2VA alignment line, then integrated_multimodal_description / overall_soundscape / non_diegetic_music.
- Prefer single-shot FL2VA paths; continuous motion from Picture 1 to Picture 2.
- Camera motion in natural English (type + amplitude + speed when meaningful).
- Dialogue inside <d>[Language] ...</d> with speaker (S1) IDs.
- Do not repeat AVOID content. Keep each beat specific and filmable.
"""