SUPIR

Runtime error

App Files Files Community

Fabrice-TIERCELIN commited on Jul 5, 2025

Commit

76d0f82

verified ·

1 Parent(s): 1bd869c

Upload 2 files

Browse files

Files changed (2) hide show

README.md +1 -1
app.py +521 -102

README.md CHANGED Viewed

@@ -5,7 +5,7 @@ colorFrom: pink
 colorTo: gray
 sdk: gradio
 sdk_version: 5.29.1
-app_file: app_start_end.py
 license: apache-2.0
 short_description: Text-to-Video/Image-to-Video/Video extender (timed prompt)
 tags:

 colorTo: gray
 sdk: gradio
 sdk_version: 5.29.1
+app_file: app.py
 license: apache-2.0
 short_description: Text-to-Video/Image-to-Video/Video extender (timed prompt)
 tags:

app.py CHANGED Viewed

@@ -13,9 +13,10 @@ import torch
 import traceback
 import einops
 import safetensors.torch as sf
-import numpy as np
 import random
 import time
 import math
 # 20250506 pftq: Added for video input loading
 import decord
@@ -38,74 +39,77 @@ from diffusers_helper.hunyuan import encode_prompt_conds, vae_decode, vae_encode
 from diffusers_helper.utils import save_bcthw_as_mp4, crop_or_pad_yield_mask, soft_append_bcthw, resize_and_center_crop, state_dict_weighted_merge, state_dict_offset_merge, generate_timestamp
 from diffusers_helper.models.hunyuan_video_packed import HunyuanVideoTransformer3DModelPacked
 from diffusers_helper.pipelines.k_diffusion_hunyuan import sample_hunyuan
-if torch.cuda.device_count() > 0:
-    from diffusers_helper.memory import cpu, gpu, get_cuda_free_memory_gb, move_model_to_device_with_memory_preservation, offload_model_from_device_for_memory_preservation, fake_diffusers_current_device, DynamicSwapInstaller, unload_complete_models, load_model_as_complete
 from diffusers_helper.thread_utils import AsyncStream, async_run
 from diffusers_helper.gradio.progress_bar import make_progress_bar_css, make_progress_bar_html
 from transformers import SiglipImageProcessor, SiglipVisionModel
 from diffusers_helper.clip_vision import hf_clip_vision_encode
 from diffusers_helper.bucket_tools import find_nearest_bucket
-from diffusers import BitsAndBytesConfig as DiffusersBitsAndBytesConfig, HunyuanVideoTransformer3DModel, HunyuanVideoPipeline
-import pillow_heif
-pillow_heif.register_heif_opener()
-high_vram = False
-free_mem_gb = 0
-if torch.cuda.device_count() > 0:
-    free_mem_gb = get_cuda_free_memory_gb(gpu)
-    high_vram = free_mem_gb > 60
-    #print(f'Free VRAM {free_mem_gb} GB')
-    #print(f'High-VRAM Mode: {high_vram}')
-    text_encoder = LlamaModel.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='text_encoder', torch_dtype=torch.float16).cpu()
-    text_encoder_2 = CLIPTextModel.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='text_encoder_2', torch_dtype=torch.float16).cpu()
-    tokenizer = LlamaTokenizerFast.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='tokenizer')
-    tokenizer_2 = CLIPTokenizer.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='tokenizer_2')
-    vae = AutoencoderKLHunyuanVideo.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='vae', torch_dtype=torch.float16).cpu()
-    feature_extractor = SiglipImageProcessor.from_pretrained("lllyasviel/flux_redux_bfl", subfolder='feature_extractor')
-    image_encoder = SiglipVisionModel.from_pretrained("lllyasviel/flux_redux_bfl", subfolder='image_encoder', torch_dtype=torch.float16).cpu()
-    transformer = HunyuanVideoTransformer3DModelPacked.from_pretrained('lllyasviel/FramePack_F1_I2V_HY_20250503', torch_dtype=torch.bfloat16).cpu()
-    vae.eval()
-    text_encoder.eval()
-    text_encoder_2.eval()
-    image_encoder.eval()
-    transformer.eval()
-    if not high_vram:
-        vae.enable_slicing()
-        vae.enable_tiling()
-    transformer.high_quality_fp32_output_for_inference = True
-    #print('transformer.high_quality_fp32_output_for_inference = True')
-    transformer.to(dtype=torch.bfloat16)
-    vae.to(dtype=torch.float16)
-    image_encoder.to(dtype=torch.float16)
-    text_encoder.to(dtype=torch.float16)
-    text_encoder_2.to(dtype=torch.float16)
-    vae.requires_grad_(False)
-    text_encoder.requires_grad_(False)
-    text_encoder_2.requires_grad_(False)
-    image_encoder.requires_grad_(False)
-    transformer.requires_grad_(False)
-    if not high_vram:
-        # DynamicSwapInstaller is same as huggingface's enable_sequential_offload but 3x faster
-        DynamicSwapInstaller.install_model(transformer, device=gpu)
-        DynamicSwapInstaller.install_model(text_encoder, device=gpu)
-    else:
-        text_encoder.to(gpu)
-        text_encoder_2.to(gpu)
-        image_encoder.to(gpu)
-        vae.to(gpu)
-        transformer.to(gpu)
 stream = AsyncStream()
@@ -114,6 +118,7 @@ os.makedirs(outputs_folder, exist_ok=True)
 input_image_debug_value = [None]
 input_video_debug_value = [None]
 prompt_debug_value = [None]
 total_second_length_debug_value = [None]
@@ -308,7 +313,7 @@ def set_mp4_comments_imageio_ffmpeg(input_file, comments):
         return False
 @torch.no_grad()
-def worker(input_image, image_position, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number):
     def encode_prompt(prompt, n_prompt):
         llama_vec, clip_l_pooler = encode_prompt_conds(prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
@@ -577,6 +582,269 @@ def worker(input_image, image_position, prompts, n_prompt, seed, resolution, tot
     stream.output_queue.push(('end', None))
     return
 # 20250506 pftq: Modified worker to accept video input and clean frame count
 @torch.no_grad()
 def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch):
@@ -857,18 +1125,18 @@ def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_
     stream.output_queue.push(('end', None))
     return
-def get_duration(input_image, image_position, prompts, generation_mode, n_prompt, seed, resolution, total_second_length, allocation_time, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number):
     return allocation_time
 # Remove this decorator if you run on local
 @spaces.GPU(duration=get_duration)
-def process_on_gpu(input_image, image_position, prompts, generation_mode, n_prompt, seed, resolution, total_second_length, allocation_time, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number
            ):
     start = time.time()
     global stream
     stream = AsyncStream()
-    async_run(worker, input_image, image_position, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number)
     output_filename = None
@@ -899,6 +1167,7 @@ def process_on_gpu(input_image, image_position, prompts, generation_mode, n_prom
 def process(input_image,
             image_position=0,
             prompt="",
             generation_mode="image",
             n_prompt="",
@@ -909,12 +1178,12 @@ def process(input_image,
             resolution=640,
             total_second_length=5,
             latent_window_size=9,
-            steps=25,
             cfg=1.0,
             gs=10.0,
             rs=0.0,
             gpu_memory_preservation=6,
-            enable_preview=True,
             use_teacache=False,
             mp4_crf=16,
             fps_number=30
@@ -922,12 +1191,13 @@ def process(input_image,
     if auto_allocation:
         allocation_time = min(total_second_length * 60 * (1.5 if use_teacache else 3.0) * (1 + ((steps - 25) / 25))**2, 600)
-    if input_image_debug_value[0] is not None or prompt_debug_value[0] is not None or total_second_length_debug_value[0] is not None:
         input_image = input_image_debug_value[0]
         prompt = prompt_debug_value[0]
         total_second_length = total_second_length_debug_value[0]
         allocation_time = min(total_second_length_debug_value[0] * 60 * 100, 600)
-        input_image_debug_value[0] = prompt_debug_value[0] = total_second_length_debug_value[0] = None
     if torch.cuda.device_count() == 0:
         gr.Warning('Set this space to GPU config to make it work.')
@@ -949,6 +1219,7 @@ def process(input_image,
     yield from process_on_gpu(input_image,
             image_position,
             prompts,
             generation_mode,
             n_prompt,
@@ -1019,7 +1290,7 @@ def process_video(input_video, prompt, n_prompt, randomize_seed, seed, auto_allo
         prompt = prompt_debug_value[0]
         total_second_length = total_second_length_debug_value[0]
         allocation_time = min(total_second_length_debug_value[0] * 60 * 100, 600)
-        input_video_debug_value[0] = prompt_debug_value[0] = total_second_length_debug_value[0] = None
     if torch.cuda.device_count() == 0:
         gr.Warning('Set this space to GPU config to make it work.')
@@ -1119,9 +1390,10 @@ with block:
     local_storage = gr.BrowserState(default_local_storage)
     with gr.Row():
         with gr.Column():
-            generation_mode = gr.Radio([["Text-to-Video", "text"], ["Image-to-Video", "image"], ["Video Extension", "video"]], elem_id="generation-mode", label="Generation mode", value = "image")
             text_to_video_hint = gr.HTML("Text-to-Video badly works with a flash effect at the start. I discourage to use the Text-to-Video feature. You should rather generate an image with Flux and use Image-to-Video. You will save time.")
             input_image = gr.Image(sources='upload', type="numpy", label="Image", height=320)
             image_position = gr.Slider(label="Image position", minimum=0, maximum=100, value=0, step=1, info='0=Video start; 100=Video end (lower quality)')
             input_video = gr.Video(sources='upload', label="Input Video", height=320)
             timeless_prompt = gr.Textbox(label="Timeless prompt", info='Used on the whole duration of the generation', value='', placeholder="The creature starts to move, fast motion, fixed camera, focus motion, consistent arm, consistent position, mute colors, insanely detailed")
@@ -1147,7 +1419,7 @@ with block:
                 enable_preview = gr.Checkbox(label='Enable preview', value=True, info='Display a preview around each second generated but it costs 2 sec. for each second generated.')
                 use_teacache = gr.Checkbox(label='Use TeaCache', value=False, info='Faster speed and no break in brightness, but often makes hands and fingers slightly worse.')
-                n_prompt = gr.Textbox(label="Negative Prompt", value="Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", info='Requires using normal CFG (undistilled) instead of Distilled (set Distilled=1 and CFG > 1).')
                 fps_number = gr.Slider(label="Frame per seconds", info="The model is trained for 30 fps so other fps may generate weird results", minimum=10, maximum=60, value=30, step=1)
@@ -1197,6 +1469,7 @@ with block:
             with gr.Accordion("Debug", open=False):
                 input_image_debug = gr.Image(type="numpy", label="Image Debug", height=320)
                 input_video_debug = gr.Video(sources='upload', label="Input Video Debug", height=320)
                 prompt_debug = gr.Textbox(label="Prompt Debug", value='')
                 total_second_length_debug = gr.Slider(label="Additional Video Length to Generate (seconds) Debug", minimum=1, maximum=120, value=1, step=0.1)
@@ -1208,8 +1481,7 @@ with block:
             progress_desc = gr.Markdown('', elem_classes='no-generating-animation')
             progress_bar = gr.HTML('', elem_classes='no-generating-animation')
-    # 20250506 pftq: Updated inputs to include num_clean_frames
-    ips = [input_image, image_position, final_prompt, generation_mode, n_prompt, randomize_seed, seed, auto_allocation, allocation_time, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number]
     ips_video = [input_video, final_prompt, n_prompt, randomize_seed, seed, auto_allocation, allocation_time, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch]
     with gr.Row(elem_id="text_examples", visible=False):
@@ -1219,9 +1491,10 @@ with block:
                     [
                         None, # input_image
                         0, # image_position
                         "Overcrowed street in Japan, photorealistic, realistic, intricate details, 8k, insanely detailed",
                         "text", # generation_mode
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1254,9 +1527,10 @@ with block:
                     [
                         "./img_examples/Example2.webp", # input_image
                         0, # image_position
                         "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks, the man stops talking and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens",
                         "image", # generation_mode
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1277,9 +1551,10 @@ with block:
                     [
                         "./img_examples/Example1.png", # input_image
                         0, # image_position
                         "A dolphin emerges from the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
                         "image", # generation_mode
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1300,9 +1575,10 @@ with block:
                     [
                         "./img_examples/Example4.webp", # input_image
                         1, # image_position
                         "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                         "image", # generation_mode
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1323,9 +1599,10 @@ with block:
                     [
                         "./img_examples/Example4.webp", # input_image
                         50, # image_position
                         "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                         "image", # generation_mode
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1346,9 +1623,46 @@ with block:
                     [
                         "./img_examples/Example4.webp", # input_image
                         100, # image_position
                         "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                         "image", # generation_mode
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1381,7 +1695,7 @@ with block:
                     [
                         "./img_examples/Example1.mp4", # input_video
                         "View of the sea as far as the eye can see, from the seaside, a piece of land is barely visible on the horizon at the middle, the sky is radiant, reflections of the sun in the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1405,7 +1719,7 @@ with block:
                     [
                         "./img_examples/Example1.mp4", # input_video
                         "View of the sea as far as the eye can see, from the seaside, a piece of land is barely visible on the horizon at the middle, the sky is radiant, reflections of the sun in the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
-                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
@@ -1440,9 +1754,10 @@ with block:
                 [
                     None, # input_image
                     0, # image_position
                     "Overcrowed street in Japan, photorealistic, realistic, intricate details, 8k, insanely detailed",
                     "text", # generation_mode
-                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
@@ -1474,9 +1789,10 @@ with block:
                 [
                     "./img_examples/Example1.png", # input_image
                     0, # image_position
                     "A dolphin emerges from the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
                     "image", # generation_mode
-                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
@@ -1497,9 +1813,10 @@ with block:
                 [
                     "./img_examples/Example2.webp", # input_image
                     0, # image_position
                     "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks, the man stops talking and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens",
                     "image", # generation_mode
-                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
@@ -1520,9 +1837,10 @@ with block:
                 [
                     "./img_examples/Example2.webp", # input_image
                     0, # image_position
                     "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks, the woman stops talking and the woman listens A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens",
                     "image", # generation_mode
-                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
@@ -1543,9 +1861,10 @@ with block:
                 [
                     "./img_examples/Example3.jpg", # input_image
                     0, # image_position
                     "A boy is walking to the right, full view, full-length view, cartoon",
                     "image", # generation_mode
-                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
@@ -1566,9 +1885,10 @@ with block:
                 [
                     "./img_examples/Example4.webp", # input_image
                     100, # image_position
                     "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                     "image", # generation_mode
-                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
@@ -1594,13 +1914,48 @@ with block:
         cache_examples = False,
     )
     gr.Examples(
         label = "🎥 Examples from video",
         examples = [
                 [
                     "./img_examples/Example1.mp4", # input_video
                     "View of the sea as far as the eye can see, from the seaside, a piece of land is barely visible on the horizon at the middle, the sky is radiant, reflections of the sun in the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
-                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
@@ -1651,42 +2006,106 @@ with block:
     def handle_generation_mode_change(generation_mode_data):
         if generation_mode_data == "text":
-            return [gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = True)]
         elif generation_mode_data == "image":
-            return [gr.update(visible = False), gr.update(visible = True), gr.update(visible = True), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = True)]
         elif generation_mode_data == "video":
-            return [gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = False)]
-    def handle_field_debug_change(input_image_debug_data, input_video_debug_data, prompt_debug_data, total_second_length_debug_data):
         print("handle_field_debug_change")
         input_image_debug_value[0] = input_image_debug_data
         input_video_debug_value[0] = input_video_debug_data
         prompt_debug_value[0] = prompt_debug_data
         total_second_length_debug_value[0] = total_second_length_debug_data
         return []
     input_image_debug.upload(
         fn=handle_field_debug_change,
-        inputs=[input_image_debug, input_video_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
     input_video_debug.upload(
         fn=handle_field_debug_change,
-        inputs=[input_image_debug, input_video_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
     prompt_debug.change(
         fn=handle_field_debug_change,
-        inputs=[input_image_debug, input_video_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
     total_second_length_debug.change(
         fn=handle_field_debug_change,
-        inputs=[input_image_debug, input_video_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
@@ -1710,7 +2129,7 @@ with block:
     generation_mode.change(
         fn=handle_generation_mode_change,
         inputs=[generation_mode],
-        outputs=[text_to_video_hint, image_position, input_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint, fps_number]
     )
     # Update display when the page loads
@@ -1718,7 +2137,7 @@ with block:
         fn=handle_generation_mode_change, inputs = [
         generation_mode
     ], outputs = [
-       text_to_video_hint, image_position, input_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint, fps_number
     ]
     )

 import traceback
 import einops
 import safetensors.torch as sf
 import random
 import time
+import numpy as np
+import argparse
 import math
 # 20250506 pftq: Added for video input loading
 import decord
 from diffusers_helper.utils import save_bcthw_as_mp4, crop_or_pad_yield_mask, soft_append_bcthw, resize_and_center_crop, state_dict_weighted_merge, state_dict_offset_merge, generate_timestamp
 from diffusers_helper.models.hunyuan_video_packed import HunyuanVideoTransformer3DModelPacked
 from diffusers_helper.pipelines.k_diffusion_hunyuan import sample_hunyuan
+from diffusers_helper.memory import cpu, gpu, get_cuda_free_memory_gb, move_model_to_device_with_memory_preservation, offload_model_from_device_for_memory_preservation, fake_diffusers_current_device, DynamicSwapInstaller, unload_complete_models, load_model_as_complete
 from diffusers_helper.thread_utils import AsyncStream, async_run
 from diffusers_helper.gradio.progress_bar import make_progress_bar_css, make_progress_bar_html
 from transformers import SiglipImageProcessor, SiglipVisionModel
 from diffusers_helper.clip_vision import hf_clip_vision_encode
 from diffusers_helper.bucket_tools import find_nearest_bucket
+parser = argparse.ArgumentParser()
+parser.add_argument('--share', action='store_true')
+parser.add_argument("--server", type=str, default='0.0.0.0')
+parser.add_argument("--port", type=int, required=False)
+parser.add_argument("--inbrowser", action='store_true')
+args = parser.parse_args()
+# for win desktop probably use --server 127.0.0.1 --inbrowser
+# For linux server probably use --server 127.0.0.1 or do not use any cmd flags
+print(args)
+free_mem_gb = get_cuda_free_memory_gb(gpu)
+high_vram = free_mem_gb > 60
+print(f'Free VRAM {free_mem_gb} GB')
+print(f'High-VRAM Mode: {high_vram}')
+text_encoder = LlamaModel.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='text_encoder', torch_dtype=torch.float16).cpu()
+text_encoder_2 = CLIPTextModel.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='text_encoder_2', torch_dtype=torch.float16).cpu()
+tokenizer = LlamaTokenizerFast.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='tokenizer')
+tokenizer_2 = CLIPTokenizer.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='tokenizer_2')
+vae = AutoencoderKLHunyuanVideo.from_pretrained("hunyuanvideo-community/HunyuanVideo", subfolder='vae', torch_dtype=torch.float16).cpu()
+feature_extractor = SiglipImageProcessor.from_pretrained("lllyasviel/flux_redux_bfl", subfolder='feature_extractor')
+image_encoder = SiglipVisionModel.from_pretrained("lllyasviel/flux_redux_bfl", subfolder='image_encoder', torch_dtype=torch.float16).cpu()
+transformer = HunyuanVideoTransformer3DModelPacked.from_pretrained('lllyasviel/FramePackI2V_HY', torch_dtype=torch.bfloat16).cpu()
+vae.eval()
+text_encoder.eval()
+text_encoder_2.eval()
+image_encoder.eval()
+transformer.eval()
+if not high_vram:
+    vae.enable_slicing()
+    vae.enable_tiling()
+transformer.high_quality_fp32_output_for_inference = True
+print('transformer.high_quality_fp32_output_for_inference = True')
+transformer.to(dtype=torch.bfloat16)
+vae.to(dtype=torch.float16)
+image_encoder.to(dtype=torch.float16)
+text_encoder.to(dtype=torch.float16)
+text_encoder_2.to(dtype=torch.float16)
+vae.requires_grad_(False)
+text_encoder.requires_grad_(False)
+text_encoder_2.requires_grad_(False)
+image_encoder.requires_grad_(False)
+transformer.requires_grad_(False)
+if not high_vram:
+    # DynamicSwapInstaller is same as huggingface's enable_sequential_offload but 3x faster
+    DynamicSwapInstaller.install_model(transformer, device=gpu)
+    DynamicSwapInstaller.install_model(text_encoder, device=gpu)
+else:
+    text_encoder.to(gpu)
+    text_encoder_2.to(gpu)
+    image_encoder.to(gpu)
+    vae.to(gpu)
+    transformer.to(gpu)
 stream = AsyncStream()
 input_image_debug_value = [None]
 input_video_debug_value = [None]
+end_image_debug_value = [None]
 prompt_debug_value = [None]
 total_second_length_debug_value = [None]
         return False
 @torch.no_grad()
+def worker(input_image, image_position, end_image, prompts, n_prompt, seed, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, use_teacache, mp4_crf, fps_number):
     def encode_prompt(prompt, n_prompt):
         llama_vec, clip_l_pooler = encode_prompt_conds(prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
     stream.output_queue.push(('end', None))
     return
+@torch.no_grad()
+def worker_start_end(input_image, image_position, end_image, prompts, n_prompt, seed, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, use_teacache, mp4_crf, fps_number):
+    def encode_prompt(prompt, n_prompt):
+        llama_vec, clip_l_pooler = encode_prompt_conds(prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
+        if cfg == 1:
+            llama_vec_n, clip_l_pooler_n = torch.zeros_like(llama_vec), torch.zeros_like(clip_l_pooler)
+        else:
+            llama_vec_n, clip_l_pooler_n = encode_prompt_conds(n_prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
+        llama_vec, llama_attention_mask = crop_or_pad_yield_mask(llama_vec, length=512)
+        llama_vec_n, llama_attention_mask_n = crop_or_pad_yield_mask(llama_vec_n, length=512)
+        llama_vec = llama_vec.to(transformer.dtype)
+        llama_vec_n = llama_vec_n.to(transformer.dtype)
+        clip_l_pooler = clip_l_pooler.to(transformer.dtype)
+        clip_l_pooler_n = clip_l_pooler_n.to(transformer.dtype)
+        return [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n]
+    total_latent_sections = (total_second_length * fps_number) / (latent_window_size * 4)
+    total_latent_sections = int(max(round(total_latent_sections), 1))
+    job_id = generate_timestamp()
+    stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Starting ...'))))
+    try:
+        # Clean GPU
+        if not high_vram:
+            unload_complete_models(
+                text_encoder, text_encoder_2, image_encoder, vae, transformer
+            )
+        # Text encoding
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Text encoding ...'))))
+        if not high_vram:
+            fake_diffusers_current_device(text_encoder, gpu)  # since we only encode one text - that is one model move and one encode, offload is same time consumption since it is also one load and one encode.
+            load_model_as_complete(text_encoder_2, target_device=gpu)
+        prompt_parameters = []
+        for prompt_part in prompts[:total_latent_sections]:
+            prompt_parameters.append(encode_prompt(prompt_part, n_prompt))
+        # Clean GPU
+        if not high_vram:
+            unload_complete_models(
+                text_encoder, text_encoder_2
+            )
+        # Processing input image (start frame)
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Processing start frame ...'))))
+        H, W, C = input_image.shape
+        height, width = find_nearest_bucket(H, W, resolution=640)
+        input_image_np = resize_and_center_crop(input_image, target_width=width, target_height=height)
+        Image.fromarray(input_image_np).save(os.path.join(outputs_folder, f'{job_id}_start.png'))
+        input_image_pt = torch.from_numpy(input_image_np).float() / 127.5 - 1
+        input_image_pt = input_image_pt.permute(2, 0, 1)[None, :, None]
+        # Processing end image (if provided)
+        has_end_image = end_image is not None
+        if has_end_image:
+            stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Processing end frame ...'))))
+            H_end, W_end, C_end = end_image.shape
+            end_image_np = resize_and_center_crop(end_image, target_width=width, target_height=height)
+            Image.fromarray(end_image_np).save(os.path.join(outputs_folder, f'{job_id}_end.png'))
+            end_image_pt = torch.from_numpy(end_image_np).float() / 127.5 - 1
+            end_image_pt = end_image_pt.permute(2, 0, 1)[None, :, None]
+        # VAE encoding
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'VAE encoding ...'))))
+        if not high_vram:
+            load_model_as_complete(vae, target_device=gpu)
+        start_latent = vae_encode(input_image_pt, vae)
+        if has_end_image:
+            end_latent = vae_encode(end_image_pt, vae)
+        # CLIP Vision
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'CLIP Vision encoding ...'))))
+        if not high_vram:
+            load_model_as_complete(image_encoder, target_device=gpu)
+        image_encoder_output = hf_clip_vision_encode(input_image_np, feature_extractor, image_encoder)
+        image_encoder_last_hidden_state = image_encoder_output.last_hidden_state
+        if has_end_image:
+            end_image_encoder_output = hf_clip_vision_encode(end_image_np, feature_extractor, image_encoder)
+            end_image_encoder_last_hidden_state = end_image_encoder_output.last_hidden_state
+            # Combine both image embeddings or use a weighted approach
+            image_encoder_last_hidden_state = (image_encoder_last_hidden_state + end_image_encoder_last_hidden_state) / 2
+        # Clean GPU
+        if not high_vram:
+            unload_complete_models(
+                image_encoder
+            )
+        # Dtype
+        image_encoder_last_hidden_state = image_encoder_last_hidden_state.to(transformer.dtype)
+        # Sampling
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Start sampling ...'))))
+        rnd = torch.Generator("cpu").manual_seed(seed)
+        num_frames = latent_window_size * 4 - 3
+        history_latents = torch.zeros(size=(1, 16, 1 + 2 + 16, height // 8, width // 8), dtype=torch.float32, device=cpu)
+        history_pixels = None
+        total_generated_latent_frames = 0
+        # 将迭代器转换为列表
+        latent_paddings = list(reversed(range(total_latent_sections)))
+        if total_latent_sections > 4:
+            # In theory the latent_paddings should follow the above sequence, but it seems that duplicating some
+            # items looks better than expanding it when total_latent_sections > 4
+            # One can try to remove below trick and just
+            # use `latent_paddings = list(reversed(range(total_latent_sections)))` to compare
+            latent_paddings = [3] + [2] * (total_latent_sections - 3) + [1, 0]
+        for latent_padding in latent_paddings:
+            is_last_section = latent_padding == 0
+            is_first_section = latent_padding == latent_paddings[0]
+            latent_padding_size = latent_padding * latent_window_size
+            if stream.input_queue.top() == 'end':
+                stream.output_queue.push(('end', None))
+                return
+            print(f'latent_padding_size = {latent_padding_size}, is_last_section = {is_last_section}, is_first_section = {is_first_section}')
+            if len(prompt_parameters) > 0:
+                [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n] = prompt_parameters.pop(len(prompt_parameters) - 1)
+            indices = torch.arange(0, sum([1, latent_padding_size, latent_window_size, 1, 2, 16])).unsqueeze(0)
+            clean_latent_indices_pre, blank_indices, latent_indices, clean_latent_indices_post, clean_latent_2x_indices, clean_latent_4x_indices = indices.split([1, latent_padding_size, latent_window_size, 1, 2, 16], dim=1)
+            clean_latent_indices = torch.cat([clean_latent_indices_pre, clean_latent_indices_post], dim=1)
+            clean_latents_pre = start_latent.to(history_latents)
+            clean_latents_post, clean_latents_2x, clean_latents_4x = history_latents[:, :, :1 + 2 + 16, :, :].split([1, 2, 16], dim=2)
+            clean_latents = torch.cat([clean_latents_pre, clean_latents_post], dim=2)
+            # Use end image latent for the first section if provided
+            if has_end_image and is_first_section:
+                clean_latents_post = end_latent.to(history_latents)
+                clean_latents = torch.cat([clean_latents_pre, clean_latents_post], dim=2)
+            if not high_vram:
+                unload_complete_models()
+                move_model_to_device_with_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=gpu_memory_preservation)
+            if use_teacache:
+                transformer.initialize_teacache(enable_teacache=True, num_steps=steps)
+            else:
+                transformer.initialize_teacache(enable_teacache=False)
+            def callback(d):
+                preview = d['denoised']
+                preview = vae_decode_fake(preview)
+                preview = (preview * 255.0).detach().cpu().numpy().clip(0, 255).astype(np.uint8)
+                preview = einops.rearrange(preview, 'b c t h w -> (b h) (t w) c')
+                if stream.input_queue.top() == 'end':
+                    stream.output_queue.push(('end', None))
+                    raise KeyboardInterrupt('User ends the task.')
+                current_step = d['i'] + 1
+                percentage = int(100.0 * current_step / steps)
+                hint = f'Sampling {current_step}/{steps}'
+                desc = f'Total generated frames: {int(max(0, total_generated_latent_frames * 4 - 3))}, Video length: {max(0, (total_generated_latent_frames * 4 - 3) / fps_number) :.2f} seconds (FPS-30). The video is being extended now ...'
+                stream.output_queue.push(('progress', (preview, desc, make_progress_bar_html(percentage, hint))))
+                return
+            generated_latents = sample_hunyuan(
+                transformer=transformer,
+                sampler='unipc',
+                width=width,
+                height=height,
+                frames=num_frames,
+                real_guidance_scale=cfg,
+                distilled_guidance_scale=gs,
+                guidance_rescale=rs,
+                # shift=3.0,
+                num_inference_steps=steps,
+                generator=rnd,
+                prompt_embeds=llama_vec,
+                prompt_embeds_mask=llama_attention_mask,
+                prompt_poolers=clip_l_pooler,
+                negative_prompt_embeds=llama_vec_n,
+                negative_prompt_embeds_mask=llama_attention_mask_n,
+                negative_prompt_poolers=clip_l_pooler_n,
+                device=gpu,
+                dtype=torch.bfloat16,
+                image_embeddings=image_encoder_last_hidden_state,
+                latent_indices=latent_indices,
+                clean_latents=clean_latents,
+                clean_latent_indices=clean_latent_indices,
+                clean_latents_2x=clean_latents_2x,
+                clean_latent_2x_indices=clean_latent_2x_indices,
+                clean_latents_4x=clean_latents_4x,
+                clean_latent_4x_indices=clean_latent_4x_indices,
+                callback=callback,
+            )
+            if is_last_section:
+                generated_latents = torch.cat([start_latent.to(generated_latents), generated_latents], dim=2)
+            total_generated_latent_frames += int(generated_latents.shape[2])
+            history_latents = torch.cat([generated_latents.to(history_latents), history_latents], dim=2)
+            if not high_vram:
+                offload_model_from_device_for_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=8)
+                load_model_as_complete(vae, target_device=gpu)
+            real_history_latents = history_latents[:, :, :total_generated_latent_frames, :, :]
+            if history_pixels is None:
+                history_pixels = vae_decode(real_history_latents, vae).cpu()
+            else:
+                section_latent_frames = (latent_window_size * 2 + 1) if is_last_section else (latent_window_size * 2)
+                overlapped_frames = latent_window_size * 4 - 3
+                current_pixels = vae_decode(real_history_latents[:, :, :section_latent_frames], vae).cpu()
+                history_pixels = soft_append_bcthw(current_pixels, history_pixels, overlapped_frames)
+            if not high_vram:
+                unload_complete_models(vae)
+            output_filename = os.path.join(outputs_folder, f'{job_id}_{total_generated_latent_frames}.mp4')
+            save_bcthw_as_mp4(history_pixels, output_filename, fps=fps_number, crf=mp4_crf)
+            print(f'Decoded. Current latent shape {real_history_latents.shape}; pixel shape {history_pixels.shape}')
+            stream.output_queue.push(('file', output_filename))
+            if is_last_section:
+                break
+    except:
+        traceback.print_exc()
+        if not high_vram:
+            unload_complete_models(
+                text_encoder, text_encoder_2, image_encoder, vae, transformer
+            )
+    stream.output_queue.push(('end', None))
+    return
 # 20250506 pftq: Modified worker to accept video input and clean frame count
 @torch.no_grad()
 def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch):
     stream.output_queue.push(('end', None))
     return
+def get_duration(input_image, image_position, end_image, prompts, generation_mode, n_prompt, seed, resolution, total_second_length, allocation_time, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number):
     return allocation_time
 # Remove this decorator if you run on local
 @spaces.GPU(duration=get_duration)
+def process_on_gpu(input_image, image_position, end_image, prompts, generation_mode, n_prompt, seed, resolution, total_second_length, allocation_time, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number
            ):
     start = time.time()
     global stream
     stream = AsyncStream()
+    async_run(worker_start_end if generation_mode == "start_end" else worker, input_image, image_position, end_image, prompts, n_prompt, seed, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, use_teacache, mp4_crf, fps_number)
     output_filename = None
 def process(input_image,
             image_position=0,
+            end_image=None,
             prompt="",
             generation_mode="image",
             n_prompt="",
             resolution=640,
             total_second_length=5,
             latent_window_size=9,
+            steps=30,
             cfg=1.0,
             gs=10.0,
             rs=0.0,
             gpu_memory_preservation=6,
+            enable_preview=False,
             use_teacache=False,
             mp4_crf=16,
             fps_number=30
     if auto_allocation:
         allocation_time = min(total_second_length * 60 * (1.5 if use_teacache else 3.0) * (1 + ((steps - 25) / 25))**2, 600)
+    if input_image_debug_value[0] is not None or end_image_debug_value[0] is not None or prompt_debug_value[0] is not None or total_second_length_debug_value[0] is not None:
         input_image = input_image_debug_value[0]
+        end_image = end_image_debug_value[0]
         prompt = prompt_debug_value[0]
         total_second_length = total_second_length_debug_value[0]
         allocation_time = min(total_second_length_debug_value[0] * 60 * 100, 600)
+        input_image_debug_value[0] = end_image_debug_value[0] = input_video_debug_value[0] = prompt_debug_value[0] = total_second_length_debug_value[0] = None
     if torch.cuda.device_count() == 0:
         gr.Warning('Set this space to GPU config to make it work.')
     yield from process_on_gpu(input_image,
             image_position,
+            end_image,
             prompts,
             generation_mode,
             n_prompt,
         prompt = prompt_debug_value[0]
         total_second_length = total_second_length_debug_value[0]
         allocation_time = min(total_second_length_debug_value[0] * 60 * 100, 600)
+        input_image_debug_value[0] = end_image_debug_value[0] = input_video_debug_value[0] = prompt_debug_value[0] = total_second_length_debug_value[0] = None
     if torch.cuda.device_count() == 0:
         gr.Warning('Set this space to GPU config to make it work.')
     local_storage = gr.BrowserState(default_local_storage)
     with gr.Row():
         with gr.Column():
+            generation_mode = gr.Radio([["Text-to-Video", "text"], ["Image-to-Video", "image"], ["Start & end frames", "start_end"], ["Video Extension", "video"]], elem_id="generation-mode", label="Generation mode", value = "image")
             text_to_video_hint = gr.HTML("Text-to-Video badly works with a flash effect at the start. I discourage to use the Text-to-Video feature. You should rather generate an image with Flux and use Image-to-Video. You will save time.")
             input_image = gr.Image(sources='upload', type="numpy", label="Image", height=320)
+            end_image = gr.Image(sources='upload', type="numpy", label="End Frame (Optional)", height=320)
             image_position = gr.Slider(label="Image position", minimum=0, maximum=100, value=0, step=1, info='0=Video start; 100=Video end (lower quality)')
             input_video = gr.Video(sources='upload', label="Input Video", height=320)
             timeless_prompt = gr.Textbox(label="Timeless prompt", info='Used on the whole duration of the generation', value='', placeholder="The creature starts to move, fast motion, fixed camera, focus motion, consistent arm, consistent position, mute colors, insanely detailed")
                 enable_preview = gr.Checkbox(label='Enable preview', value=True, info='Display a preview around each second generated but it costs 2 sec. for each second generated.')
                 use_teacache = gr.Checkbox(label='Use TeaCache', value=False, info='Faster speed and no break in brightness, but often makes hands and fingers slightly worse.')
+                n_prompt = gr.Textbox(label="Negative Prompt", value="Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", info='Requires using normal CFG (undistilled) instead of Distilled (set Distilled=1 and CFG > 1).')
                 fps_number = gr.Slider(label="Frame per seconds", info="The model is trained for 30 fps so other fps may generate weird results", minimum=10, maximum=60, value=30, step=1)
             with gr.Accordion("Debug", open=False):
                 input_image_debug = gr.Image(type="numpy", label="Image Debug", height=320)
+                end_image_debug = gr.Image(type="numpy", label="End Image Debug", height=320)
                 input_video_debug = gr.Video(sources='upload', label="Input Video Debug", height=320)
                 prompt_debug = gr.Textbox(label="Prompt Debug", value='')
                 total_second_length_debug = gr.Slider(label="Additional Video Length to Generate (seconds) Debug", minimum=1, maximum=120, value=1, step=0.1)
             progress_desc = gr.Markdown('', elem_classes='no-generating-animation')
             progress_bar = gr.HTML('', elem_classes='no-generating-animation')
+    ips = [input_image, image_position, end_image, final_prompt, generation_mode, n_prompt, randomize_seed, seed, auto_allocation, allocation_time, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, fps_number]
     ips_video = [input_video, final_prompt, n_prompt, randomize_seed, seed, auto_allocation, allocation_time, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch]
     with gr.Row(elem_id="text_examples", visible=False):
                     [
                         None, # input_image
                         0, # image_position
+                        None, # end_image
                         "Overcrowed street in Japan, photorealistic, realistic, intricate details, 8k, insanely detailed",
                         "text", # generation_mode
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                     [
                         "./img_examples/Example2.webp", # input_image
                         0, # image_position
+                        None, # end_image
                         "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks, the man stops talking and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens",
                         "image", # generation_mode
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                     [
                         "./img_examples/Example1.png", # input_image
                         0, # image_position
+                        None, # end_image
                         "A dolphin emerges from the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
                         "image", # generation_mode
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                     [
                         "./img_examples/Example4.webp", # input_image
                         1, # image_position
+                        None, # end_image
                         "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                         "image", # generation_mode
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                     [
                         "./img_examples/Example4.webp", # input_image
                         50, # image_position
+                        None, # end_image
                         "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                         "image", # generation_mode
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                     [
                         "./img_examples/Example4.webp", # input_image
                         100, # image_position
+                        None, # end_image
                         "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                         "image", # generation_mode
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
+                        True, # randomize_seed
+                        42, # seed
+                        True, # auto_allocation
+                        180, # allocation_time
+                        672, # resolution
+                        1, # total_second_length
+                        9, # latent_window_size
+                        30, # steps
+                        1.0, # cfg
+                        10.0, # gs
+                        0.0, # rs
+                        6, # gpu_memory_preservation
+                        False, # enable_preview
+                        False, # use_teacache
+                        16, # mp4_crf
+                        30 # fps_number
+                    ],
+                ],
+            run_on_click = True,
+            fn = process,
+	        inputs = ips,
+            outputs = [result_video, preview_image, progress_desc, progress_bar, start_button, end_button, warning],
+            cache_examples = torch.cuda.device_count() > 0,
+        )
+    with gr.Row(elem_id="start_end_examples", visible=False):
+        gr.Examples(
+        label = "Examples from start and end frames",
+            examples = [
+                    [
+                        "./img_examples/Example2.webp", # input_image
+                        0, # image_position
+                        None, # end_image
+                        "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks, the man stops talking and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens",
+                        "start_end", # generation_mode
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                     [
                         "./img_examples/Example1.mp4", # input_video
                         "View of the sea as far as the eye can see, from the seaside, a piece of land is barely visible on the horizon at the middle, the sky is radiant, reflections of the sun in the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                     [
                         "./img_examples/Example1.mp4", # input_video
                         "View of the sea as far as the eye can see, from the seaside, a piece of land is barely visible on the horizon at the middle, the sky is radiant, reflections of the sun in the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
+                        "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                         True, # randomize_seed
                         42, # seed
                         True, # auto_allocation
                 [
                     None, # input_image
                     0, # image_position
+                    None, # end_image
                     "Overcrowed street in Japan, photorealistic, realistic, intricate details, 8k, insanely detailed",
                     "text", # generation_mode
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
                 [
                     "./img_examples/Example1.png", # input_image
                     0, # image_position
+                    None, # end_image
                     "A dolphin emerges from the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
                     "image", # generation_mode
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
                 [
                     "./img_examples/Example2.webp", # input_image
                     0, # image_position
+                    None, # end_image
                     "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks, the man stops talking and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens",
                     "image", # generation_mode
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
                 [
                     "./img_examples/Example2.webp", # input_image
                     0, # image_position
+                    None, # end_image
                     "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks, the woman stops talking and the woman listens A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens",
                     "image", # generation_mode
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
                 [
                     "./img_examples/Example3.jpg", # input_image
                     0, # image_position
+                    None, # end_image
                     "A boy is walking to the right, full view, full-length view, cartoon",
                     "image", # generation_mode
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
                 [
                     "./img_examples/Example4.webp", # input_image
                     100, # image_position
+                    None, # end_image
                     "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
                     "image", # generation_mode
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
         cache_examples = False,
     )
+    gr.Examples(
+        label = "🖼️ Examples from start and end frames",
+        examples = [
+                [
+                    "./img_examples/Example1.png", # input_image
+                    0, # image_position
+                    None, # end_image
+                    "A dolphin emerges from the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
+                    "start_end", # generation_mode
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
+                    True, # randomize_seed
+                    42, # seed
+                    True, # auto_allocation
+                    180, # allocation_time
+                    672, # resolution
+                    1, # total_second_length
+                    9, # latent_window_size
+                    30, # steps
+                    1.0, # cfg
+                    10.0, # gs
+                    0.0, # rs
+                    6, # gpu_memory_preservation
+                    False, # enable_preview
+                    True, # use_teacache
+                    16, # mp4_crf
+                    30 # fps_number
+                ],
+            ],
+        run_on_click = True,
+        fn = process,
+	    inputs = ips,
+        outputs = [result_video, preview_image, progress_desc, progress_bar, start_button, end_button, warning],
+        cache_examples = False,
+    )
     gr.Examples(
         label = "🎥 Examples from video",
         examples = [
                 [
                     "./img_examples/Example1.mp4", # input_video
                     "View of the sea as far as the eye can see, from the seaside, a piece of land is barely visible on the horizon at the middle, the sky is radiant, reflections of the sun in the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
+                    "Missing arm, long hand, unrealistic position, impossible contortion, visible bone, muscle contraction, poorly framed, blurred, blurry, over-smooth", # n_prompt
                     True, # randomize_seed
                     42, # seed
                     True, # auto_allocation
     def handle_generation_mode_change(generation_mode_data):
         if generation_mode_data == "text":
+            return [
+            gr.update(visible = True),  # text_to_video_hint
+            gr.update(visible = False), # image_position
+            gr.update(visible = False), # input_image
+            gr.update(visible = False), # end_image
+            gr.update(visible = False), # input_video
+            gr.update(visible = True),  # start_button
+            gr.update(visible = False), # start_button_video
+            gr.update(visible = False), # no_resize
+            gr.update(visible = False), # batch
+            gr.update(visible = False), # num_clean_frames
+            gr.update(visible = False), # vae_batch
+            gr.update(visible = False), # prompt_hint
+            gr.update(visible = True)   # fps_number
+            ]
         elif generation_mode_data == "image":
+            return [
+            gr.update(visible = False), # text_to_video_hint
+            gr.update(visible = True),  # image_position
+            gr.update(visible = True),  # input_image
+            gr.update(visible = False), # end_image
+            gr.update(visible = False), # input_video
+            gr.update(visible = True),  # start_button
+            gr.update(visible = False), # start_button_video
+            gr.update(visible = False), # no_resize
+            gr.update(visible = False), # batch
+            gr.update(visible = False), # num_clean_frames
+            gr.update(visible = False), # vae_batch
+            gr.update(visible = False), # prompt_hint
+            gr.update(visible = True)   # fps_number
+            ]
+        elif generation_mode_data == "start_end":
+            return [
+            gr.update(visible = False), # text_to_video_hint
+            gr.update(visible = False), # image_position
+            gr.update(visible = True),  # input_image
+            gr.update(visible = True),  # end_image
+            gr.update(visible = False), # input_video
+            gr.update(visible = True),  # start_button
+            gr.update(visible = False), # start_button_video
+            gr.update(visible = False), # no_resize
+            gr.update(visible = False), # batch
+            gr.update(visible = False), # num_clean_frames
+            gr.update(visible = False), # vae_batch
+            gr.update(visible = False), # prompt_hint
+            gr.update(visible = True)   # fps_number
+            ]
         elif generation_mode_data == "video":
+            return [
+            gr.update(visible = False), # text_to_video_hint
+            gr.update(visible = False), # image_position
+            gr.update(visible = False), # input_image
+            gr.update(visible = False), # end_image
+            gr.update(visible = True),  # input_video
+            gr.update(visible = False), # start_button
+            gr.update(visible = True),  # start_button_video
+            gr.update(visible = True),  # no_resize
+            gr.update(visible = True),  # batch
+            gr.update(visible = True),  # num_clean_frames
+            gr.update(visible = True),  # vae_batch
+            gr.update(visible = True),  # prompt_hint
+            gr.update(visible = False)  # fps_number
+            ]
+    def handle_field_debug_change(input_image_debug_data, input_video_debug_data, end_image_debug_data, prompt_debug_data, total_second_length_debug_data):
         print("handle_field_debug_change")
         input_image_debug_value[0] = input_image_debug_data
         input_video_debug_value[0] = input_video_debug_data
+        end_image_debug_value[0] = end_image_debug_data
         prompt_debug_value[0] = prompt_debug_data
         total_second_length_debug_value[0] = total_second_length_debug_data
         return []
     input_image_debug.upload(
         fn=handle_field_debug_change,
+        inputs=[input_image_debug, input_video_debug, end_image_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
     input_video_debug.upload(
         fn=handle_field_debug_change,
+        inputs=[input_image_debug, input_video_debug, end_image_debug, prompt_debug, total_second_length_debug],
+        outputs=[]
+    )
+    end_image_debug.upload(
+        fn=handle_field_debug_change,
+        inputs=[input_image_debug, input_video_debug, end_image_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
     prompt_debug.change(
         fn=handle_field_debug_change,
+        inputs=[input_image_debug, input_video_debug, end_image_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
     total_second_length_debug.change(
         fn=handle_field_debug_change,
+        inputs=[input_image_debug, input_video_debug, end_image_debug, prompt_debug, total_second_length_debug],
         outputs=[]
     )
     generation_mode.change(
         fn=handle_generation_mode_change,
         inputs=[generation_mode],
+        outputs=[text_to_video_hint, image_position, input_image, end_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint, fps_number]
     )
     # Update display when the page loads
         fn=handle_generation_mode_change, inputs = [
         generation_mode
     ], outputs = [
+       text_to_video_hint, image_position, input_image, end_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint, fps_number
     ]
     )