FramePack

Running

App Files Files Community

Fabrice-TIERCELIN commited on 25 days ago

Commit

50254df

verified ·

1 Parent(s): c28f966

Last frame

Browse files

Files changed (1) hide show

app.py +330 -94

app.py CHANGED Viewed

@@ -355,31 +355,36 @@ def worker(input_image, prompts, n_prompt, seed, resolution, total_second_length
         H, W, C = input_image.shape
         height, width = find_nearest_bucket(H, W, resolution=resolution)
-        input_image_np = resize_and_center_crop(input_image, target_width=width, target_height=height)
-        Image.fromarray(input_image_np).save(os.path.join(outputs_folder, f'{job_id}.png'))
-        input_image_pt = torch.from_numpy(input_image_np).float() / 127.5 - 1
-        input_image_pt = input_image_pt.permute(2, 0, 1)[None, :, None]
-        # VAE encoding
-        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'VAE encoding ...'))))
-        if not high_vram:
-            load_model_as_complete(vae, target_device=gpu)
-        start_latent = vae_encode(input_image_pt, vae)
-        # CLIP Vision
-        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'CLIP Vision encoding ...'))))
-        if not high_vram:
-            load_model_as_complete(image_encoder, target_device=gpu)
-        image_encoder_output = hf_clip_vision_encode(input_image_np, feature_extractor, image_encoder)
-        image_encoder_last_hidden_state = image_encoder_output.last_hidden_state
         # Dtype
@@ -438,7 +443,7 @@ def worker(input_image, prompts, n_prompt, seed, resolution, total_second_length
                 section_latent_frames = latent_window_size * 2
                 overlapped_frames = latent_window_size * 4 - 3
-                real_history_latents = history_latents[:, :, max(-section_latent_frames, -total_generated_latent_frames):, :, :]
                 history_pixels = soft_append_bcthw(history_pixels, vae_decode(real_history_latents, vae).cpu(), overlapped_frames)
             if not high_vram:
@@ -519,78 +524,226 @@ def worker(input_image, prompts, n_prompt, seed, resolution, total_second_length
     stream.output_queue.push(('end', None))
     return
-def get_duration(input_image, prompt, generation_mode, n_prompt, randomize_seed, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf):
-    return total_second_length * 60 * (0.9 if use_teacache else 1.5) * (1 + ((steps - 25) / 100))
-@spaces.GPU(duration=get_duration)
-def process(input_image, prompt,
-            generation_mode="image",
-            n_prompt="",
-            randomize_seed=True,
-            seed=31337,
-            resolution=640,
-            total_second_length=5,
-            latent_window_size=9,
-            steps=25,
-            cfg=1.0,
-            gs=10.0,
-            rs=0.0,
-            gpu_memory_preservation=6,
-            enable_preview=True,
-            use_teacache=False,
-            mp4_crf=16
-           ):
-    start = time.time()
-    global stream
-    if torch.cuda.device_count() == 0:
-        gr.Warning('Set this space to GPU config to make it work.')
-        yield gr.update(), gr.update(), gr.update(), gr.update(), gr.update(), gr.update()
-        return
-    if randomize_seed:
-        seed = random.randint(0, np.iinfo(np.int32).max)
-    prompts = prompt.split(";")
-    # assert input_image is not None, 'No input image!'
-    if generation_mode == "text":
-        default_height, default_width = 640, 640
-        input_image = np.ones((default_height, default_width, 3), dtype=np.uint8) * 255
-        print("No input image provided. Using a blank white image.")
-    yield None, None, '', '', gr.update(interactive=False), gr.update(interactive=True)
-    stream = AsyncStream()
-    async_run(worker, input_image, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf)
-    output_filename = None
-    while True:
-        flag, data = stream.output_queue.next()
-        if flag == 'file':
-            output_filename = data
-            yield output_filename, gr.update(), gr.update(), gr.update(), gr.update(interactive=False), gr.update(interactive=True)
-        if flag == 'progress':
-            preview, desc, html = data
-            yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
-        if flag == 'end':
-            end = time.time()
-            secondes = int(end - start)
-            minutes = math.floor(secondes / 60)
-            secondes = secondes - (minutes * 60)
-            hours = math.floor(minutes / 60)
-            minutes = minutes - (hours * 60)
-            yield output_filename, gr.update(visible=False), gr.update(), "The video has been generated in " + \
-            ((str(hours) + " h, ") if hours != 0 else "") + \
-            ((str(minutes) + " min, ") if hours != 0 or minutes != 0 else "") + \
-            str(secondes) + " sec. " + \
-            "You can upscale the result with RIFE. To make all your generated scenes consistent, you can then apply a face swap on the main character.", gr.update(interactive=True), gr.update(interactive=False)
-            break
 # 20250506 pftq: Modified worker to accept video input and clean frame count
 @spaces.GPU()
@@ -860,12 +1013,90 @@ def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_
     stream.output_queue.push(('end', None))
     return
 def get_duration_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch):
     return total_second_length * 60 * (0.9 if use_teacache else 2.3) * (1 + ((steps - 25) / 100))
 # 20250506 pftq: Modified process to pass clean frame count, etc from video_encode
 @spaces.GPU(duration=get_duration_video)
-def process_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch):
     start = time.time()
     global stream, high_vram
@@ -913,6 +1144,7 @@ def process_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, re
         if flag == 'progress':
             preview, desc, html = data
             #yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
             yield output_filename, gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True) # 20250506 pftq: Keep refreshing the video in case it got hidden when the tab was in the background
@@ -1002,6 +1234,7 @@ with block:
             generation_mode = gr.Radio([["Text-to-Video", "text"], ["Image-to-Video", "image"], ["Video Extension", "video"]], elem_id="generation-mode", label="Generation mode", value = "image")
             text_to_video_hint = gr.HTML("I discourage to use the Text-to-Video feature. You should rather generate an image with Flux and use Image-to-Video. You will save time.")
             input_image = gr.Image(sources='upload', type="numpy", label="Image", height=320)
             input_video = gr.Video(sources='upload', label="Input Video", height=320)
             timeless_prompt = gr.Textbox(label="Timeless prompt", info='Used on the whole duration of the generation', value='', placeholder="The creature starts to move, fast motion, fixed camera, focus motion, consistent arm, consistent position, mute colors, insanely detailed")
             prompt_number = gr.Slider(label="Timed prompt number", minimum=0, maximum=1000, value=0, step=1, info='Prompts will automatically appear')
@@ -1076,7 +1309,7 @@ with block:
             progress_bar = gr.HTML('', elem_classes='no-generating-animation')
     # 20250506 pftq: Updated inputs to include num_clean_frames
-    ips = [input_image, final_prompt, generation_mode, n_prompt, randomize_seed, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf]
     ips_video = [input_video, final_prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch]
     gr.Examples(
@@ -1084,6 +1317,7 @@ with block:
         examples = [
                 [
                     "./img_examples/Example1.png", # input_image
                     "A dolphin emerges from the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
@@ -1103,7 +1337,8 @@ with block:
                 ],
                 [
                     "./img_examples/Example2.webp", # input_image
-                    "A black man on the left and an Asian woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens; A black man on the left and an Asian woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
                     True, # randomize_seed
@@ -1122,7 +1357,8 @@ with block:
                 ],
                 [
                     "./img_examples/Example2.webp", # input_image
-                    "A black man on the left and an Asian woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens; A black man on the left and an Asian woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
                     True, # randomize_seed
@@ -1141,6 +1377,7 @@ with block:
                 ],
                 [
                     "./img_examples/Example3.jpg", # input_image
                     "A boy is walking to the right, full view, full-length view, cartoon",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
@@ -1221,13 +1458,12 @@ with block:
     def handle_generation_mode_change(generation_mode_data):
         if generation_mode_data == "text":
-            return [gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False)]
         elif generation_mode_data == "image":
-            return [gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False)]
         elif generation_mode_data == "video":
-            return [gr.update(visible = False), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True)]
     prompt_number.change(fn=handle_prompt_number_change, inputs=[], outputs=[])
     timeless_prompt.change(fn=handle_timeless_prompt_change, inputs=[timeless_prompt], outputs=[final_prompt])
     start_button.click(fn = check_parameters, inputs = [
@@ -1248,7 +1484,7 @@ with block:
     generation_mode.change(
         fn=handle_generation_mode_change,
         inputs=[generation_mode],
-        outputs=[text_to_video_hint, input_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint]
     )
     # Update display when the page loads
@@ -1256,7 +1492,7 @@ with block:
         fn=handle_generation_mode_change, inputs = [
         generation_mode
     ], outputs = [
-       text_to_video_hint, input_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint
     ]
     )

         H, W, C = input_image.shape
         height, width = find_nearest_bucket(H, W, resolution=resolution)
+        def get_start_latent(input_image, height, width, vae, gpu, image_encoder, high_vram):
+            input_image_np = resize_and_center_crop(input_image, target_width=width, target_height=height)
+            #Image.fromarray(input_image_np).save(os.path.join(outputs_folder, f'{job_id}.png'))
+            input_image_pt = torch.from_numpy(input_image_np).float() / 127.5 - 1
+            input_image_pt = input_image_pt.permute(2, 0, 1)[None, :, None]
+            # VAE encoding
+            stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'VAE encoding ...'))))
+            if not high_vram:
+                load_model_as_complete(vae, target_device=gpu)
+            start_latent = vae_encode(input_image_pt, vae)
+            # CLIP Vision
+            stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'CLIP Vision encoding ...'))))
+            if not high_vram:
+                load_model_as_complete(image_encoder, target_device=gpu)
+            image_encoder_last_hidden_state = hf_clip_vision_encode(input_image_np, feature_extractor, image_encoder).last_hidden_state
+            return [start_latent, image_encoder_last_hidden_state]
+        [start_latent, image_encoder_last_hidden_state] = get_start_latent(input_image, height, width, vae, gpu, image_encoder, high_vram)
         # Dtype
                 section_latent_frames = latent_window_size * 2
                 overlapped_frames = latent_window_size * 4 - 3
+                real_history_latents = history_latents[:, :, -min(section_latent_frames, total_generated_latent_frames):, :, :]
                 history_pixels = soft_append_bcthw(history_pixels, vae_decode(real_history_latents, vae).cpu(), overlapped_frames)
             if not high_vram:
     stream.output_queue.push(('end', None))
     return
+@torch.no_grad()
+def worker_last_frame(input_image, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf):
+    def encode_prompt(prompt, n_prompt):
+        llama_vec, clip_l_pooler = encode_prompt_conds(prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
+        if cfg == 1:
+            llama_vec_n, clip_l_pooler_n = torch.zeros_like(llama_vec), torch.zeros_like(clip_l_pooler)
+        else:
+            llama_vec_n, clip_l_pooler_n = encode_prompt_conds(n_prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
+        llama_vec, llama_attention_mask = crop_or_pad_yield_mask(llama_vec, length=512)
+        llama_vec_n, llama_attention_mask_n = crop_or_pad_yield_mask(llama_vec_n, length=512)
+        llama_vec = llama_vec.to(transformer.dtype)
+        llama_vec_n = llama_vec_n.to(transformer.dtype)
+        clip_l_pooler = clip_l_pooler.to(transformer.dtype)
+        clip_l_pooler_n = clip_l_pooler_n.to(transformer.dtype)
+        return [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n]
+    total_latent_sections = (total_second_length * 30) / (latent_window_size * 4)
+    total_latent_sections = int(max(round(total_latent_sections), 1))
+    job_id = generate_timestamp()
+    stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Starting ...'))))
+    try:
+        # Clean GPU
+        if not high_vram:
+            unload_complete_models(
+                text_encoder, text_encoder_2, image_encoder, vae, transformer
+            )
+        # Text encoding
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Text encoding ...'))))
+        if not high_vram:
+            fake_diffusers_current_device(text_encoder, gpu)  # since we only encode one text - that is one model move and one encode, offload is same time consumption since it is also one load and one encode.
+            load_model_as_complete(text_encoder_2, target_device=gpu)
+        prompt_parameters = []
+        for prompt_part in prompts:
+            prompt_parameters.append(encode_prompt(prompt_part, n_prompt))
+        # Processing input image
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Image processing ...'))))
+        H, W, C = input_image.shape
+        height, width = find_nearest_bucket(H, W, resolution=resolution)
+        def get_start_latent(input_image, height, width, vae, gpu, image_encoder, high_vram):
+            input_image_np = resize_and_center_crop(input_image, target_width=width, target_height=height)
+            #Image.fromarray(input_image_np).save(os.path.join(outputs_folder, f'{job_id}.png'))
+            input_image_pt = torch.from_numpy(input_image_np).float() / 127.5 - 1
+            input_image_pt = input_image_pt.permute(2, 0, 1)[None, :, None]
+            # VAE encoding
+            stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'VAE encoding ...'))))
+            if not high_vram:
+                load_model_as_complete(vae, target_device=gpu)
+            start_latent = vae_encode(input_image_pt, vae)
+            # CLIP Vision
+            stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'CLIP Vision encoding ...'))))
+            if not high_vram:
+                load_model_as_complete(image_encoder, target_device=gpu)
+            image_encoder_last_hidden_state = hf_clip_vision_encode(input_image_np, feature_extractor, image_encoder).last_hidden_state
+            return [start_latent, image_encoder_last_hidden_state]
+        [start_latent, image_encoder_last_hidden_state] = get_start_latent(input_image, height, width, vae, gpu, image_encoder, high_vram)
+        # Dtype
+        image_encoder_last_hidden_state = image_encoder_last_hidden_state.to(transformer.dtype)
+        # Sampling
+        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Start sampling ...'))))
+        rnd = torch.Generator("cpu").manual_seed(seed)
+        history_latents = torch.zeros(size=(1, 16, 16 + 2 + 1, height // 8, width // 8), dtype=torch.float32).cpu()
+        history_pixels = None
+        history_latents = torch.cat([start_latent.to(history_latents), history_latents], dim=2)
+        total_generated_latent_frames = 1
+        if enable_preview:
+            def callback(d):
+                preview = d['denoised']
+                preview = vae_decode_fake(preview)
+                preview = (preview * 255.0).detach().cpu().numpy().clip(0, 255).astype(np.uint8)
+                preview = einops.rearrange(preview, 'b c t h w -> (b h) (t w) c')
+                if stream.input_queue.top() == 'end':
+                    stream.output_queue.push(('end', None))
+                    raise KeyboardInterrupt('User ends the task.')
+                current_step = d['i'] + 1
+                percentage = int(100.0 * current_step / steps)
+                hint = f'Sampling {current_step}/{steps}'
+                desc = f'Total generated frames: {int(max(0, total_generated_latent_frames * 4 - 3))}, Video length: {max(0, (total_generated_latent_frames * 4 - 3) / 30) :.2f} seconds (FPS-30), Resolution: {height}px * {width}px. The video is being extended now ...'
+                stream.output_queue.push(('progress', (preview, desc, make_progress_bar_html(percentage, hint))))
+                return
+        else:
+            def callback(d):
+                return
+        indices = torch.arange(0, sum([1, 16, 2, 1, latent_window_size])).unsqueeze(0)
+        latent_indices, clean_latent_1x_indices, clean_latent_2x_indices, clean_latent_4x_indices, clean_latent_indices_start = indices.split([latent_window_size, 1, 2, 16, 1], dim=1)
+        clean_latent_indices = torch.cat([clean_latent_1x_indices, clean_latent_indices_start], dim=1)
+        def post_process(generated_latents, total_generated_latent_frames, history_latents, high_vram, transformer, gpu, vae, history_pixels, latent_window_size, enable_preview, section_index, total_latent_sections, outputs_folder, mp4_crf, stream):
+            total_generated_latent_frames += int(generated_latents.shape[2])
+            history_latents = torch.cat([generated_latents.to(history_latents), history_latents], dim=2)
+            if not high_vram:
+                offload_model_from_device_for_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=8)
+                load_model_as_complete(vae, target_device=gpu)
+            if history_pixels is None:
+                real_history_latents = history_latents[:, :, :total_generated_latent_frames, :, :]
+                history_pixels = vae_decode(real_history_latents, vae).cpu()
+            else:
+                section_latent_frames = latent_window_size * 2
+                overlapped_frames = latent_window_size * 4 - 3
+                real_history_latents = history_latents[:, :, :min(section_latent_frames, total_generated_latent_frames), :, :]
+                history_pixels = soft_append_bcthw(vae_decode(real_history_latents, vae).cpu(), history_pixels, overlapped_frames)
+            if not high_vram:
+                unload_complete_models()
+            if enable_preview or section_index == 0:
+                output_filename = os.path.join(outputs_folder, f'{job_id}_{total_generated_latent_frames}.mp4')
+                save_bcthw_as_mp4(history_pixels, output_filename, fps=30, crf=mp4_crf)
+                print(f'Decoded. Current latent shape pixel shape {history_pixels.shape}')
+                stream.output_queue.push(('file', output_filename))
+            return [total_generated_latent_frames, history_latents, history_pixels]
+        for section_index in range(total_latent_sections - 1, -1, -1):
+            if stream.input_queue.top() == 'end':
+                stream.output_queue.push(('end', None))
+                return
+            print(f'section_index = {section_index}, total_latent_sections = {total_latent_sections}')
+            if len(prompt_parameters) > 0:
+                [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n] = prompt_parameters.pop(len(prompt_parameters) - 1)
+            if not high_vram:
+                unload_complete_models()
+                move_model_to_device_with_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=gpu_memory_preservation)
+            if use_teacache:
+                transformer.initialize_teacache(enable_teacache=True, num_steps=steps)
+            else:
+                transformer.initialize_teacache(enable_teacache=False)
+            clean_latents_1x, clean_latents_2x, clean_latents_4x = history_latents[:, :, :sum([1, 2, 16]), :, :].split([1, 2, 16], dim=2)
+            clean_latents = torch.cat([clean_latents_1x, start_latent.to(history_latents)], dim=2)
+            generated_latents = sample_hunyuan(
+                transformer=transformer,
+                sampler='unipc',
+                width=width,
+                height=height,
+                frames=latent_window_size * 4 - 3,
+                real_guidance_scale=cfg,
+                distilled_guidance_scale=gs,
+                guidance_rescale=rs,
+                # shift=3.0,
+                num_inference_steps=steps,
+                generator=rnd,
+                prompt_embeds=llama_vec,
+                prompt_embeds_mask=llama_attention_mask,
+                prompt_poolers=clip_l_pooler,
+                negative_prompt_embeds=llama_vec_n,
+                negative_prompt_embeds_mask=llama_attention_mask_n,
+                negative_prompt_poolers=clip_l_pooler_n,
+                device=gpu,
+                dtype=torch.bfloat16,
+                image_embeddings=image_encoder_last_hidden_state,
+                latent_indices=latent_indices,
+                clean_latents=clean_latents,
+                clean_latent_indices=clean_latent_indices,
+                clean_latents_2x=clean_latents_2x,
+                clean_latent_2x_indices=clean_latent_2x_indices,
+                clean_latents_4x=clean_latents_4x,
+                clean_latent_4x_indices=clean_latent_4x_indices,
+                callback=callback,
+            )
+            [total_generated_latent_frames, history_latents, history_pixels] = post_process(generated_latents, total_generated_latent_frames, history_latents, high_vram, transformer, gpu, vae, history_pixels, latent_window_size, enable_preview, section_index, total_latent_sections, outputs_folder, mp4_crf, stream)
+    except:
+        traceback.print_exc()
+        if not high_vram:
+            unload_complete_models(
+                text_encoder, text_encoder_2, image_encoder, vae, transformer
+            )
+    stream.output_queue.push(('end', None))
+    return
 # 20250506 pftq: Modified worker to accept video input and clean frame count
 @spaces.GPU()
     stream.output_queue.push(('end', None))
     return
+def get_duration(input_image, image_position, prompt, generation_mode, n_prompt, randomize_seed, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf):
+    return total_second_length * 60 * (0.9 if use_teacache else 1.5) * (1 + ((steps - 25) / 100))
+@spaces.GPU(duration=get_duration)
+def process(input_image,
+            image_position=0,
+            prompt="",
+            generation_mode="image",
+            n_prompt="",
+            randomize_seed=True,
+            seed=31337,
+            resolution=640,
+            total_second_length=5,
+            latent_window_size=9,
+            steps=25,
+            cfg=1.0,
+            gs=10.0,
+            rs=0.0,
+            gpu_memory_preservation=6,
+            enable_preview=True,
+            use_teacache=False,
+            mp4_crf=16,
+            progress = gr.Progress()
+           ):
+    start = time.time()
+    global stream
+    if torch.cuda.device_count() == 0:
+        gr.Warning('Set this space to GPU config to make it work.')
+        yield gr.update(), gr.update(), gr.update(), gr.update(), gr.update(), gr.update()
+        return
+    if randomize_seed:
+        seed = random.randint(0, np.iinfo(np.int32).max)
+    prompts = prompt.split(";")
+    # assert input_image is not None, 'No input image!'
+    if generation_mode == "text":
+        default_height, default_width = 640, 640
+        input_image = np.ones((default_height, default_width, 3), dtype=np.uint8) * 255
+        print("No input image provided. Using a blank white image.")
+    yield None, None, '', '', gr.update(interactive=False), gr.update(interactive=True)
+    stream = AsyncStream()
+    async_run(worker_last_frame if image_position == 100 else worker, input_image, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf)
+    output_filename = None
+    while True:
+        flag, data = stream.output_queue.next()
+        if flag == 'file':
+            output_filename = data
+            yield output_filename, gr.update(), gr.update(), gr.update(), gr.update(interactive=False), gr.update(interactive=True)
+        if flag == 'progress':
+            preview, desc, html = data
+            progress(None, desc = desc)
+            yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
+        if flag == 'end':
+            end = time.time()
+            secondes = int(end - start)
+            minutes = math.floor(secondes / 60)
+            secondes = secondes - (minutes * 60)
+            hours = math.floor(minutes / 60)
+            minutes = minutes - (hours * 60)
+            yield output_filename, gr.update(visible=False), gr.update(), "The video has been generated in " + \
+            ((str(hours) + " h, ") if hours != 0 else "") + \
+            ((str(minutes) + " min, ") if hours != 0 or minutes != 0 else "") + \
+            str(secondes) + " sec. " + \
+            "You can upscale the result with RIFE. To make all your generated scenes consistent, you can then apply a face swap on the main character.", gr.update(interactive=True), gr.update(interactive=False)
+            break
 def get_duration_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch):
     return total_second_length * 60 * (0.9 if use_teacache else 2.3) * (1 + ((steps - 25) / 100))
 # 20250506 pftq: Modified process to pass clean frame count, etc from video_encode
 @spaces.GPU(duration=get_duration_video)
+def process_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch,
+    progress = gr.Progress()):
     start = time.time()
     global stream, high_vram
         if flag == 'progress':
             preview, desc, html = data
+            progress(None, desc = desc)
             #yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
             yield output_filename, gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True) # 20250506 pftq: Keep refreshing the video in case it got hidden when the tab was in the background
             generation_mode = gr.Radio([["Text-to-Video", "text"], ["Image-to-Video", "image"], ["Video Extension", "video"]], elem_id="generation-mode", label="Generation mode", value = "image")
             text_to_video_hint = gr.HTML("I discourage to use the Text-to-Video feature. You should rather generate an image with Flux and use Image-to-Video. You will save time.")
             input_image = gr.Image(sources='upload', type="numpy", label="Image", height=320)
+            image_position = gr.Slider(label="Image position", minimum=0, maximum=100, value=0, step=100, info='0=Video start; 100=Video end')
             input_video = gr.Video(sources='upload', label="Input Video", height=320)
             timeless_prompt = gr.Textbox(label="Timeless prompt", info='Used on the whole duration of the generation', value='', placeholder="The creature starts to move, fast motion, fixed camera, focus motion, consistent arm, consistent position, mute colors, insanely detailed")
             prompt_number = gr.Slider(label="Timed prompt number", minimum=0, maximum=1000, value=0, step=1, info='Prompts will automatically appear')
             progress_bar = gr.HTML('', elem_classes='no-generating-animation')
     # 20250506 pftq: Updated inputs to include num_clean_frames
+    ips = [input_image, image_position, final_prompt, generation_mode, n_prompt, randomize_seed, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf]
     ips_video = [input_video, final_prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch]
     gr.Examples(
         examples = [
                 [
                     "./img_examples/Example1.png", # input_image
+                    0, # image_position
                     "A dolphin emerges from the water, photorealistic, realistic, intricate details, 8k, insanely detailed",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
                 ],
                 [
                     "./img_examples/Example2.webp", # input_image
+                    0, # image_position
+                    "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
                     True, # randomize_seed
                 ],
                 [
                     "./img_examples/Example2.webp", # input_image
+                    0, # image_position
+                    "A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The woman talks and the man listens; A man on the left and a woman on the right face each other ready to start a conversation, large space between the persons, full view, full-length view, 3D, pixar, 3D render, CGI. The man talks and the woman listens",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
                     True, # randomize_seed
                 ],
                 [
                     "./img_examples/Example3.jpg", # input_image
+                    0, # image_position
                     "A boy is walking to the right, full view, full-length view, cartoon",
                     "image", # generation_mode
                     "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
     def handle_generation_mode_change(generation_mode_data):
         if generation_mode_data == "text":
+            return [gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False)]
         elif generation_mode_data == "image":
+            return [gr.update(visible = False), gr.update(visible = True), gr.update(visible = True), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = False)]
         elif generation_mode_data == "video":
+            return [gr.update(visible = False), gr.update(visible = False), gr.update(visible = False), gr.update(visible = True), gr.update(visible = False), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True), gr.update(visible = True)]
     prompt_number.change(fn=handle_prompt_number_change, inputs=[], outputs=[])
     timeless_prompt.change(fn=handle_timeless_prompt_change, inputs=[timeless_prompt], outputs=[final_prompt])
     start_button.click(fn = check_parameters, inputs = [
     generation_mode.change(
         fn=handle_generation_mode_change,
         inputs=[generation_mode],
+        outputs=[text_to_video_hint, image_position, input_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint]
     )
     # Update display when the page loads
         fn=handle_generation_mode_change, inputs = [
         generation_mode
     ], outputs = [
+       text_to_video_hint, image_position, input_image, input_video, start_button, start_button_video, no_resize, batch, num_clean_frames, vae_batch, prompt_hint
     ]
     )