FramePack

Running

App Files Files Community

Fabrice-TIERCELIN commited on 24 days ago

Commit

b0d63fe

verified ·

1 Parent(s): c1f8bed

Merge code

Browse files

Files changed (1) hide show

app.py +53 -246

app.py CHANGED Viewed

@@ -304,7 +304,8 @@ def set_mp4_comments_imageio_ffmpeg(input_file, comments):
         return False
 @torch.no_grad()
-def worker(input_image, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf):
     def encode_prompt(prompt, n_prompt):
         llama_vec, clip_l_pooler = encode_prompt_conds(prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
@@ -397,9 +398,10 @@ def worker(input_image, prompts, n_prompt, seed, resolution, total_second_length
         rnd = torch.Generator("cpu").manual_seed(seed)
         history_latents = torch.zeros(size=(1, 16, 16 + 2 + 1, height // 8, width // 8), dtype=torch.float32).cpu()
         history_pixels = None
-        history_latents = torch.cat([history_latents, start_latent.to(history_latents)], dim=2)
         total_generated_latent_frames = 1
         if enable_preview:
@@ -425,252 +427,35 @@ def worker(input_image, prompts, n_prompt, seed, resolution, total_second_length
                 return
         indices = torch.arange(0, sum([1, 16, 2, 1, latent_window_size])).unsqueeze(0)
-        clean_latent_indices_start, clean_latent_4x_indices, clean_latent_2x_indices, clean_latent_1x_indices, latent_indices = indices.split([1, 16, 2, 1, latent_window_size], dim=1)
-        clean_latent_indices = torch.cat([clean_latent_indices_start, clean_latent_1x_indices], dim=1)
-        def post_process(generated_latents, total_generated_latent_frames, history_latents, high_vram, transformer, gpu, vae, history_pixels, latent_window_size, enable_preview, section_index, total_latent_sections, outputs_folder, mp4_crf, stream):
-            total_generated_latent_frames += int(generated_latents.shape[2])
-            history_latents = torch.cat([history_latents, generated_latents.to(history_latents)], dim=2)
-            if not high_vram:
-                offload_model_from_device_for_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=8)
-                load_model_as_complete(vae, target_device=gpu)
-            if history_pixels is None:
-                real_history_latents = history_latents[:, :, -total_generated_latent_frames:, :, :]
-                history_pixels = vae_decode(real_history_latents, vae).cpu()
-            else:
-                section_latent_frames = latent_window_size * 2
-                overlapped_frames = latent_window_size * 4 - 3
-                real_history_latents = history_latents[:, :, -min(section_latent_frames, total_generated_latent_frames):, :, :]
-                history_pixels = soft_append_bcthw(history_pixels, vae_decode(real_history_latents, vae).cpu(), overlapped_frames)
-            if not high_vram:
-                unload_complete_models()
-            if enable_preview or section_index == total_latent_sections - 1:
-                output_filename = os.path.join(outputs_folder, f'{job_id}_{total_generated_latent_frames}.mp4')
-                save_bcthw_as_mp4(history_pixels, output_filename, fps=30, crf=mp4_crf)
-                print(f'Decoded. Current latent shape pixel shape {history_pixels.shape}')
-                stream.output_queue.push(('file', output_filename))
-            return [total_generated_latent_frames, history_latents, history_pixels]
-        for section_index in range(total_latent_sections):
-            if stream.input_queue.top() == 'end':
-                stream.output_queue.push(('end', None))
-                return
-            print(f'section_index = {section_index}, total_latent_sections = {total_latent_sections}')
-            if len(prompt_parameters) > 0:
-                [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n] = prompt_parameters.pop(0)
-            if not high_vram:
-                unload_complete_models()
-                move_model_to_device_with_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=gpu_memory_preservation)
-            if use_teacache:
-                transformer.initialize_teacache(enable_teacache=True, num_steps=steps)
-            else:
-                transformer.initialize_teacache(enable_teacache=False)
-            clean_latents_4x, clean_latents_2x, clean_latents_1x = history_latents[:, :, -sum([16, 2, 1]):, :, :].split([16, 2, 1], dim=2)
-            clean_latents = torch.cat([start_latent.to(history_latents), clean_latents_1x], dim=2)
-            generated_latents = sample_hunyuan(
-                transformer=transformer,
-                sampler='unipc',
-                width=width,
-                height=height,
-                frames=latent_window_size * 4 - 3,
-                real_guidance_scale=cfg,
-                distilled_guidance_scale=gs,
-                guidance_rescale=rs,
-                # shift=3.0,
-                num_inference_steps=steps,
-                generator=rnd,
-                prompt_embeds=llama_vec,
-                prompt_embeds_mask=llama_attention_mask,
-                prompt_poolers=clip_l_pooler,
-                negative_prompt_embeds=llama_vec_n,
-                negative_prompt_embeds_mask=llama_attention_mask_n,
-                negative_prompt_poolers=clip_l_pooler_n,
-                device=gpu,
-                dtype=torch.bfloat16,
-                image_embeddings=image_encoder_last_hidden_state,
-                latent_indices=latent_indices,
-                clean_latents=clean_latents,
-                clean_latent_indices=clean_latent_indices,
-                clean_latents_2x=clean_latents_2x,
-                clean_latent_2x_indices=clean_latent_2x_indices,
-                clean_latents_4x=clean_latents_4x,
-                clean_latent_4x_indices=clean_latent_4x_indices,
-                callback=callback,
-            )
-            [total_generated_latent_frames, history_latents, history_pixels] = post_process(generated_latents, total_generated_latent_frames, history_latents, high_vram, transformer, gpu, vae, history_pixels, latent_window_size, enable_preview, section_index, total_latent_sections, outputs_folder, mp4_crf, stream)
-    except:
-        traceback.print_exc()
-        if not high_vram:
-            unload_complete_models(
-                text_encoder, text_encoder_2, image_encoder, vae, transformer
-            )
-    stream.output_queue.push(('end', None))
-    return
-@torch.no_grad()
-def worker_last_frame(input_image, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf):
-    def encode_prompt(prompt, n_prompt):
-        llama_vec, clip_l_pooler = encode_prompt_conds(prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
-        if cfg == 1:
-            llama_vec_n, clip_l_pooler_n = torch.zeros_like(llama_vec), torch.zeros_like(clip_l_pooler)
-        else:
-            llama_vec_n, clip_l_pooler_n = encode_prompt_conds(n_prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
-        llama_vec, llama_attention_mask = crop_or_pad_yield_mask(llama_vec, length=512)
-        llama_vec_n, llama_attention_mask_n = crop_or_pad_yield_mask(llama_vec_n, length=512)
-        llama_vec = llama_vec.to(transformer.dtype)
-        llama_vec_n = llama_vec_n.to(transformer.dtype)
-        clip_l_pooler = clip_l_pooler.to(transformer.dtype)
-        clip_l_pooler_n = clip_l_pooler_n.to(transformer.dtype)
-        return [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n]
-    total_latent_sections = (total_second_length * 30) / (latent_window_size * 4)
-    total_latent_sections = int(max(round(total_latent_sections), 1))
-    job_id = generate_timestamp()
-    stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Starting ...'))))
-    try:
-        # Clean GPU
-        if not high_vram:
-            unload_complete_models(
-                text_encoder, text_encoder_2, image_encoder, vae, transformer
-            )
-        # Text encoding
-        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Text encoding ...'))))
-        if not high_vram:
-            fake_diffusers_current_device(text_encoder, gpu)  # since we only encode one text - that is one model move and one encode, offload is same time consumption since it is also one load and one encode.
-            load_model_as_complete(text_encoder_2, target_device=gpu)
-        prompt_parameters = []
-        for prompt_part in prompts:
-            prompt_parameters.append(encode_prompt(prompt_part, n_prompt))
-        # Processing input image
-        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Image processing ...'))))
-        H, W, C = input_image.shape
-        height, width = find_nearest_bucket(H, W, resolution=resolution)
-        def get_start_latent(input_image, height, width, vae, gpu, image_encoder, high_vram):
-            input_image_np = resize_and_center_crop(input_image, target_width=width, target_height=height)
-            #Image.fromarray(input_image_np).save(os.path.join(outputs_folder, f'{job_id}.png'))
-            input_image_pt = torch.from_numpy(input_image_np).float() / 127.5 - 1
-            input_image_pt = input_image_pt.permute(2, 0, 1)[None, :, None]
-            # VAE encoding
-            stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'VAE encoding ...'))))
-            if not high_vram:
-                load_model_as_complete(vae, target_device=gpu)
-            start_latent = vae_encode(input_image_pt, vae)
-            # CLIP Vision
-            stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'CLIP Vision encoding ...'))))
-            if not high_vram:
-                load_model_as_complete(image_encoder, target_device=gpu)
-            image_encoder_last_hidden_state = hf_clip_vision_encode(input_image_np, feature_extractor, image_encoder).last_hidden_state
-            return [start_latent, image_encoder_last_hidden_state]
-        [start_latent, image_encoder_last_hidden_state] = get_start_latent(input_image, height, width, vae, gpu, image_encoder, high_vram)
-        # Dtype
-        image_encoder_last_hidden_state = image_encoder_last_hidden_state.to(transformer.dtype)
-        # Sampling
-        stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Start sampling ...'))))
-        rnd = torch.Generator("cpu").manual_seed(seed)
-        history_latents = torch.zeros(size=(1, 16, 16 + 2 + 1, height // 8, width // 8), dtype=torch.float32).cpu()
-        history_pixels = None
-        history_latents = torch.cat([start_latent.to(history_latents), history_latents], dim=2)
-        total_generated_latent_frames = 1
-        if enable_preview:
-            def callback(d):
-                preview = d['denoised']
-                preview = vae_decode_fake(preview)
-                preview = (preview * 255.0).detach().cpu().numpy().clip(0, 255).astype(np.uint8)
-                preview = einops.rearrange(preview, 'b c t h w -> (b h) (t w) c')
-                if stream.input_queue.top() == 'end':
-                    stream.output_queue.push(('end', None))
-                    raise KeyboardInterrupt('User ends the task.')
-                current_step = d['i'] + 1
-                percentage = int(100.0 * current_step / steps)
-                hint = f'Sampling {current_step}/{steps}'
-                desc = f'Total generated frames: {int(max(0, total_generated_latent_frames * 4 - 3))}, Video length: {max(0, (total_generated_latent_frames * 4 - 3) / 30) :.2f} seconds (FPS-30), Resolution: {height}px * {width}px. The video is being extended now ...'
-                stream.output_queue.push(('progress', (preview, desc, make_progress_bar_html(percentage, hint))))
-                return
         else:
-            def callback(d):
-                return
-        indices = torch.arange(0, sum([1, 16, 2, 1, latent_window_size])).unsqueeze(0)
-        latent_indices, clean_latent_1x_indices, clean_latent_2x_indices, clean_latent_4x_indices, clean_latent_indices_start = indices.split([latent_window_size, 1, 2, 16, 1], dim=1)
-        clean_latent_indices = torch.cat([clean_latent_1x_indices, clean_latent_indices_start], dim=1)
         def post_process(generated_latents, total_generated_latent_frames, history_latents, high_vram, transformer, gpu, vae, history_pixels, latent_window_size, enable_preview, section_index, total_latent_sections, outputs_folder, mp4_crf, stream):
             total_generated_latent_frames += int(generated_latents.shape[2])
-            history_latents = torch.cat([generated_latents.to(history_latents), history_latents], dim=2)
             if not high_vram:
                 offload_model_from_device_for_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=8)
                 load_model_as_complete(vae, target_device=gpu)
             if history_pixels is None:
-                real_history_latents = history_latents[:, :, :total_generated_latent_frames, :, :]
                 history_pixels = vae_decode(real_history_latents, vae).cpu()
             else:
                 section_latent_frames = latent_window_size * 2
                 overlapped_frames = latent_window_size * 4 - 3
-                real_history_latents = history_latents[:, :, :min(section_latent_frames, total_generated_latent_frames), :, :]
-                history_pixels = soft_append_bcthw(vae_decode(real_history_latents, vae).cpu(), history_pixels, overlapped_frames)
             if not high_vram:
                 unload_complete_models()
-            if enable_preview or section_index == 0:
                 output_filename = os.path.join(outputs_folder, f'{job_id}_{total_generated_latent_frames}.mp4')
                 save_bcthw_as_mp4(history_pixels, output_filename, fps=30, crf=mp4_crf)
@@ -680,7 +465,7 @@ def worker_last_frame(input_image, prompts, n_prompt, seed, resolution, total_se
                 stream.output_queue.push(('file', output_filename))
             return [total_generated_latent_frames, history_latents, history_pixels]
-        for section_index in range(total_latent_sections - 1, -1, -1):
             if stream.input_queue.top() == 'end':
                 stream.output_queue.push(('end', None))
                 return
@@ -688,7 +473,7 @@ def worker_last_frame(input_image, prompts, n_prompt, seed, resolution, total_se
             print(f'section_index = {section_index}, total_latent_sections = {total_latent_sections}')
             if len(prompt_parameters) > 0:
-                [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n] = prompt_parameters.pop(len(prompt_parameters) - 1)
             if not high_vram:
                 unload_complete_models()
@@ -699,8 +484,12 @@ def worker_last_frame(input_image, prompts, n_prompt, seed, resolution, total_se
             else:
                 transformer.initialize_teacache(enable_teacache=False)
-            clean_latents_1x, clean_latents_2x, clean_latents_4x = history_latents[:, :, :sum([1, 2, 16]), :, :].split([1, 2, 16], dim=2)
-            clean_latents = torch.cat([clean_latents_1x, start_latent.to(history_latents)], dim=2)
             generated_latents = sample_hunyuan(
                 transformer=transformer,
@@ -791,7 +580,9 @@ def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_
         stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Video processing ...'))))
         # 20250506 pftq: Encode video
-        start_latent, input_image_np, video_latents, fps, height, width, input_video_pixels  = video_encode(input_video, resolution, no_resize, vae, vae_batch_size=vae_batch, device=gpu)
         # CLIP Vision
         stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'CLIP Vision encoding ...'))))
@@ -881,7 +672,7 @@ def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_
                     if effective_clean_frames > 0 and split_idx < len(splits):
                         clean_latents_1x = splits[split_idx]
-            clean_latents = torch.cat([start_latent.to(history_latents), clean_latents_1x], dim=2)
             # 20250507 pftq: Fix for <=1 sec videos.
             max_frames = min(latent_window_size * 4 - 3, history_latents.shape[2] * 4)
@@ -900,7 +691,7 @@ def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_
             rnd = torch.Generator("cpu").manual_seed(seed)
             # 20250506 pftq: Initialize history_latents with video latents
-            history_latents = video_latents.cpu()
             total_generated_latent_frames = history_latents.shape[2]
             # 20250506 pftq: Initialize history_pixels to fix UnboundLocalError
             history_pixels = None
@@ -1013,7 +804,7 @@ def worker_video(input_video, prompts, n_prompt, seed, batch, resolution, total_
     stream.output_queue.push(('end', None))
     return
-def get_duration(input_image, image_position, prompt, generation_mode, n_prompt, randomize_seed, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf, progress = None):
     return total_second_length * 60 * (0.9 if use_teacache else 1.5) * (1 + ((steps - 25) / 100))
 @spaces.GPU(duration=get_duration)
@@ -1034,8 +825,7 @@ def process(input_image,
             gpu_memory_preservation=6,
             enable_preview=True,
             use_teacache=False,
-            mp4_crf=16,
-            progress = gr.Progress()
            ):
     start = time.time()
     global stream
@@ -1060,7 +850,7 @@ def process(input_image,
     stream = AsyncStream()
-    async_run(worker_last_frame if image_position == 100 else worker, input_image, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf)
     output_filename = None
@@ -1073,7 +863,6 @@ def process(input_image,
         if flag == 'progress':
             preview, desc, html = data
-            progress(None, desc = desc)
             yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
         if flag == 'end':
@@ -1090,13 +879,12 @@ def process(input_image,
             "You can upscale the result with RIFE. To make all your generated scenes consistent, you can then apply a face swap on the main character.", gr.update(interactive=True), gr.update(interactive=False)
             break
-def get_duration_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch, progress = None):
     return total_second_length * 60 * (0.9 if use_teacache else 2.3) * (1 + ((steps - 25) / 100))
 # 20250506 pftq: Modified process to pass clean frame count, etc from video_encode
 @spaces.GPU(duration=get_duration_video)
-def process_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch,
-    progress = gr.Progress()):
     start = time.time()
     global stream, high_vram
@@ -1144,7 +932,6 @@ def process_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, re
         if flag == 'progress':
             preview, desc, html = data
-            progress(None, desc = desc)
             #yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
             yield output_filename, gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True) # 20250506 pftq: Keep refreshing the video in case it got hidden when the tab was in the background
@@ -1234,7 +1021,7 @@ with block:
             generation_mode = gr.Radio([["Text-to-Video", "text"], ["Image-to-Video", "image"], ["Video Extension", "video"]], elem_id="generation-mode", label="Generation mode", value = "image")
             text_to_video_hint = gr.HTML("I discourage to use the Text-to-Video feature. You should rather generate an image with Flux and use Image-to-Video. You will save time.")
             input_image = gr.Image(sources='upload', type="numpy", label="Image", height=320)
-            image_position = gr.Slider(label="Image position", minimum=0, maximum=100, value=0, step=100, info='0=Video start; 100=Video end')
             input_video = gr.Video(sources='upload', label="Input Video", height=320)
             timeless_prompt = gr.Textbox(label="Timeless prompt", info='Used on the whole duration of the generation', value='', placeholder="The creature starts to move, fast motion, fixed camera, focus motion, consistent arm, consistent position, mute colors, insanely detailed")
             prompt_number = gr.Slider(label="Timed prompt number", minimum=0, maximum=1000, value=0, step=1, info='Prompts will automatically appear')
@@ -1394,6 +1181,26 @@ with block:
                     False, # enable_preview
                     True, # use_teacache
                     16 # mp4_crf
                 ]
             ],
         run_on_click = True,

         return False
 @torch.no_grad()
+def worker(input_image, image_position, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf):
+    is_last_frame = (image_position == 100)
     def encode_prompt(prompt, n_prompt):
         llama_vec, clip_l_pooler = encode_prompt_conds(prompt, text_encoder, text_encoder_2, tokenizer, tokenizer_2)
         rnd = torch.Generator("cpu").manual_seed(seed)
         history_latents = torch.zeros(size=(1, 16, 16 + 2 + 1, height // 8, width // 8), dtype=torch.float32).cpu()
+        start_latent = start_latent.to(history_latents)
         history_pixels = None
+        history_latents = torch.cat([start_latent, history_latents] if is_last_frame else [history_latents, start_latent], dim=2)
         total_generated_latent_frames = 1
         if enable_preview:
                 return
         indices = torch.arange(0, sum([1, 16, 2, 1, latent_window_size])).unsqueeze(0)
+        if is_last_frame:
+            latent_indices, clean_latent_1x_indices, clean_latent_2x_indices, clean_latent_4x_indices, clean_latent_indices_start = indices.split([latent_window_size, 1, 2, 16, 1], dim=1)
+            clean_latent_indices = torch.cat([clean_latent_1x_indices, clean_latent_indices_start], dim=1)
         else:
+            clean_latent_indices_start, clean_latent_4x_indices, clean_latent_2x_indices, clean_latent_1x_indices, latent_indices = indices.split([1, 16, 2, 1, latent_window_size], dim=1)
+            clean_latent_indices = torch.cat([clean_latent_indices_start, clean_latent_1x_indices], dim=1)
         def post_process(generated_latents, total_generated_latent_frames, history_latents, high_vram, transformer, gpu, vae, history_pixels, latent_window_size, enable_preview, section_index, total_latent_sections, outputs_folder, mp4_crf, stream):
             total_generated_latent_frames += int(generated_latents.shape[2])
+            history_latents = torch.cat([generated_latents.to(history_latents), history_latents], dim=2) if is_last_frame else torch.cat([history_latents, generated_latents.to(history_latents)], dim=2)
             if not high_vram:
                 offload_model_from_device_for_memory_preservation(transformer, target_device=gpu, preserved_memory_gb=8)
                 load_model_as_complete(vae, target_device=gpu)
             if history_pixels is None:
+                real_history_latents = history_latents[:, :, :total_generated_latent_frames, :, :] if is_last_frame else history_latents[:, :, -total_generated_latent_frames:, :, :]
                 history_pixels = vae_decode(real_history_latents, vae).cpu()
             else:
                 section_latent_frames = latent_window_size * 2
                 overlapped_frames = latent_window_size * 4 - 3
+                real_history_latents = history_latents[:, :, :min(section_latent_frames, total_generated_latent_frames), :, :] if is_last_frame else history_latents[:, :, -min(section_latent_frames, total_generated_latent_frames):, :, :]
+                history_pixels = soft_append_bcthw(vae_decode(real_history_latents, vae).cpu(), history_pixels, overlapped_frames) if is_last_frame else soft_append_bcthw(history_pixels, vae_decode(real_history_latents, vae).cpu(), overlapped_frames)
             if not high_vram:
                 unload_complete_models()
+            if enable_preview or section_index == (0 if is_last_frame else (total_latent_sections - 1)):
                 output_filename = os.path.join(outputs_folder, f'{job_id}_{total_generated_latent_frames}.mp4')
                 save_bcthw_as_mp4(history_pixels, output_filename, fps=30, crf=mp4_crf)
                 stream.output_queue.push(('file', output_filename))
             return [total_generated_latent_frames, history_latents, history_pixels]
+        for section_index in range(total_latent_sections - 1, -1, -1) if is_last_frame else range(total_latent_sections):
             if stream.input_queue.top() == 'end':
                 stream.output_queue.push(('end', None))
                 return
             print(f'section_index = {section_index}, total_latent_sections = {total_latent_sections}')
             if len(prompt_parameters) > 0:
+                [llama_vec, clip_l_pooler, llama_vec_n, clip_l_pooler_n, llama_attention_mask, llama_attention_mask_n] = prompt_parameters.pop((len(prompt_parameters) - 1) if is_last_frame else 0)
             if not high_vram:
                 unload_complete_models()
             else:
                 transformer.initialize_teacache(enable_teacache=False)
+            if is_last_frame:
+                clean_latents_1x, clean_latents_2x, clean_latents_4x = history_latents[:, :, :sum([1, 2, 16]), :, :].split([1, 2, 16], dim=2)
+                clean_latents = torch.cat([clean_latents_1x, start_latent], dim=2)
+            else:
+                clean_latents_4x, clean_latents_2x, clean_latents_1x = history_latents[:, :, -sum([16, 2, 1]):, :, :].split([16, 2, 1], dim=2)
+                clean_latents = torch.cat([start_latent, clean_latents_1x], dim=2)
             generated_latents = sample_hunyuan(
                 transformer=transformer,
         stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'Video processing ...'))))
         # 20250506 pftq: Encode video
+        start_latent, input_image_np, video_latents, fps, height, width = video_encode(input_video, resolution, no_resize, vae, vae_batch_size=vae_batch, device=gpu)[:6]
+        start_latent = start_latent.to(dtype=torch.float32).cpu()
+        video_latents = video_latents.cpu()
         # CLIP Vision
         stream.output_queue.push(('progress', (None, '', make_progress_bar_html(0, 'CLIP Vision encoding ...'))))
                     if effective_clean_frames > 0 and split_idx < len(splits):
                         clean_latents_1x = splits[split_idx]
+            clean_latents = torch.cat([start_latent, clean_latents_1x], dim=2)
             # 20250507 pftq: Fix for <=1 sec videos.
             max_frames = min(latent_window_size * 4 - 3, history_latents.shape[2] * 4)
             rnd = torch.Generator("cpu").manual_seed(seed)
             # 20250506 pftq: Initialize history_latents with video latents
+            history_latents = video_latents
             total_generated_latent_frames = history_latents.shape[2]
             # 20250506 pftq: Initialize history_pixels to fix UnboundLocalError
             history_pixels = None
     stream.output_queue.push(('end', None))
     return
+def get_duration(input_image, image_position, prompt, generation_mode, n_prompt, randomize_seed, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf):
     return total_second_length * 60 * (0.9 if use_teacache else 1.5) * (1 + ((steps - 25) / 100))
 @spaces.GPU(duration=get_duration)
             gpu_memory_preservation=6,
             enable_preview=True,
             use_teacache=False,
+            mp4_crf=16
            ):
     start = time.time()
     global stream
     stream = AsyncStream()
+    async_run(worker, input_image, image_position, prompts, n_prompt, seed, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, mp4_crf)
     output_filename = None
         if flag == 'progress':
             preview, desc, html = data
             yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
         if flag == 'end':
             "You can upscale the result with RIFE. To make all your generated scenes consistent, you can then apply a face swap on the main character.", gr.update(interactive=True), gr.update(interactive=False)
             break
+def get_duration_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch):
     return total_second_length * 60 * (0.9 if use_teacache else 2.3) * (1 + ((steps - 25) / 100))
 # 20250506 pftq: Modified process to pass clean frame count, etc from video_encode
 @spaces.GPU(duration=get_duration_video)
+def process_video(input_video, prompt, n_prompt, randomize_seed, seed, batch, resolution, total_second_length, latent_window_size, steps, cfg, gs, rs, gpu_memory_preservation, enable_preview, use_teacache, no_resize, mp4_crf, num_clean_frames, vae_batch):
     start = time.time()
     global stream, high_vram
         if flag == 'progress':
             preview, desc, html = data
             #yield gr.update(), gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True)
             yield output_filename, gr.update(visible=True, value=preview), desc, html, gr.update(interactive=False), gr.update(interactive=True) # 20250506 pftq: Keep refreshing the video in case it got hidden when the tab was in the background
             generation_mode = gr.Radio([["Text-to-Video", "text"], ["Image-to-Video", "image"], ["Video Extension", "video"]], elem_id="generation-mode", label="Generation mode", value = "image")
             text_to_video_hint = gr.HTML("I discourage to use the Text-to-Video feature. You should rather generate an image with Flux and use Image-to-Video. You will save time.")
             input_image = gr.Image(sources='upload', type="numpy", label="Image", height=320)
+            image_position = gr.Slider(label="Image position", minimum=0, maximum=100, value=0, step=100, info='0=Video start; 100=Video end (lower quality)')
             input_video = gr.Video(sources='upload', label="Input Video", height=320)
             timeless_prompt = gr.Textbox(label="Timeless prompt", info='Used on the whole duration of the generation', value='', placeholder="The creature starts to move, fast motion, fixed camera, focus motion, consistent arm, consistent position, mute colors, insanely detailed")
             prompt_number = gr.Slider(label="Timed prompt number", minimum=0, maximum=1000, value=0, step=1, info='Prompts will automatically appear')
                     False, # enable_preview
                     True, # use_teacache
                     16 # mp4_crf
+                ],
+                [
+                    "./img_examples/Example4.webp", # input_image
+                    100, # image_position
+                    "A building starting to explode, photorealistic, realisitc, 8k, insanely detailed",
+                    "image", # generation_mode
+                    "Missing arm, unrealistic position, impossible contortion, visible bone, muscle contraction, blurred, blurry", # n_prompt
+                    True, # randomize_seed
+                    42, # seed
+                    672, # resolution
+                    1, # total_second_length
+                    9, # latent_window_size
+                    25, # steps
+                    1.0, # cfg
+                    10.0, # gs
+                    0.0, # rs
+                    6, # gpu_memory_preservation
+                    False, # enable_preview
+                    False, # use_teacache
+                    16 # mp4_crf
                 ]
             ],
         run_on_click = True,