From 50483a3b545b28774cb27346a11c2e4c51f00752 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Thu, 16 Oct 2025 00:42:33 +0300 Subject: [PATCH] FlashVSR tweaks --- FlashVSR/TCDecoder.py | 2 +- .../wanvideo_1_3B_FlashVSR_upscale_example.json | 2 +- nodes_sampler.py | 6 +++--- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/FlashVSR/TCDecoder.py b/FlashVSR/TCDecoder.py index 3e50e01..1dbef42 100644 --- a/FlashVSR/TCDecoder.py +++ b/FlashVSR/TCDecoder.py @@ -241,7 +241,7 @@ class TAEHV(nn.Module): trim_flag = self.mem[-8] is None # keeps original relative check if cond is not None: - shuffled = self.pixel_shuffle(cond) + shuffled = self.pixel_shuffle(cond.to(x)) x = torch.cat([shuffled[:, :x.shape[1]], x], dim=2) x, self.mem = apply_model_with_memblocks(self.decoder, x, parallel, show_progress_bar, mem=self.mem) diff --git a/example_workflows/wanvideo_1_3B_FlashVSR_upscale_example.json b/example_workflows/wanvideo_1_3B_FlashVSR_upscale_example.json index 7f21f46..783c468 100644 --- a/example_workflows/wanvideo_1_3B_FlashVSR_upscale_example.json +++ b/example_workflows/wanvideo_1_3B_FlashVSR_upscale_example.json @@ -875,7 +875,7 @@ "Node name for S&R": "WanVideoExtraModelSelect" }, "widgets_values": [ - "WanVideo\\FlashVSR\\Wan2_1_FlashVSR_PQ_proj_model_bf16.safetensors" + "WanVideo\\FlashVSR\\Wan2_1_FlashVSR_LQ_proj_model_bf16.safetensors" ], "color": "#223", "bgcolor": "#335" diff --git a/nodes_sampler.py b/nodes_sampler.py index 7059022..7ed6829 100644 --- a/nodes_sampler.py +++ b/nodes_sampler.py @@ -838,9 +838,9 @@ class WanVideoSampler: missing_frames = num_frames + 4 - flashvsr_LQ_images.shape[0] last_frame = flashvsr_LQ_images[-1:].repeat(missing_frames, 1, 1, 1) flashvsr_LQ_images = torch.cat([flashvsr_LQ_images, last_frame], dim=0) - LQ_images = flashvsr_LQ_images[:num_frames+4].unsqueeze(0).movedim(-1, 1).to(device, dtype) * 2 - 1 + LQ_images = flashvsr_LQ_images[:num_frames+4].unsqueeze(0).movedim(-1, 1).to(dtype) * 2 - 1 if context_options is None: - flashvsr_LQ_latent = transformer.LQ_proj_in(LQ_images) + flashvsr_LQ_latent = transformer.LQ_proj_in(LQ_images.to(device)) log.info(f"flashvsr_LQ_latent: {flashvsr_LQ_latent[0].shape}") seq_len = math.ceil((noise.shape[2] * noise.shape[3]) / 4 * noise.shape[1]) @@ -1955,7 +1955,7 @@ class WanVideoSampler: end = c[-1] * 4 + 1 + 4 center_indices = torch.arange(start, end, 1) center_indices = torch.clamp(center_indices, min=0, max=LQ_images.shape[2] - 1) - partial_flashvsr_LQ_images = LQ_images[:, :, center_indices].to(device, dtype) + partial_flashvsr_LQ_images = LQ_images[:, :, center_indices].to(device) partial_flashvsr_LQ_latent = transformer.LQ_proj_in(partial_flashvsr_LQ_images) if len(timestep.shape) != 1: