diff --git a/nodes_sampler.py b/nodes_sampler.py index c506bdd..61007a3 100644 --- a/nodes_sampler.py +++ b/nodes_sampler.py @@ -1218,11 +1218,11 @@ class WanVideoSampler: # [latent_frames, 1 mask frame, mocha_num_refs] latent_end = latent_frames mask_end = latent_end + 1 - partial_latents = mocha_embeds[:, :, context_window] # windowed latents - mask_frame = mocha_embeds[:, :, latent_end:mask_end] # single mask frame - ref_frames = mocha_embeds[:, :, -mocha_num_refs:] # reference frames + partial_latents = mocha_embeds[:, context_window] # windowed latents + mask_frame = mocha_embeds[:, latent_end:mask_end] # single mask frame + ref_frames = mocha_embeds[:, -mocha_num_refs:] # reference frames - partial_mocha_embeds = torch.cat([partial_latents, mask_frame, ref_frames], dim=2) + partial_mocha_embeds = torch.cat([partial_latents, mask_frame, ref_frames], dim=1) z = torch.cat([z, partial_mocha_embeds.to(z)], dim=1) else: z = torch.cat([z, mocha_embeds.to(z)], dim=1)