From 66527a866f5b99caa080ded4aa880f5a8924113c Mon Sep 17 00:00:00 2001 From: Hawk Lee Date: Tue, 20 Jan 2026 00:14:02 +0800 Subject: [PATCH] fix(test): correct clip encoder input shape --- tests/test_official_style.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_official_style.py b/tests/test_official_style.py index 311821f..a7d8724 100644 --- a/tests/test_official_style.py +++ b/tests/test_official_style.py @@ -145,8 +145,10 @@ def run_official_style_inference(): mask_latents = mask_latents.to(device, dtype=dtype) if mask_latents is not None else None # Explicitly handle clip image context + # WanImageEncoder expects [C, T, H, W]. For image, T=1. + # TF.to_tensor gives [C, H, W]. unsqueeze(1) gives [C, 1, H, W]. clip_image_pixel_values = TF.to_tensor(clip_image_pixel_values).sub_(0.5).div_(0.5).to(device, dtype=dtype) - clip_context = clip_image_encoder([clip_image_pixel_values.unsqueeze(0)]) + clip_context = clip_image_encoder([clip_image_pixel_values.unsqueeze(1)]) print("Starting generation loop...") video = pipeline(