Fix Fun 2.2 reference image input
Turns out the reference image doesn't work in the initial fp8_scaled models I shared due to the ref_conv layer ending up scaled as well, I've fixed the models and added error indicating the issue when trying to use reference image on such bugged model. No change in behaviour when using start image instead.
This commit is contained in:
@@ -1806,6 +1806,9 @@ class WanVideoSampler:
|
||||
image_cond = torch.cat([mask_latents, masked_video_latents_input], dim=0).to(device)
|
||||
clip_fea = None
|
||||
fun_ref_image = control_embeds.get("fun_ref_image", None)
|
||||
if fun_ref_image is not None:
|
||||
if transformer.ref_conv.weight.dtype in [torch.float8_e4m3fn, torch.float8_e5m2]:
|
||||
raise ValueError("Fun-Control reference image won't work with this specific fp8_scaled model, it's been fixed in latest version of the model")
|
||||
control_start_percent = control_embeds.get("start_percent", 0.0)
|
||||
control_end_percent = control_embeds.get("end_percent", 1.0)
|
||||
else:
|
||||
|
||||
@@ -1048,7 +1048,7 @@ class WanVideoModelLoader:
|
||||
dtype = torch.float8_e5m2
|
||||
else:
|
||||
dtype = base_dtype
|
||||
params_to_keep = {"norm", "bias", "time_in", "patch_embedding", "time_", "img_emb", "modulation", "text_embedding", "adapter", "add"}
|
||||
params_to_keep = {"norm", "bias", "time_in", "patch_embedding", "time_", "img_emb", "modulation", "text_embedding", "adapter", "add", "ref_conv"}
|
||||
if not lora_low_mem_load:
|
||||
log.info("Using accelerate to load and assign model weights to device...")
|
||||
param_count = sum(1 for _ in transformer.named_parameters())
|
||||
|
||||
Reference in New Issue
Block a user