Squashed commit of the following:
commit 73dd1a06d33953912f5dd684f168028b14e42a36 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 19:47:38 2025 +0300 cleanup commit 39bc2cecf493e2eb176b55e8841d933f0da1ec39 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 19:24:20 2025 +0300 Allow scheduling ovi cfg commit 2c153c5f324dbd59670ad9c51a7995459504a3cd Merge: dba766732eb6b4Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 17:48:20 2025 +0300 Merge branch 'main' into ovi commit dba76674c71af7bf94c82834a0b0e40d94043c99 Merge: 0f11a435a0456eAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Sun Oct 12 22:45:43 2025 +0300 Merge branch 'main' into ovi commit 0f11a439622799ad8070f8a2b8cc8e6a041b761d Merge: 0999f50e2d8c9bAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Sat Oct 11 07:48:06 2025 +0300 Merge branch 'main' into ovi commit 0999f50cfe025290cd7ce88a8dd1acff0b38d9bd Merge: d45df1ff1d1c83Author: kijai <40791699+kijai@users.noreply.github.com> Date: Fri Oct 10 22:16:09 2025 +0300 Merge branch 'main' into ovi commit d45df1fb5b7c629b15eabc197357d62bdc232aaf Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 20:21:37 2025 +0300 Remove dependency for librosa commit d8e7533fdf7eab1d2489c3e025a908c02d997444 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 19:57:28 2025 +0300 Remove omegaconf dependency commit f4e27ff018e98cb5b09655dceda399baea36b240 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 19:31:06 2025 +0300 Fix VACE commit 35d3df39294831e5e7568b6f7e16d2ecf2d790a0 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 00:26:40 2025 +0300 small update commit 96f8ea1d26869ab7e49e12a07f19d5d5a2023253 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 22:32:57 2025 +0300 Create wanvideo_2_2_5B_ovi_testing.json commit a2511be73b9da7019fd21aeb0b521af941c09150 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 22:32:54 2025 +0300 Update nodes_sampler.py commit d3688b8db71452ea1f7c9a2bc0216441d524e56c Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 21:43:02 2025 +0300 Allow EasyCache to work with ovi commit 586d9148a0306ef5d30e9a971a9c3be4cd3ecc97 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 19:09:06 2025 +0300 Update model.py commit 61eedd2839decdb7d4c2ddd5f1310fdaf49d36ad Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 19:09:02 2025 +0300 I2V fix commit a97fcb1b9ae9fb7bbfdf668c24816e014a1b58d1 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 17:57:28 2025 +0300 Add nodes to set audio latent size commit d41e42a697f3d561dabbc22566f633b5f1bbd952 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 16:42:04 2025 +0300 Support loading mmaudio vae from .safetensors commit 1b0e28ec41e3c97fe1f2f057fef9b9bbcb87bca7 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 16:19:53 2025 +0300 Update nodes_sampler.py commit fbd18f45fe85ede8edcb5aebaea7ceb5b6eab5a2 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 10:16:44 2025 +0300 Fixes for other workflows commit b06993b637198f7fad92208f3b3dc9a7d7f57c7f Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 09:46:27 2025 +0300 initial commit T2V works
This commit is contained in:
+54
-22
@@ -892,8 +892,8 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None,
|
||||
cnt += 1
|
||||
if cnt % 100 == 0:
|
||||
pbar.update(100)
|
||||
#for name, param in transformer.named_parameters():
|
||||
# print(name, param.device, param.dtype)
|
||||
|
||||
#[print(name, param.device, param.dtype) for name, param in transformer.named_parameters()]
|
||||
|
||||
pbar.update_absolute(0)
|
||||
|
||||
@@ -1112,7 +1112,8 @@ class WanVideoModelLoader:
|
||||
if "vace_blocks.0.after_proj.weight" in sd and not "patch_embedding.weight" in sd:
|
||||
raise ValueError("You are attempting to load a VACE module as a WanVideo model, instead you should use the vace_model input and matching T2V base model")
|
||||
|
||||
# currently this can be VAE or MTV-Crafter weights
|
||||
# currently this can be VACE, MTV-Crafter, Lynx or Ovi-audio weights
|
||||
extra_audio_model = False
|
||||
if extra_model is not None:
|
||||
for _model in extra_model:
|
||||
print("Loading extra model: ", _model["path"])
|
||||
@@ -1126,32 +1127,38 @@ class WanVideoModelLoader:
|
||||
if _model["path"].endswith(".gguf"):
|
||||
raise ValueError("With GGUF extra model the main model must also be GGUF quantized model")
|
||||
extra_sd = load_torch_file(_model["path"], device=transformer_load_device, safe_load=True)
|
||||
if "audio_model.patch_embedding.0.weight" in extra_sd:
|
||||
extra_audio_model = True
|
||||
sd.update(extra_sd)
|
||||
del extra_sd
|
||||
|
||||
first_key = next(iter(sd))
|
||||
if first_key.startswith("audio_model.") and not extra_audio_model:
|
||||
sd = {key.replace("audio_model.", "", 1): value for key, value in sd.items()}
|
||||
if first_key.startswith("model.diffusion_model."):
|
||||
new_sd = {}
|
||||
for key, value in sd.items():
|
||||
new_key = key.replace("model.diffusion_model.", "", 1)
|
||||
new_sd[new_key] = value
|
||||
sd = new_sd
|
||||
sd = {key.replace("model.diffusion_model.", "", 1): value for key, value in sd.items()}
|
||||
elif first_key.startswith("model."):
|
||||
new_sd = {}
|
||||
for key, value in sd.items():
|
||||
new_key = key.replace("model.", "", 1)
|
||||
new_sd[new_key] = value
|
||||
sd = new_sd
|
||||
if not "patch_embedding.weight" in sd:
|
||||
raise ValueError("Invalid WanVideo model selected")
|
||||
dim = sd["patch_embedding.weight"].shape[0]
|
||||
sd = {key.replace("model.", "", 1): value for key, value in sd.items()}
|
||||
|
||||
if "patch_embedding.weight" in sd:
|
||||
dim = sd["patch_embedding.weight"].shape[0]
|
||||
in_channels = sd["patch_embedding.weight"].shape[1]
|
||||
elif "patch_embedding.0.weight" in sd:
|
||||
dim = sd["patch_embedding.0.weight"].shape[0]
|
||||
in_channels = sd["patch_embedding.0.weight"].shape[1]
|
||||
else:
|
||||
raise ValueError("No patch_embedding weight found, is the selected model a full WanVideo model?")
|
||||
|
||||
in_features = sd["blocks.0.self_attn.k.weight"].shape[1]
|
||||
out_features = sd["blocks.0.self_attn.k.weight"].shape[0]
|
||||
in_channels = sd["patch_embedding.weight"].shape[1]
|
||||
log.info(f"Detected model in_channels: {in_channels}")
|
||||
ffn_dim = sd["blocks.0.ffn.0.bias"].shape[0]
|
||||
ffn2_dim = sd["blocks.0.ffn.2.weight"].shape[1]
|
||||
|
||||
patch_size=(1, 2, 2)
|
||||
if "patch_embedding.0.weight" in sd:
|
||||
patch_size = [1]
|
||||
|
||||
is_humo = "audio_proj.audio_proj_glob_1.layer.weight" in sd
|
||||
is_wananimate = "pose_patch_embedding.weight" in sd
|
||||
|
||||
@@ -1273,6 +1280,7 @@ class WanVideoModelLoader:
|
||||
"dim": dim,
|
||||
"in_features": in_features,
|
||||
"out_features": out_features,
|
||||
"patch_size": patch_size,
|
||||
"ffn_dim": ffn_dim,
|
||||
"ffn2_dim": ffn2_dim,
|
||||
"eps": 1e-06,
|
||||
@@ -1309,8 +1317,32 @@ class WanVideoModelLoader:
|
||||
}
|
||||
|
||||
with init_empty_weights():
|
||||
transformer = WanModel(**TRANSFORMER_CONFIG)
|
||||
transformer.eval()
|
||||
transformer = WanModel(**TRANSFORMER_CONFIG).eval()
|
||||
|
||||
if extra_audio_model:
|
||||
log.info("Ovi extra audio model detected, initializing...")
|
||||
TRANSFORMER_CONFIG.update({
|
||||
"patch_size": [1],
|
||||
"in_dim": 20,
|
||||
"out_dim": 20,
|
||||
})
|
||||
|
||||
with init_empty_weights():
|
||||
transformer.audio_model = WanModel(**TRANSFORMER_CONFIG).eval()
|
||||
|
||||
from .wanvideo.modules.model import WanLayerNorm, WanRMSNorm
|
||||
|
||||
for block in transformer.blocks:
|
||||
block.cross_attn.k_fusion = nn.Linear(block.dim, block.dim)
|
||||
block.cross_attn.v_fusion = nn.Linear(block.dim, block.dim)
|
||||
block.cross_attn.pre_attn_norm_fusion = WanLayerNorm(block.dim, elementwise_affine=True)
|
||||
block.cross_attn.norm_k_fusion = WanRMSNorm(block.dim, eps=1e-6) if block.qk_norm else nn.Identity()
|
||||
|
||||
for block in transformer.audio_model.blocks:
|
||||
block.cross_attn.k_fusion = nn.Linear(block.dim, block.dim)
|
||||
block.cross_attn.v_fusion = nn.Linear(block.dim, block.dim)
|
||||
block.cross_attn.pre_attn_norm_fusion = WanLayerNorm(block.dim, elementwise_affine=True)
|
||||
block.cross_attn.norm_k_fusion = WanRMSNorm(block.dim, eps=1e-6) if block.qk_norm else nn.Identity()
|
||||
|
||||
#ReCamMaster
|
||||
if "blocks.0.cam_encoder.weight" in sd:
|
||||
@@ -1422,10 +1454,10 @@ class WanVideoModelLoader:
|
||||
for k, v in sd.items():
|
||||
if k.endswith(".scale_weight"):
|
||||
scale_weights[k] = v.to(device, base_dtype)
|
||||
|
||||
if "fp8_e4m3fn" in quantization:
|
||||
|
||||
if quantization == "fp8_e4m3fn":
|
||||
weight_dtype = torch.float8_e4m3fn
|
||||
elif "fp8_e5m2" in quantization:
|
||||
elif quantization == "fp8_e5m2":
|
||||
weight_dtype = torch.float8_e5m2
|
||||
else:
|
||||
weight_dtype = base_dtype
|
||||
|
||||
Reference in New Issue
Block a user