commit 73dd1a06d33953912f5dd684f168028b14e42a36 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 19:47:38 2025 +0300 cleanup commit 39bc2cecf493e2eb176b55e8841d933f0da1ec39 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 19:24:20 2025 +0300 Allow scheduling ovi cfg commit 2c153c5f324dbd59670ad9c51a7995459504a3cd Merge: dba766732eb6b4Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 17:48:20 2025 +0300 Merge branch 'main' into ovi commit dba76674c71af7bf94c82834a0b0e40d94043c99 Merge: 0f11a435a0456eAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Sun Oct 12 22:45:43 2025 +0300 Merge branch 'main' into ovi commit 0f11a439622799ad8070f8a2b8cc8e6a041b761d Merge: 0999f50e2d8c9bAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Sat Oct 11 07:48:06 2025 +0300 Merge branch 'main' into ovi commit 0999f50cfe025290cd7ce88a8dd1acff0b38d9bd Merge: d45df1ff1d1c83Author: kijai <40791699+kijai@users.noreply.github.com> Date: Fri Oct 10 22:16:09 2025 +0300 Merge branch 'main' into ovi commit d45df1fb5b7c629b15eabc197357d62bdc232aaf Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 20:21:37 2025 +0300 Remove dependency for librosa commit d8e7533fdf7eab1d2489c3e025a908c02d997444 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 19:57:28 2025 +0300 Remove omegaconf dependency commit f4e27ff018e98cb5b09655dceda399baea36b240 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 19:31:06 2025 +0300 Fix VACE commit 35d3df39294831e5e7568b6f7e16d2ecf2d790a0 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 00:26:40 2025 +0300 small update commit 96f8ea1d26869ab7e49e12a07f19d5d5a2023253 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 22:32:57 2025 +0300 Create wanvideo_2_2_5B_ovi_testing.json commit a2511be73b9da7019fd21aeb0b521af941c09150 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 22:32:54 2025 +0300 Update nodes_sampler.py commit d3688b8db71452ea1f7c9a2bc0216441d524e56c Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 21:43:02 2025 +0300 Allow EasyCache to work with ovi commit 586d9148a0306ef5d30e9a971a9c3be4cd3ecc97 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 19:09:06 2025 +0300 Update model.py commit 61eedd2839decdb7d4c2ddd5f1310fdaf49d36ad Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 19:09:02 2025 +0300 I2V fix commit a97fcb1b9ae9fb7bbfdf668c24816e014a1b58d1 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 17:57:28 2025 +0300 Add nodes to set audio latent size commit d41e42a697f3d561dabbc22566f633b5f1bbd952 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 16:42:04 2025 +0300 Support loading mmaudio vae from .safetensors commit 1b0e28ec41e3c97fe1f2f057fef9b9bbcb87bca7 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 16:19:53 2025 +0300 Update nodes_sampler.py commit fbd18f45fe85ede8edcb5aebaea7ceb5b6eab5a2 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 10:16:44 2025 +0300 Fixes for other workflows commit b06993b637198f7fad92208f3b3dc9a7d7f57c7f Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 09:46:27 2025 +0300 initial commit T2V works
257 lines
8.4 KiB
Python
257 lines
8.4 KiB
Python
import torch
|
|
import torch.nn as nn
|
|
import folder_paths
|
|
import os
|
|
|
|
from .mel_converter import get_mel_converter
|
|
from .vae.autoencoder import AutoEncoderModule
|
|
from .vae.distributions import DiagonalGaussianDistribution
|
|
import torchaudio
|
|
|
|
from comfy import model_management as mm
|
|
device = mm.get_torch_device()
|
|
offload_device = mm.unet_offload_device()
|
|
|
|
class FeaturesUtils(nn.Module):
|
|
|
|
def __init__(
|
|
self,
|
|
*,
|
|
tod_vae_ckpt: str,
|
|
bigvgan_vocoder_ckpt = None,
|
|
mode=['16k', '44k'],
|
|
need_vae_encoder: bool = True,
|
|
):
|
|
super().__init__()
|
|
|
|
self.mel_converter = get_mel_converter(mode)
|
|
self.tod = AutoEncoderModule(vae_ckpt_path=tod_vae_ckpt,
|
|
vocoder_ckpt_path=bigvgan_vocoder_ckpt,
|
|
mode=mode,
|
|
need_vae_encoder=need_vae_encoder)
|
|
|
|
def encode_audio(self, x) -> DiagonalGaussianDistribution:
|
|
assert self.tod is not None, 'VAE is not loaded'
|
|
# x: (B * L)
|
|
mel = self.mel_converter(x)
|
|
dist = self.tod.encode(mel)
|
|
|
|
return dist
|
|
|
|
def vocode(self, mel: torch.Tensor) -> torch.Tensor:
|
|
assert self.tod is not None, 'VAE is not loaded'
|
|
return self.tod.vocode(mel)
|
|
|
|
def decode(self, z: torch.Tensor) -> torch.Tensor:
|
|
assert self.tod is not None, 'VAE is not loaded'
|
|
return self.tod.decode(z)
|
|
|
|
@property
|
|
def device(self):
|
|
return next(self.parameters()).device
|
|
|
|
@property
|
|
def dtype(self):
|
|
return next(self.parameters()).dtype
|
|
|
|
def wrapped_decode(self, z):
|
|
with torch.amp.autocast('cuda', dtype=self.dtype):
|
|
mel_decoded = self.decode(z)
|
|
audio = self.vocode(mel_decoded)
|
|
|
|
return audio
|
|
|
|
def wrapped_encode(self, audio):
|
|
with torch.amp.autocast('cuda', dtype=self.dtype):
|
|
dist = self.encode_audio(audio)
|
|
|
|
return dist.mean
|
|
|
|
if not "mmaudio" in folder_paths.folder_names_and_paths:
|
|
folder_paths.add_model_folder_path("mmaudio", os.path.join(folder_paths.models_dir, "mmaudio"))
|
|
|
|
class OviMMAudioVAELoader:
|
|
"""Loads MMAudio VAE for audio encoding/decoding in Ovi"""
|
|
@classmethod
|
|
def INPUT_TYPES(s):
|
|
s.vae_files = folder_paths.get_filename_list("vae")
|
|
s.mmaudio_files = folder_paths.get_filename_list("mmaudio")
|
|
s.all_files = s.vae_files + s.mmaudio_files
|
|
|
|
return {
|
|
"required": {
|
|
"vae": (s.all_files, {"tooltip": "MMAudio VAE 16k (v1-16.pth) model from models/vae or models/mmaudio"}),
|
|
"vocoder": (s.all_files, {"tooltip": "BigVGAN vocoder (best_netG.pt) from models/vae or models/mmaudio"}),
|
|
"precision": (["bf16", "fp16", "fp32"], {"default": "bf16"}),
|
|
}
|
|
}
|
|
|
|
RETURN_TYPES = ("MMAUDIOVAE",)
|
|
RETURN_NAMES = ("mmaudio_vae",)
|
|
FUNCTION = "loadmodel"
|
|
CATEGORY = "WanVideoWrapper/Ovi"
|
|
DESCRIPTION = "Loads MMAudio VAE for Ovi audio generation"
|
|
|
|
def loadmodel(self, vae, vocoder, precision):
|
|
dtype = {"bf16": torch.bfloat16, "fp16": torch.float16, "fp32": torch.float32}[precision]
|
|
|
|
vae_path = folder_paths.get_full_path("vae", vae) if vae in self.vae_files else folder_paths.get_full_path("mmaudio", vae)
|
|
vocoder_path = folder_paths.get_full_path("vae", vocoder) if vocoder in self.vae_files else folder_paths.get_full_path("mmaudio", vocoder)
|
|
|
|
vae = FeaturesUtils(
|
|
tod_vae_ckpt=vae_path,
|
|
bigvgan_vocoder_ckpt=vocoder_path,
|
|
mode='16k',
|
|
need_vae_encoder=True
|
|
)
|
|
|
|
vae.to(device=offload_device, dtype=dtype)
|
|
vae.eval()
|
|
|
|
return (vae,)
|
|
|
|
class WanVideoDecodeOviAudio:
|
|
@classmethod
|
|
def INPUT_TYPES(s):
|
|
return {"required": {
|
|
"mmaudio_vae": ("MMAUDIOVAE",),
|
|
"samples": ("LATENT",),
|
|
}
|
|
}
|
|
|
|
RETURN_TYPES = ("AUDIO",)
|
|
RETURN_NAMES = ("audio",)
|
|
FUNCTION = "decode"
|
|
CATEGORY = "WanVideoWrapper/Ovi"
|
|
|
|
def decode(self, mmaudio_vae, samples):
|
|
mm.soft_empty_cache()
|
|
audio_latents = samples.get("latent_ovi_audio", None)
|
|
if audio_latents is None:
|
|
raise ValueError("No Ovi audio latents found in input samples")
|
|
|
|
mmaudio_vae.to(device)
|
|
|
|
waveform = mmaudio_vae.wrapped_decode(audio_latents.to(device=device, dtype=mmaudio_vae.dtype))
|
|
audio = {"waveform": waveform.unsqueeze(0).cpu().float(), "sample_rate": 16000}
|
|
|
|
mmaudio_vae.to(offload_device)
|
|
mm.soft_empty_cache()
|
|
|
|
return (audio,)
|
|
|
|
class WanVideoEncodeOviAudio:
|
|
@classmethod
|
|
def INPUT_TYPES(s):
|
|
return {"required": {
|
|
"mmaudio_vae": ("MMAUDIOVAE",),
|
|
"audio": ("AUDIO",),
|
|
}
|
|
}
|
|
|
|
RETURN_TYPES = ("LATENT",)
|
|
RETURN_NAMES = ("samples",)
|
|
FUNCTION = "decode"
|
|
CATEGORY = "WanVideoWrapper/Ovi"
|
|
|
|
def decode(self, mmaudio_vae, audio):
|
|
|
|
mmaudio_vae.to(device)
|
|
|
|
waveform = audio.get("waveform", None)
|
|
sample_rate = audio.get("sample_rate", None)
|
|
if sample_rate != 16000:
|
|
waveform = torchaudio.functional.resample(waveform, sample_rate, 16000)
|
|
waveform = waveform.to(device=device, dtype=mmaudio_vae.dtype)[0][0].unsqueeze(0)
|
|
|
|
samples = mmaudio_vae.wrapped_encode(waveform)
|
|
|
|
mmaudio_vae.to(offload_device)
|
|
mm.soft_empty_cache()
|
|
|
|
return ({"latent_ovi_audio": samples},)
|
|
|
|
|
|
class WanVideoAddOviAudioToLatents:
|
|
@classmethod
|
|
def INPUT_TYPES(s):
|
|
return {"required": {
|
|
"original_samples": ("LATENT",),
|
|
"audio_samples": ("LATENT",),
|
|
}
|
|
}
|
|
|
|
RETURN_TYPES = ("LATENT",)
|
|
RETURN_NAMES = ("samples",)
|
|
FUNCTION = "decode"
|
|
CATEGORY = "WanVideoWrapper/Ovi"
|
|
|
|
def decode(self, original_samples, audio_samples):
|
|
samples = original_samples.copy()
|
|
samples.update(audio_samples)
|
|
|
|
return (samples,)
|
|
|
|
class WanVideoEmptyMMAudioLatents:
|
|
@classmethod
|
|
def INPUT_TYPES(s):
|
|
return {"required": {
|
|
"length": ("INT", {"default": 157, "min": 1, "max": 10000, "step": 1, "tooltip": "Length of the audio latent sequence"}),
|
|
}
|
|
}
|
|
|
|
RETURN_TYPES = ("LATENT",)
|
|
RETURN_NAMES = ("samples",)
|
|
FUNCTION = "decode"
|
|
CATEGORY = "WanVideoWrapper/Ovi"
|
|
|
|
def decode(self, length):
|
|
audio_latents = torch.zeros((length, 20), device=torch.device("cpu"), dtype=torch.float32) # 1, l c -> l, c
|
|
|
|
return ({"latent_ovi_audio": audio_latents},)
|
|
|
|
|
|
class WanVideoOviCFG:
|
|
@classmethod
|
|
def INPUT_TYPES(s):
|
|
return {"required": {
|
|
"original_text_embeds": ("WANVIDEOTEXTEMBEDS",),
|
|
"ovi_negative_text_embeds": ("WANVIDEOTEXTEMBEDS",),
|
|
"ovi_audio_cfg": ("FLOAT", {"default": 3.0, "min": 0.0, "max": 100.0, "step": 0.01}),
|
|
},
|
|
}
|
|
|
|
RETURN_TYPES = ("WANVIDEOTEXTEMBEDS", )
|
|
RETURN_NAMES = ("text_embeds",)
|
|
FUNCTION = "process"
|
|
CATEGORY = "WanVideoWrapper/Ovi"
|
|
DESCRIPTION = "Adds Ovi negative text embeddings and audio CFG scale to the text embeddings dictionary"
|
|
|
|
def process(self, original_text_embeds, ovi_negative_text_embeds, ovi_audio_cfg):
|
|
negative_text_embeds = ovi_negative_text_embeds.get("negative_prompt_embeds", None)
|
|
if negative_text_embeds is None:
|
|
negative_text_embeds = original_text_embeds["prompt_embeds"]
|
|
|
|
prompt_embeds_dict_copy = original_text_embeds.copy()
|
|
prompt_embeds_dict_copy.update({
|
|
"ovi_negative_prompt_embeds": negative_text_embeds,
|
|
"ovi_audio_cfg": ovi_audio_cfg,
|
|
})
|
|
return (prompt_embeds_dict_copy,)
|
|
|
|
NODE_CLASS_MAPPINGS = {
|
|
"OviMMAudioVAELoader": OviMMAudioVAELoader,
|
|
"WanVideoDecodeOviAudio": WanVideoDecodeOviAudio,
|
|
"WanVideoEncodeOviAudio": WanVideoEncodeOviAudio,
|
|
"WanVideoOviCFG": WanVideoOviCFG,
|
|
"WanVideoAddOviAudioToLatents": WanVideoAddOviAudioToLatents,
|
|
"WanVideoEmptyMMAudioLatents": WanVideoEmptyMMAudioLatents,
|
|
}
|
|
NODE_DISPLAY_NAME_MAPPINGS = {
|
|
"OviMMAudioVAELoader": "Ovi MMAudio VAE Loader",
|
|
"WanVideoDecodeOviAudio": "WanVideo Decode Ovi Audio",
|
|
"WanVideoEncodeOviAudio": "WanVideo Encode Ovi Audio",
|
|
"WanVideoOviCFG": "WanVideo Ovi CFG",
|
|
"WanVideoAddOviAudioToLatents": "WanVideo Add MMAudio To Latents",
|
|
"WanVideoEmptyMMAudioLatents": "WanVideo Empty MMAudio Latents",
|
|
} |