Files
kijai-ComfyUI-WanVideoWrapper/Ovi/nodes_ovi.py
T
kijai 139bdf827f Squashed commit of the following:
commit 73dd1a06d33953912f5dd684f168028b14e42a36
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Mon Oct 13 19:47:38 2025 +0300

    cleanup

commit 39bc2cecf493e2eb176b55e8841d933f0da1ec39
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Mon Oct 13 19:24:20 2025 +0300

    Allow scheduling ovi cfg

commit 2c153c5f324dbd59670ad9c51a7995459504a3cd
Merge: dba7667 32eb6b4
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Mon Oct 13 17:48:20 2025 +0300

    Merge branch 'main' into ovi

commit dba76674c71af7bf94c82834a0b0e40d94043c99
Merge: 0f11a43 5a0456e
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Sun Oct 12 22:45:43 2025 +0300

    Merge branch 'main' into ovi

commit 0f11a439622799ad8070f8a2b8cc8e6a041b761d
Merge: 0999f50 e2d8c9b
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Sat Oct 11 07:48:06 2025 +0300

    Merge branch 'main' into ovi

commit 0999f50cfe025290cd7ce88a8dd1acff0b38d9bd
Merge: d45df1f f1d1c83
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Fri Oct 10 22:16:09 2025 +0300

    Merge branch 'main' into ovi

commit d45df1fb5b7c629b15eabc197357d62bdc232aaf
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 20:21:37 2025 +0300

    Remove dependency for librosa

commit d8e7533fdf7eab1d2489c3e025a908c02d997444
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 19:57:28 2025 +0300

    Remove omegaconf dependency

commit f4e27ff018e98cb5b09655dceda399baea36b240
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 19:31:06 2025 +0300

    Fix VACE

commit 35d3df39294831e5e7568b6f7e16d2ecf2d790a0
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 00:26:40 2025 +0300

    small update

commit 96f8ea1d26869ab7e49e12a07f19d5d5a2023253
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 22:32:57 2025 +0300

    Create wanvideo_2_2_5B_ovi_testing.json

commit a2511be73b9da7019fd21aeb0b521af941c09150
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 22:32:54 2025 +0300

    Update nodes_sampler.py

commit d3688b8db71452ea1f7c9a2bc0216441d524e56c
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 21:43:02 2025 +0300

    Allow EasyCache to work with ovi

commit 586d9148a0306ef5d30e9a971a9c3be4cd3ecc97
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 19:09:06 2025 +0300

    Update model.py

commit 61eedd2839decdb7d4c2ddd5f1310fdaf49d36ad
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 19:09:02 2025 +0300

    I2V fix

commit a97fcb1b9ae9fb7bbfdf668c24816e014a1b58d1
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 17:57:28 2025 +0300

    Add nodes to set audio latent size

commit d41e42a697f3d561dabbc22566f633b5f1bbd952
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 16:42:04 2025 +0300

    Support loading mmaudio vae from .safetensors

commit 1b0e28ec41e3c97fe1f2f057fef9b9bbcb87bca7
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 16:19:53 2025 +0300

    Update nodes_sampler.py

commit fbd18f45fe85ede8edcb5aebaea7ceb5b6eab5a2
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 10:16:44 2025 +0300

    Fixes for other workflows

commit b06993b637198f7fad92208f3b3dc9a7d7f57c7f
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 09:46:27 2025 +0300

    initial commit

    T2V works
2025-10-13 20:16:53 +03:00

257 lines
8.4 KiB
Python

import torch
import torch.nn as nn
import folder_paths
import os
from .mel_converter import get_mel_converter
from .vae.autoencoder import AutoEncoderModule
from .vae.distributions import DiagonalGaussianDistribution
import torchaudio
from comfy import model_management as mm
device = mm.get_torch_device()
offload_device = mm.unet_offload_device()
class FeaturesUtils(nn.Module):
def __init__(
self,
*,
tod_vae_ckpt: str,
bigvgan_vocoder_ckpt = None,
mode=['16k', '44k'],
need_vae_encoder: bool = True,
):
super().__init__()
self.mel_converter = get_mel_converter(mode)
self.tod = AutoEncoderModule(vae_ckpt_path=tod_vae_ckpt,
vocoder_ckpt_path=bigvgan_vocoder_ckpt,
mode=mode,
need_vae_encoder=need_vae_encoder)
def encode_audio(self, x) -> DiagonalGaussianDistribution:
assert self.tod is not None, 'VAE is not loaded'
# x: (B * L)
mel = self.mel_converter(x)
dist = self.tod.encode(mel)
return dist
def vocode(self, mel: torch.Tensor) -> torch.Tensor:
assert self.tod is not None, 'VAE is not loaded'
return self.tod.vocode(mel)
def decode(self, z: torch.Tensor) -> torch.Tensor:
assert self.tod is not None, 'VAE is not loaded'
return self.tod.decode(z)
@property
def device(self):
return next(self.parameters()).device
@property
def dtype(self):
return next(self.parameters()).dtype
def wrapped_decode(self, z):
with torch.amp.autocast('cuda', dtype=self.dtype):
mel_decoded = self.decode(z)
audio = self.vocode(mel_decoded)
return audio
def wrapped_encode(self, audio):
with torch.amp.autocast('cuda', dtype=self.dtype):
dist = self.encode_audio(audio)
return dist.mean
if not "mmaudio" in folder_paths.folder_names_and_paths:
folder_paths.add_model_folder_path("mmaudio", os.path.join(folder_paths.models_dir, "mmaudio"))
class OviMMAudioVAELoader:
"""Loads MMAudio VAE for audio encoding/decoding in Ovi"""
@classmethod
def INPUT_TYPES(s):
s.vae_files = folder_paths.get_filename_list("vae")
s.mmaudio_files = folder_paths.get_filename_list("mmaudio")
s.all_files = s.vae_files + s.mmaudio_files
return {
"required": {
"vae": (s.all_files, {"tooltip": "MMAudio VAE 16k (v1-16.pth) model from models/vae or models/mmaudio"}),
"vocoder": (s.all_files, {"tooltip": "BigVGAN vocoder (best_netG.pt) from models/vae or models/mmaudio"}),
"precision": (["bf16", "fp16", "fp32"], {"default": "bf16"}),
}
}
RETURN_TYPES = ("MMAUDIOVAE",)
RETURN_NAMES = ("mmaudio_vae",)
FUNCTION = "loadmodel"
CATEGORY = "WanVideoWrapper/Ovi"
DESCRIPTION = "Loads MMAudio VAE for Ovi audio generation"
def loadmodel(self, vae, vocoder, precision):
dtype = {"bf16": torch.bfloat16, "fp16": torch.float16, "fp32": torch.float32}[precision]
vae_path = folder_paths.get_full_path("vae", vae) if vae in self.vae_files else folder_paths.get_full_path("mmaudio", vae)
vocoder_path = folder_paths.get_full_path("vae", vocoder) if vocoder in self.vae_files else folder_paths.get_full_path("mmaudio", vocoder)
vae = FeaturesUtils(
tod_vae_ckpt=vae_path,
bigvgan_vocoder_ckpt=vocoder_path,
mode='16k',
need_vae_encoder=True
)
vae.to(device=offload_device, dtype=dtype)
vae.eval()
return (vae,)
class WanVideoDecodeOviAudio:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"mmaudio_vae": ("MMAUDIOVAE",),
"samples": ("LATENT",),
}
}
RETURN_TYPES = ("AUDIO",)
RETURN_NAMES = ("audio",)
FUNCTION = "decode"
CATEGORY = "WanVideoWrapper/Ovi"
def decode(self, mmaudio_vae, samples):
mm.soft_empty_cache()
audio_latents = samples.get("latent_ovi_audio", None)
if audio_latents is None:
raise ValueError("No Ovi audio latents found in input samples")
mmaudio_vae.to(device)
waveform = mmaudio_vae.wrapped_decode(audio_latents.to(device=device, dtype=mmaudio_vae.dtype))
audio = {"waveform": waveform.unsqueeze(0).cpu().float(), "sample_rate": 16000}
mmaudio_vae.to(offload_device)
mm.soft_empty_cache()
return (audio,)
class WanVideoEncodeOviAudio:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"mmaudio_vae": ("MMAUDIOVAE",),
"audio": ("AUDIO",),
}
}
RETURN_TYPES = ("LATENT",)
RETURN_NAMES = ("samples",)
FUNCTION = "decode"
CATEGORY = "WanVideoWrapper/Ovi"
def decode(self, mmaudio_vae, audio):
mmaudio_vae.to(device)
waveform = audio.get("waveform", None)
sample_rate = audio.get("sample_rate", None)
if sample_rate != 16000:
waveform = torchaudio.functional.resample(waveform, sample_rate, 16000)
waveform = waveform.to(device=device, dtype=mmaudio_vae.dtype)[0][0].unsqueeze(0)
samples = mmaudio_vae.wrapped_encode(waveform)
mmaudio_vae.to(offload_device)
mm.soft_empty_cache()
return ({"latent_ovi_audio": samples},)
class WanVideoAddOviAudioToLatents:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"original_samples": ("LATENT",),
"audio_samples": ("LATENT",),
}
}
RETURN_TYPES = ("LATENT",)
RETURN_NAMES = ("samples",)
FUNCTION = "decode"
CATEGORY = "WanVideoWrapper/Ovi"
def decode(self, original_samples, audio_samples):
samples = original_samples.copy()
samples.update(audio_samples)
return (samples,)
class WanVideoEmptyMMAudioLatents:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"length": ("INT", {"default": 157, "min": 1, "max": 10000, "step": 1, "tooltip": "Length of the audio latent sequence"}),
}
}
RETURN_TYPES = ("LATENT",)
RETURN_NAMES = ("samples",)
FUNCTION = "decode"
CATEGORY = "WanVideoWrapper/Ovi"
def decode(self, length):
audio_latents = torch.zeros((length, 20), device=torch.device("cpu"), dtype=torch.float32) # 1, l c -> l, c
return ({"latent_ovi_audio": audio_latents},)
class WanVideoOviCFG:
@classmethod
def INPUT_TYPES(s):
return {"required": {
"original_text_embeds": ("WANVIDEOTEXTEMBEDS",),
"ovi_negative_text_embeds": ("WANVIDEOTEXTEMBEDS",),
"ovi_audio_cfg": ("FLOAT", {"default": 3.0, "min": 0.0, "max": 100.0, "step": 0.01}),
},
}
RETURN_TYPES = ("WANVIDEOTEXTEMBEDS", )
RETURN_NAMES = ("text_embeds",)
FUNCTION = "process"
CATEGORY = "WanVideoWrapper/Ovi"
DESCRIPTION = "Adds Ovi negative text embeddings and audio CFG scale to the text embeddings dictionary"
def process(self, original_text_embeds, ovi_negative_text_embeds, ovi_audio_cfg):
negative_text_embeds = ovi_negative_text_embeds.get("negative_prompt_embeds", None)
if negative_text_embeds is None:
negative_text_embeds = original_text_embeds["prompt_embeds"]
prompt_embeds_dict_copy = original_text_embeds.copy()
prompt_embeds_dict_copy.update({
"ovi_negative_prompt_embeds": negative_text_embeds,
"ovi_audio_cfg": ovi_audio_cfg,
})
return (prompt_embeds_dict_copy,)
NODE_CLASS_MAPPINGS = {
"OviMMAudioVAELoader": OviMMAudioVAELoader,
"WanVideoDecodeOviAudio": WanVideoDecodeOviAudio,
"WanVideoEncodeOviAudio": WanVideoEncodeOviAudio,
"WanVideoOviCFG": WanVideoOviCFG,
"WanVideoAddOviAudioToLatents": WanVideoAddOviAudioToLatents,
"WanVideoEmptyMMAudioLatents": WanVideoEmptyMMAudioLatents,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"OviMMAudioVAELoader": "Ovi MMAudio VAE Loader",
"WanVideoDecodeOviAudio": "WanVideo Decode Ovi Audio",
"WanVideoEncodeOviAudio": "WanVideo Encode Ovi Audio",
"WanVideoOviCFG": "WanVideo Ovi CFG",
"WanVideoAddOviAudioToLatents": "WanVideo Add MMAudio To Latents",
"WanVideoEmptyMMAudioLatents": "WanVideo Empty MMAudio Latents",
}