Files
kijai-ComfyUI-WanVideoWrapper/wanvideo/modules/s2v/audio_encoder.py
T
kijai 139bdf827f Squashed commit of the following:
commit 73dd1a06d33953912f5dd684f168028b14e42a36
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Mon Oct 13 19:47:38 2025 +0300

    cleanup

commit 39bc2cecf493e2eb176b55e8841d933f0da1ec39
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Mon Oct 13 19:24:20 2025 +0300

    Allow scheduling ovi cfg

commit 2c153c5f324dbd59670ad9c51a7995459504a3cd
Merge: dba7667 32eb6b4
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Mon Oct 13 17:48:20 2025 +0300

    Merge branch 'main' into ovi

commit dba76674c71af7bf94c82834a0b0e40d94043c99
Merge: 0f11a43 5a0456e
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Sun Oct 12 22:45:43 2025 +0300

    Merge branch 'main' into ovi

commit 0f11a439622799ad8070f8a2b8cc8e6a041b761d
Merge: 0999f50 e2d8c9b
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Sat Oct 11 07:48:06 2025 +0300

    Merge branch 'main' into ovi

commit 0999f50cfe025290cd7ce88a8dd1acff0b38d9bd
Merge: d45df1f f1d1c83
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Fri Oct 10 22:16:09 2025 +0300

    Merge branch 'main' into ovi

commit d45df1fb5b7c629b15eabc197357d62bdc232aaf
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 20:21:37 2025 +0300

    Remove dependency for librosa

commit d8e7533fdf7eab1d2489c3e025a908c02d997444
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 19:57:28 2025 +0300

    Remove omegaconf dependency

commit f4e27ff018e98cb5b09655dceda399baea36b240
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 19:31:06 2025 +0300

    Fix VACE

commit 35d3df39294831e5e7568b6f7e16d2ecf2d790a0
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Thu Oct 9 00:26:40 2025 +0300

    small update

commit 96f8ea1d26869ab7e49e12a07f19d5d5a2023253
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 22:32:57 2025 +0300

    Create wanvideo_2_2_5B_ovi_testing.json

commit a2511be73b9da7019fd21aeb0b521af941c09150
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 22:32:54 2025 +0300

    Update nodes_sampler.py

commit d3688b8db71452ea1f7c9a2bc0216441d524e56c
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 21:43:02 2025 +0300

    Allow EasyCache to work with ovi

commit 586d9148a0306ef5d30e9a971a9c3be4cd3ecc97
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 19:09:06 2025 +0300

    Update model.py

commit 61eedd2839decdb7d4c2ddd5f1310fdaf49d36ad
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 19:09:02 2025 +0300

    I2V fix

commit a97fcb1b9ae9fb7bbfdf668c24816e014a1b58d1
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 17:57:28 2025 +0300

    Add nodes to set audio latent size

commit d41e42a697f3d561dabbc22566f633b5f1bbd952
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 16:42:04 2025 +0300

    Support loading mmaudio vae from .safetensors

commit 1b0e28ec41e3c97fe1f2f057fef9b9bbcb87bca7
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 16:19:53 2025 +0300

    Update nodes_sampler.py

commit fbd18f45fe85ede8edcb5aebaea7ceb5b6eab5a2
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 10:16:44 2025 +0300

    Fixes for other workflows

commit b06993b637198f7fad92208f3b3dc9a7d7f57c7f
Author: kijai <40791699+kijai@users.noreply.github.com>
Date:   Wed Oct 8 09:46:27 2025 +0300

    initial commit

    T2V works
2025-10-13 20:16:53 +03:00

163 lines
6.3 KiB
Python

# Copyright 2024-2025 The Alibaba Wan Team Authors. All rights reserved.
import math
import numpy as np
import torch
import torch.nn.functional as F
from transformers import Wav2Vec2ForCTC, Wav2Vec2Processor
def get_sample_indices(original_fps,
total_frames,
target_fps,
num_sample,
fixed_start=None):
required_duration = num_sample / target_fps
required_origin_frames = int(np.ceil(required_duration * original_fps))
if required_duration > total_frames / original_fps:
raise ValueError("required_duration must be less than video length")
if not fixed_start is None and fixed_start >= 0:
start_frame = fixed_start
else:
max_start = total_frames - required_origin_frames
if max_start < 0:
raise ValueError("video length is too short")
start_frame = np.random.randint(0, max_start + 1)
start_time = start_frame / original_fps
end_time = start_time + required_duration
time_points = np.linspace(start_time, end_time, num_sample, endpoint=False)
frame_indices = np.round(np.array(time_points) * original_fps).astype(int)
frame_indices = np.clip(frame_indices, 0, total_frames - 1)
return frame_indices
def linear_interpolation(features, input_fps, output_fps, output_len=None):
"""
features: shape=[1, T, 512]
input_fps: fps for audio, f_a
output_fps: fps for video, f_m
output_len: video length
"""
features = features.transpose(1, 2) # [1, 512, T]
seq_len = features.shape[2] / float(input_fps) # T/f_a
if output_len is None:
output_len = int(seq_len * output_fps) # f_m*T/f_a
output_features = F.interpolate(
features, size=output_len, align_corners=True,
mode='linear') # [1, 512, output_len]
return output_features.transpose(1, 2) # [1, output_len, 512]
class AudioEncoder():
def __init__(self, device='cpu', model_id="facebook/wav2vec2-base-960h"):
# load pretrained model
self.processor = Wav2Vec2Processor.from_pretrained(model_id)
self.model = Wav2Vec2ForCTC.from_pretrained(model_id)
self.model = self.model.to(device)
self.video_rate = 30
def get_audio_embed_bucket(self,
audio_embed,
stride=2,
batch_frames=12,
m=2):
num_layers, audio_frame_num, audio_dim = audio_embed.shape
if num_layers > 1:
return_all_layers = True
else:
return_all_layers = False
min_batch_num = int(audio_frame_num / (batch_frames * stride)) + 1
bucket_num = min_batch_num * batch_frames
batch_idx = [stride * i for i in range(bucket_num)]
batch_audio_eb = []
for bi in batch_idx:
if bi < audio_frame_num:
audio_sample_stride = 2
chosen_idx = list(
range(bi - m * audio_sample_stride,
bi + (m + 1) * audio_sample_stride,
audio_sample_stride))
chosen_idx = [0 if c < 0 else c for c in chosen_idx]
chosen_idx = [
audio_frame_num - 1 if c >= audio_frame_num else c
for c in chosen_idx
]
if return_all_layers:
frame_audio_embed = audio_embed[:, chosen_idx].flatten(
start_dim=-2, end_dim=-1)
else:
frame_audio_embed = audio_embed[0][chosen_idx].flatten()
else:
frame_audio_embed = \
torch.zeros([audio_dim * (2 * m + 1)], device=audio_embed.device) if not return_all_layers \
else torch.zeros([num_layers, audio_dim * (2 * m + 1)], device=audio_embed.device)
batch_audio_eb.append(frame_audio_embed)
batch_audio_eb = torch.cat([c.unsqueeze(0) for c in batch_audio_eb],
dim=0)
return batch_audio_eb, min_batch_num
def get_audio_embed_bucket_fps(self,
audio_embed,
fps=16,
batch_frames=81,
m=0):
num_layers, audio_frame_num, audio_dim = audio_embed.shape
if num_layers > 1:
return_all_layers = True
else:
return_all_layers = False
scale = self.video_rate / fps
min_batch_num = int(audio_frame_num / (batch_frames * scale)) + 1
bucket_num = min_batch_num * batch_frames
padd_audio_num = math.ceil(min_batch_num * batch_frames / fps *
self.video_rate) - audio_frame_num
batch_idx = get_sample_indices(
original_fps=self.video_rate,
total_frames=audio_frame_num + padd_audio_num,
target_fps=fps,
num_sample=bucket_num,
fixed_start=0)
batch_audio_eb = []
audio_sample_stride = int(self.video_rate / fps)
for bi in batch_idx:
if bi < audio_frame_num:
chosen_idx = list(
range(bi - m * audio_sample_stride,
bi + (m + 1) * audio_sample_stride,
audio_sample_stride))
chosen_idx = [0 if c < 0 else c for c in chosen_idx]
chosen_idx = [
audio_frame_num - 1 if c >= audio_frame_num else c
for c in chosen_idx
]
if return_all_layers:
frame_audio_embed = audio_embed[:, chosen_idx].flatten(
start_dim=-2, end_dim=-1)
else:
frame_audio_embed = audio_embed[0][chosen_idx].flatten()
else:
frame_audio_embed = \
torch.zeros([audio_dim * (2 * m + 1)], device=audio_embed.device) if not return_all_layers \
else torch.zeros([num_layers, audio_dim * (2 * m + 1)], device=audio_embed.device)
batch_audio_eb.append(frame_audio_embed)
batch_audio_eb = torch.cat([c.unsqueeze(0) for c in batch_audio_eb],
dim=0)
return batch_audio_eb, min_batch_num