commit 73dd1a06d33953912f5dd684f168028b14e42a36 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 19:47:38 2025 +0300 cleanup commit 39bc2cecf493e2eb176b55e8841d933f0da1ec39 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 19:24:20 2025 +0300 Allow scheduling ovi cfg commit 2c153c5f324dbd59670ad9c51a7995459504a3cd Merge: dba766732eb6b4Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Oct 13 17:48:20 2025 +0300 Merge branch 'main' into ovi commit dba76674c71af7bf94c82834a0b0e40d94043c99 Merge: 0f11a435a0456eAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Sun Oct 12 22:45:43 2025 +0300 Merge branch 'main' into ovi commit 0f11a439622799ad8070f8a2b8cc8e6a041b761d Merge: 0999f50e2d8c9bAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Sat Oct 11 07:48:06 2025 +0300 Merge branch 'main' into ovi commit 0999f50cfe025290cd7ce88a8dd1acff0b38d9bd Merge: d45df1ff1d1c83Author: kijai <40791699+kijai@users.noreply.github.com> Date: Fri Oct 10 22:16:09 2025 +0300 Merge branch 'main' into ovi commit d45df1fb5b7c629b15eabc197357d62bdc232aaf Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 20:21:37 2025 +0300 Remove dependency for librosa commit d8e7533fdf7eab1d2489c3e025a908c02d997444 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 19:57:28 2025 +0300 Remove omegaconf dependency commit f4e27ff018e98cb5b09655dceda399baea36b240 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 19:31:06 2025 +0300 Fix VACE commit 35d3df39294831e5e7568b6f7e16d2ecf2d790a0 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Thu Oct 9 00:26:40 2025 +0300 small update commit 96f8ea1d26869ab7e49e12a07f19d5d5a2023253 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 22:32:57 2025 +0300 Create wanvideo_2_2_5B_ovi_testing.json commit a2511be73b9da7019fd21aeb0b521af941c09150 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 22:32:54 2025 +0300 Update nodes_sampler.py commit d3688b8db71452ea1f7c9a2bc0216441d524e56c Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 21:43:02 2025 +0300 Allow EasyCache to work with ovi commit 586d9148a0306ef5d30e9a971a9c3be4cd3ecc97 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 19:09:06 2025 +0300 Update model.py commit 61eedd2839decdb7d4c2ddd5f1310fdaf49d36ad Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 19:09:02 2025 +0300 I2V fix commit a97fcb1b9ae9fb7bbfdf668c24816e014a1b58d1 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 17:57:28 2025 +0300 Add nodes to set audio latent size commit d41e42a697f3d561dabbc22566f633b5f1bbd952 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 16:42:04 2025 +0300 Support loading mmaudio vae from .safetensors commit 1b0e28ec41e3c97fe1f2f057fef9b9bbcb87bca7 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 16:19:53 2025 +0300 Update nodes_sampler.py commit fbd18f45fe85ede8edcb5aebaea7ceb5b6eab5a2 Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 10:16:44 2025 +0300 Fixes for other workflows commit b06993b637198f7fad92208f3b3dc9a7d7f57c7f Author: kijai <40791699+kijai@users.noreply.github.com> Date: Wed Oct 8 09:46:27 2025 +0300 initial commit T2V works
163 lines
6.3 KiB
Python
163 lines
6.3 KiB
Python
# Copyright 2024-2025 The Alibaba Wan Team Authors. All rights reserved.
|
|
import math
|
|
import numpy as np
|
|
import torch
|
|
import torch.nn.functional as F
|
|
from transformers import Wav2Vec2ForCTC, Wav2Vec2Processor
|
|
|
|
|
|
def get_sample_indices(original_fps,
|
|
total_frames,
|
|
target_fps,
|
|
num_sample,
|
|
fixed_start=None):
|
|
required_duration = num_sample / target_fps
|
|
required_origin_frames = int(np.ceil(required_duration * original_fps))
|
|
if required_duration > total_frames / original_fps:
|
|
raise ValueError("required_duration must be less than video length")
|
|
|
|
if not fixed_start is None and fixed_start >= 0:
|
|
start_frame = fixed_start
|
|
else:
|
|
max_start = total_frames - required_origin_frames
|
|
if max_start < 0:
|
|
raise ValueError("video length is too short")
|
|
start_frame = np.random.randint(0, max_start + 1)
|
|
start_time = start_frame / original_fps
|
|
|
|
end_time = start_time + required_duration
|
|
time_points = np.linspace(start_time, end_time, num_sample, endpoint=False)
|
|
|
|
frame_indices = np.round(np.array(time_points) * original_fps).astype(int)
|
|
frame_indices = np.clip(frame_indices, 0, total_frames - 1)
|
|
return frame_indices
|
|
|
|
|
|
def linear_interpolation(features, input_fps, output_fps, output_len=None):
|
|
"""
|
|
features: shape=[1, T, 512]
|
|
input_fps: fps for audio, f_a
|
|
output_fps: fps for video, f_m
|
|
output_len: video length
|
|
"""
|
|
features = features.transpose(1, 2) # [1, 512, T]
|
|
seq_len = features.shape[2] / float(input_fps) # T/f_a
|
|
if output_len is None:
|
|
output_len = int(seq_len * output_fps) # f_m*T/f_a
|
|
output_features = F.interpolate(
|
|
features, size=output_len, align_corners=True,
|
|
mode='linear') # [1, 512, output_len]
|
|
return output_features.transpose(1, 2) # [1, output_len, 512]
|
|
|
|
|
|
class AudioEncoder():
|
|
|
|
def __init__(self, device='cpu', model_id="facebook/wav2vec2-base-960h"):
|
|
# load pretrained model
|
|
self.processor = Wav2Vec2Processor.from_pretrained(model_id)
|
|
self.model = Wav2Vec2ForCTC.from_pretrained(model_id)
|
|
|
|
self.model = self.model.to(device)
|
|
|
|
self.video_rate = 30
|
|
|
|
def get_audio_embed_bucket(self,
|
|
audio_embed,
|
|
stride=2,
|
|
batch_frames=12,
|
|
m=2):
|
|
num_layers, audio_frame_num, audio_dim = audio_embed.shape
|
|
|
|
if num_layers > 1:
|
|
return_all_layers = True
|
|
else:
|
|
return_all_layers = False
|
|
|
|
min_batch_num = int(audio_frame_num / (batch_frames * stride)) + 1
|
|
|
|
bucket_num = min_batch_num * batch_frames
|
|
batch_idx = [stride * i for i in range(bucket_num)]
|
|
batch_audio_eb = []
|
|
for bi in batch_idx:
|
|
if bi < audio_frame_num:
|
|
audio_sample_stride = 2
|
|
chosen_idx = list(
|
|
range(bi - m * audio_sample_stride,
|
|
bi + (m + 1) * audio_sample_stride,
|
|
audio_sample_stride))
|
|
chosen_idx = [0 if c < 0 else c for c in chosen_idx]
|
|
chosen_idx = [
|
|
audio_frame_num - 1 if c >= audio_frame_num else c
|
|
for c in chosen_idx
|
|
]
|
|
|
|
if return_all_layers:
|
|
frame_audio_embed = audio_embed[:, chosen_idx].flatten(
|
|
start_dim=-2, end_dim=-1)
|
|
else:
|
|
frame_audio_embed = audio_embed[0][chosen_idx].flatten()
|
|
else:
|
|
frame_audio_embed = \
|
|
torch.zeros([audio_dim * (2 * m + 1)], device=audio_embed.device) if not return_all_layers \
|
|
else torch.zeros([num_layers, audio_dim * (2 * m + 1)], device=audio_embed.device)
|
|
batch_audio_eb.append(frame_audio_embed)
|
|
batch_audio_eb = torch.cat([c.unsqueeze(0) for c in batch_audio_eb],
|
|
dim=0)
|
|
|
|
return batch_audio_eb, min_batch_num
|
|
|
|
def get_audio_embed_bucket_fps(self,
|
|
audio_embed,
|
|
fps=16,
|
|
batch_frames=81,
|
|
m=0):
|
|
num_layers, audio_frame_num, audio_dim = audio_embed.shape
|
|
|
|
if num_layers > 1:
|
|
return_all_layers = True
|
|
else:
|
|
return_all_layers = False
|
|
|
|
scale = self.video_rate / fps
|
|
|
|
min_batch_num = int(audio_frame_num / (batch_frames * scale)) + 1
|
|
|
|
bucket_num = min_batch_num * batch_frames
|
|
padd_audio_num = math.ceil(min_batch_num * batch_frames / fps *
|
|
self.video_rate) - audio_frame_num
|
|
batch_idx = get_sample_indices(
|
|
original_fps=self.video_rate,
|
|
total_frames=audio_frame_num + padd_audio_num,
|
|
target_fps=fps,
|
|
num_sample=bucket_num,
|
|
fixed_start=0)
|
|
batch_audio_eb = []
|
|
audio_sample_stride = int(self.video_rate / fps)
|
|
for bi in batch_idx:
|
|
if bi < audio_frame_num:
|
|
|
|
chosen_idx = list(
|
|
range(bi - m * audio_sample_stride,
|
|
bi + (m + 1) * audio_sample_stride,
|
|
audio_sample_stride))
|
|
chosen_idx = [0 if c < 0 else c for c in chosen_idx]
|
|
chosen_idx = [
|
|
audio_frame_num - 1 if c >= audio_frame_num else c
|
|
for c in chosen_idx
|
|
]
|
|
|
|
if return_all_layers:
|
|
frame_audio_embed = audio_embed[:, chosen_idx].flatten(
|
|
start_dim=-2, end_dim=-1)
|
|
else:
|
|
frame_audio_embed = audio_embed[0][chosen_idx].flatten()
|
|
else:
|
|
frame_audio_embed = \
|
|
torch.zeros([audio_dim * (2 * m + 1)], device=audio_embed.device) if not return_all_layers \
|
|
else torch.zeros([num_layers, audio_dim * (2 * m + 1)], device=audio_embed.device)
|
|
batch_audio_eb.append(frame_audio_embed)
|
|
batch_audio_eb = torch.cat([c.unsqueeze(0) for c in batch_audio_eb],
|
|
dim=0)
|
|
|
|
return batch_audio_eb, min_batch_num
|