diff --git a/LongCat_Video/layer_streaming.py b/LongCat_Video/layer_streaming.py index c201447..502de76 100644 --- a/LongCat_Video/layer_streaming.py +++ b/LongCat_Video/layer_streaming.py @@ -25,12 +25,173 @@ from __future__ import annotations import functools import itertools import logging -from typing import Any - +from typing import Any,TypeVar +import gc import torch from torch import nn +from contextlib import contextmanager +from collections.abc import Iterator logger = logging.getLogger(__name__) +_M = TypeVar("_M", bound=torch.nn.Module) +T = TypeVar("T") + + +def cleanup_memory() -> None: + gc.collect() + torch.cuda.empty_cache() + torch.cuda.synchronize() + +# LayerStreamingWrapper from https://github.com/Lightricks/LTX-2 + +class SimpleLayerStreamingWrapper_Dual(nn.Module): + """简化版层流式处理包装器,支持多模块卸载""" + + def __init__( + self, + model: nn.Module, + layers_attrs: list[str], # 修改为列表,支持多个模块路径 + target_device: torch.device, + active_count: int = 1, + ) -> None: + super().__init__() + self._model = model + self._layers_attrs = layers_attrs + self._target_device = target_device + self._active_count = active_count + + # 解析并存储所有需要卸载的模块 + self._layer_groups: list[nn.ModuleList] = [] + self._stores: list[_SimpleLayerStore] = [] + + for attr in self._layers_attrs: + layers = _resolve_attr(model, attr) + self._layer_groups.append(layers) + self._stores.append(_SimpleLayerStore(layers, self._target_device)) + + # 将非层参数移到GPU + self._move_non_layer_params_to_gpu() + + # 为所有模块组注册钩子 + self._register_simple_hooks() + + def _move_non_layer_params_to_gpu(self) -> None: + """移动非层参数到GPU,排除所有需要流式卸载的模块参数""" + layer_tensor_ids = set() + # 收集所有卸载模块的参数 ID + for layers in self._layer_groups: + for layer in layers: + for t in itertools.chain(layer.parameters(), layer.buffers()): + layer_tensor_ids.add(id(t)) + + for p in self._model.parameters(): + if id(p) not in layer_tensor_ids: + p.data = p.data.to(self._target_device) + for b in self._model.buffers(): + if id(b) not in layer_tensor_ids: + b.data = b.data.to(self._target_device) + def forward(self, *args: Any, **kwargs: Any) -> Any: + return self._model(*args, **kwargs) + + def __getattr__(self, name: str) -> Any: + """代理属性访问到原始模型""" + try: + # 首先尝试从包装器自身获取属性 + return super().__getattr__(name) + except AttributeError: + # 如果失败,则从原始模型获取 + return getattr(self._model, name) + + def _register_simple_hooks(self) -> None: + """为所有模块组注册简单的加载/释放钩子""" + # 遍历每一个模块组及其对应的 Store + for layers, store in zip(self._layer_groups, self._stores): + idx_map = {id(layer): idx for idx, layer in enumerate(layers)} + + def _pre_hook(module: nn.Module, input, *, idx: int, s: _SimpleLayerStore): + # 加载当前层到GPU + s.load_layer_to_gpu(idx, module) + # 记录流,防止内存被提前回收 + for param in itertools.chain(module.parameters(), module.buffers()): + param.data.record_stream(torch.cuda.current_stream(self._target_device)) + + def _post_hook(module: nn.Module, input, output, *, idx: int, s: _SimpleLayerStore): + # 处理完后立即将层移回CPU + s.unload_layer_from_gpu(idx, module) + + for layer in layers: + idx = idx_map[id(layer)] + # 使用 functools.partial 将对应的 store 实例传入钩子 + pre_hook = layer.register_forward_pre_hook(functools.partial(_pre_hook, idx=idx, s=store)) + post_hook = layer.register_forward_hook(functools.partial(_post_hook, idx=idx, s=store)) + +@contextmanager +def _streaming_model( + model: _M, + layers_attr, # 允许接收 str 或 list[str] + target_device: torch.device, + prefetch_count: int, +) -> Iterator[_M]: + """Wrap *model* with :class:`LayerStreamingWrapper`, yield it, then tear down.""" + # 根据传入的 layers_attr 类型自动路由到对应的 Wrapper + if isinstance(layers_attr, list): + wrapped = SimpleLayerStreamingWrapper_Dual( + model, + layers_attrs=layers_attr, + target_device=target_device, + active_count=prefetch_count, + ) + else: + wrapped = SimpleLayerStreamingWrapper( + model, + layers_attr=layers_attr, + target_device=target_device, + active_count=prefetch_count, + ) + + try: + yield wrapped # type: ignore[misc] + finally: + wrapped.to("cpu") + cleanup_memory() + torch.cuda.synchronize(device=target_device) + try: + if hasattr(torch._C, "_host_emptyCache"): + torch._C._host_emptyCache() + except Exception: + print("Host empty cache cleanup failed; ignoring.", exc_info=True) + + + +@contextmanager +def _streaming_model_( + model: _M, + layers_attr: str, + target_device: torch.device, + prefetch_count: int, +) -> Iterator[_M]: + """Wrap *model* with :class:`LayerStreamingWrapper`, yield it, then tear down.""" + wrapped = SimpleLayerStreamingWrapper( + model, + layers_attr=layers_attr, + target_device=target_device, + active_count=prefetch_count, + ) + try: + yield wrapped # type: ignore[misc] + finally: + wrapped.to("cpu") + cleanup_memory() + # Flush the host (pinned) memory cache so that freed pinned pages are + # returned to the OS. Without this, sequential streaming models + # (e.g. text encoder then transformer) exhaust host memory because the + # CachingHostAllocator keeps freed blocks cached indefinitely. + torch.cuda.synchronize(device=target_device) + try: + if hasattr(torch._C, "_host_emptyCache"): + torch._C._host_emptyCache() + except Exception: + print("Host empty cache cleanup failed; ignoring.", exc_info=True) def _resolve_attr(module: nn.Module, dotted_path: str) -> nn.ModuleList: diff --git a/LongCat_Video/longcat_video/modules/avatar/longcat_video_dit_avatar.py b/LongCat_Video/longcat_video/modules/avatar/longcat_video_dit_avatar.py index bc0050a..613c665 100644 --- a/LongCat_Video/longcat_video/modules/avatar/longcat_video_dit_avatar.py +++ b/LongCat_Video/longcat_video/modules/avatar/longcat_video_dit_avatar.py @@ -329,6 +329,39 @@ class LongCatVideoAvatarTransformer3DModel( module.forward = self._create_multi_lora_forward(module, loras) def _create_multi_lora_forward(self, module, loras): + def multi_lora_forward(x, *args, **kwargs): + weight_dtype = x.dtype + target_device = x.device + + # 执行原始的模块前向传播(不受LoRA设备影响) + org_output = module.org_forward(x, *args, **kwargs) + + total_lora_output = 0 + for lora in loras: + if lora.use_lora: + # 1. 推理前:将 LoRA 权重动态加载到输入张量所在的 CUDA 设备 + lora.lora_down.to(target_device, dtype=weight_dtype, non_blocking=True) + lora.lora_up.to(target_device, dtype=weight_dtype, non_blocking=True) + + # 2. 执行 LoRA 计算 + lx = lora.lora_down(x) + lx = lora.lora_up(lx) + lora_output = lx * lora.multiplier * lora.alpha_scale + + total_lora_output += lora_output + + # 3. 推理后:立即将 LoRA 权重卸载回 CPU,释放显存 + lora.lora_down.to("cpu", non_blocking=True) + lora.lora_up.to("cpu", non_blocking=True) + + # 累加 LoRA 输出并转换回原始数据类型 + total_lora_output = total_lora_output.to(weight_dtype) + + return org_output + total_lora_output + + return multi_lora_forward + + def _create_multi_lora_forward_(self, module, loras): def multi_lora_forward(x, *args, **kwargs): weight_dtype = x.dtype org_output = module.org_forward(x, *args, **kwargs) diff --git a/LongCat_Video/longcat_video/modules/quantization.py b/LongCat_Video/longcat_video/modules/quantization.py index 10ba375..205e7a2 100644 --- a/LongCat_Video/longcat_video/modules/quantization.py +++ b/LongCat_Video/longcat_video/modules/quantization.py @@ -6,7 +6,7 @@ from typing import Optional, Set import torch import torch.nn as nn import torch.nn.functional as F -from safetensors.torch import save_file, load_file +from safetensors.torch import save_file, load_file as safe_load_file class QuantizedLinear(nn.Module): @@ -233,17 +233,17 @@ def load_quantized_dit(checkpoint_dir: str, subfolder: str = "base_model_int8", state_dict = {} for shard_file in sorted(shard_files): shard_path = os.path.join(quantized_dir, shard_file) - shard_dict = load_file(shard_path, device="cpu") + shard_dict = safe_load_file(shard_path, device="cpu") state_dict.update(shard_dict) else: # Single file fallback if single_file: - state_dict = load_file(single_file, device="cpu") + state_dict = safe_load_file(single_file, device="cpu") else: files = [f for f in os.listdir(quantized_dir) if f.endswith(".safetensors") and "index" not in f] state_dict = {} for f in sorted(files): - shard_dict = load_file(os.path.join(quantized_dir, f), device="cpu") + shard_dict = safe_load_file(os.path.join(quantized_dir, f), device="cpu") state_dict.update(shard_dict) X=model.load_state_dict(state_dict, strict=True,assign=True) # Load weights and cast to bfloat16 for non-quantized params diff --git a/LongCat_Video/longcat_video/pipeline_longcat_video_avatar.py b/LongCat_Video/longcat_video/pipeline_longcat_video_avatar.py index bc7280e..9dc1917 100644 --- a/LongCat_Video/longcat_video/pipeline_longcat_video_avatar.py +++ b/LongCat_Video/longcat_video/pipeline_longcat_video_avatar.py @@ -19,7 +19,7 @@ from .modules.autoencoder_kl_wan import AutoencoderKLWan from .modules.avatar.longcat_video_dit_avatar import LongCatVideoAvatarTransformer3DModel #from .context_parallel import context_parallel_util from .utils.bukcet_config import get_bucket_config -from ..utils import _streaming_model +from ..layer_streaming import _streaming_model from contextlib import AbstractContextManager import ftfy import regex as re @@ -36,6 +36,67 @@ def torch_gc(): torch.cuda.empty_cache() torch.cuda.ipc_collect() +@torch.no_grad() +def get_audio_embedding_whisper_(audio_encoder,speech_array, fps=25, device='cpu', sample_rate=16000): + """使用 Whisper encoder 提取音频特征。 + Args: + speech_array: 原始音频波形 (numpy array, 单声道, sample_rate=16000) + fps: 目标视频帧率 + device: 推理设备 + sample_rate: 音频采样率 + Returns: + audio_emb: [T, 5, D],T = int(audio_duration * fps) + """ + def linear_interpolation_fps(features, input_fps, output_fps, output_len=None): + """将音频特征从 input_fps 插值到 output_fps。 + Args: + features: [B, T, D] + input_fps: 源帧率 + output_fps: 目标帧率 + output_len: 若指定则直接用该长度,否则按帧率比换算 + Returns: + [B, output_len, D] + """ + features = features.transpose(1, 2) # [B, D, T] + if output_len is None: + output_len = int(features.shape[2] / float(input_fps) * output_fps) + output_features = F.interpolate(features, size=output_len, align_corners=True, mode='linear') + return output_features.transpose(1, 2) + # ---- 常量 ---- + MEL_CHUNK = 750 * 640 # feature extractor 滑窗大小(样本数) + ENC_CHUNK = 3000 # encoder 滑窗大小(mel 帧数) + ENC_FPS = 50 # encoder 输出帧率 + + + audio_duration = len(speech_array) / sample_rate + video_length = int(audio_duration * fps) + def _loudness_norm(audio_array, sr=16000, lufs=-23, threshold=100): + meter = pyln.Meter(sr) + loudness = meter.integrated_loudness(audio_array) + if abs(loudness) > threshold: + return audio_array + normalized_audio = pyln.normalize.loudness(audio_array, loudness, lufs) + return normalized_audio + # ---- 音频预处理 ---- + speech_array = _loudness_norm(speech_array, sample_rate) + + + # comfyui encoder + audio_encoder_output = audio_encoder.encode_audio(torch.from_numpy(speech_array).unsqueeze(0).unsqueeze(0).to(device,audio_encoder.dtype), sample_rate) + audio_emb = torch.stack(audio_encoder_output["encoded_audio_all_layers"], dim=2) + audio_len = audio_encoder_output["audio_samples"] // 640 + + audio_prompts = audio_emb[:, :audio_len * 2] + + # ---- 按层分组 + 插值到目标帧数 ---- + feat0 = linear_interpolation_fps(audio_prompts[:, :, 0: 8].mean(dim=2), ENC_FPS, fps, video_length) + feat1 = linear_interpolation_fps(audio_prompts[:, :, 8:16].mean(dim=2), ENC_FPS, fps, video_length) + feat2 = linear_interpolation_fps(audio_prompts[:, :, 16:24].mean(dim=2), ENC_FPS, fps, video_length) + feat3 = linear_interpolation_fps(audio_prompts[:, :, 24:32].mean(dim=2), ENC_FPS, fps, video_length) + feat4 = linear_interpolation_fps(audio_prompts[:, :, 32], ENC_FPS, fps, video_length) + audio_emb = torch.stack([feat0, feat1, feat2, feat3, feat4], dim=2)[0] # [T, 5, D] + return audio_emb + @torch.no_grad() def get_audio_embedding_whisper(audio_encoder,audio_feature_extractor,speech_array, fps=25, device='cpu', sample_rate=16000): @@ -80,7 +141,7 @@ def get_audio_embedding_whisper(audio_encoder,audio_feature_extractor,speech_arr return normalized_audio # ---- 音频预处理 ---- speech_array = _loudness_norm(speech_array, sample_rate) - + # ---- Whisper feature extractor:wav → mel spectrogram ---- mel_chunks = [] for i in range(0, len(speech_array), MEL_CHUNK): @@ -101,8 +162,11 @@ def get_audio_embedding_whisper(audio_encoder,audio_feature_extractor,speech_arr output_hidden_states=True, ).hidden_states # tuple: (n_layers+1,) x [1, T_enc, D] enc_chunks.append(torch.stack(chunk_hs, dim=2)) # [1, T_enc, n_layers, D] + audio_prompts = torch.cat(enc_chunks, dim=1) # [1, T_enc_total, n_layers, D] + audio_prompts = audio_prompts[:, :video_length * 2] # 截取有效帧 + # ---- 按层分组 + 插值到目标帧数 ---- feat0 = linear_interpolation_fps(audio_prompts[:, :, 0: 8].mean(dim=2), ENC_FPS, fps, video_length) @@ -250,7 +314,7 @@ class LongCatVideoAvatarPipeline: if streaming_prefetch_count is not None: return _streaming_model( self.dit, - layers_attr="blocks", + layers_attr=["blocks"], target_device=torch.device("cuda"), prefetch_count=streaming_prefetch_count, ) @@ -1665,10 +1729,10 @@ class LongCatVideoAvatarPipeline: self.device = device if self.vae is not None: self.vae = self.vae.to(device, non_blocking=True) - if hasattr(self.dit, 'lora_dict') and self.dit.lora_dict: - for lora_key, lora_network in self.dit.lora_dict.items(): - for lora in lora_network.loras: - lora.to(device, non_blocking=True) + # if hasattr(self.dit, 'lora_dict') and self.dit.lora_dict: + # for lora_key, lora_network in self.dit.lora_dict.items(): + # for lora in lora_network.loras: + # lora.to(device, non_blocking=True) return self def to(self, device: str | torch.device): diff --git a/LongCat_Video/run_demo_avatar_multi_audio_to_video.py b/LongCat_Video/run_demo_avatar_multi_audio_to_video.py index dc155ec..1483a3c 100644 --- a/LongCat_Video/run_demo_avatar_multi_audio_to_video.py +++ b/LongCat_Video/run_demo_avatar_multi_audio_to_video.py @@ -426,7 +426,7 @@ def generate_multi(pipe,condition,te_cond,device,seed,cond_image,resolution, generator=generator, output_type='both', use_kv_cache=True, - offload_kv_cache=False, + offload_kv_cache=True, enhance_hf=True if not use_distill else False, audio_emb=audio_embs, ref_latent=ref_latent, diff --git a/LongCat_Video/run_demo_avatar_single_audio_to_video.py b/LongCat_Video/run_demo_avatar_single_audio_to_video.py index 3656f01..83575f8 100644 --- a/LongCat_Video/run_demo_avatar_single_audio_to_video.py +++ b/LongCat_Video/run_demo_avatar_single_audio_to_video.py @@ -15,7 +15,7 @@ import torch from transformers import AutoTokenizer, UMT5EncoderModel from diffusers.utils import load_image -from .longcat_video.pipeline_longcat_video_avatar import LongCatVideoAvatarPipeline,get_audio_embedding_whisper +from .longcat_video.pipeline_longcat_video_avatar import LongCatVideoAvatarPipeline,get_audio_embedding_whisper,get_audio_embedding_whisper_ from .longcat_video.modules.scheduling_flow_match_euler_discrete import FlowMatchEulerDiscreteScheduler from .longcat_video.modules.autoencoder_kl_wan import AutoencoderKLWan from .longcat_video.modules.avatar.longcat_video_dit_avatar import LongCatVideoAvatarTransformer3DModel @@ -47,7 +47,7 @@ def extract_vocal_from_speech(source_path, target_path, vocal_separator, audio_o print("Audio separate failed. Using raw audio.") return None - default_vocal_path = audio_output_dir_temp / "vocals" / outputs[0] + default_vocal_path = Path(os.path.join(audio_output_dir_temp ,"vocals", f"{outputs[0]}")) default_vocal_path = default_vocal_path.resolve().as_posix() # cmd = f"mv '{default_vocal_path}' '{target_path}'" # os.system(cmd) @@ -178,7 +178,7 @@ def prepare_audio(audio, sample_rate=16000): sr = 16000 return speech_array,sr -def get_audio_emb(checkpoint_dir,audio,left_audio,audio_type,save_fps,num_segments,device,p_box,model_type='avatar-v1.5' ): +def get_audio_emb(audio_encoder,audio,left_audio,audio_type,save_fps,num_segments,device,p_box,model_type='avatar-v1.5' ): num_frames=93 num_cond_frames = 13 audio_stride=1 @@ -198,23 +198,16 @@ def get_audio_emb(checkpoint_dir,audio,left_audio,audio_type,save_fps,num_segmen left_person_bbox=p_box[0] right_person_bbox=p_box[1] other_person_bbox = p_box[2] if len(p_box) > 2 else None - # left_person_bbox = p_box.get('person1', None) - # right_person_bbox = p_box.get('person2', None) - # other_person_bbox = p_box.get('others', None) use_background_silent_audio = other_person_bbox is not None and len(other_person_bbox) > 0 - # audio embedding - # initialize audio models - audio_model_checkpoint_path = os.path.join(checkpoint_dir, 'whisper-large-v3') - audio_encoder = get_audio_encoder(audio_model_checkpoint_path, model_type).to(device) - audio_feature_extractor = get_audio_feature_extractor(audio_model_checkpoint_path, model_type) + if left_speech_array is not None: left_speech_array_ext, right_speech_array_ext = audio_prepare_multi(left_speech_array,speech_array, generate_duration, sr=sr, audio_type=audio_type) - left_full_audio_emb = get_audio_embedding_whisper(audio_encoder, audio_feature_extractor, left_speech_array_ext, fps=save_fps*audio_stride, device="cuda" if torch.cuda.is_available() else "cpu", sample_rate=sr) - full_audio_emb = get_audio_embedding_whisper(audio_encoder, audio_feature_extractor, right_speech_array_ext, fps=save_fps*audio_stride, device="cuda" if torch.cuda.is_available() else "cpu", sample_rate=sr) + left_full_audio_emb = get_audio_embedding_whisper_(audio_encoder, left_speech_array_ext, fps=save_fps*audio_stride, ) + full_audio_emb = get_audio_embedding_whisper_(audio_encoder, right_speech_array_ext, fps=save_fps*audio_stride,) if torch.isnan(left_full_audio_emb).any() or torch.isnan(full_audio_emb).any(): raise ValueError(f"broken audio embedding with nan values") if use_background_silent_audio: - back_full_audio_emb = get_audio_embedding_whisper(audio_encoder, audio_feature_extractor,np.zeros_like(left_speech_array_ext), fps=save_fps*audio_stride, device="cuda" if torch.cuda.is_available() else "cpu", sample_rate=sr) + back_full_audio_emb = get_audio_embedding_whisper_(audio_encoder,np.zeros_like(left_speech_array_ext), fps=save_fps*audio_stride, ) assert left_full_audio_emb.shape == full_audio_emb.shape, f"Inconsistent audio embedding shape." if use_background_silent_audio: assert left_full_audio_emb.shape == back_full_audio_emb.shape, f"Inconsistent audio embedding shape between speaker and background." @@ -223,19 +216,10 @@ def get_audio_emb(checkpoint_dir,audio,left_audio,audio_type,save_fps,num_segmen added_sample_nums = math.ceil((generate_duration - source_duraion) * sr) if added_sample_nums > 0: speech_array = np.append(speech_array, [0.]*added_sample_nums) - full_audio_emb = get_audio_embedding_whisper(audio_encoder, audio_feature_extractor, speech_array, fps=save_fps*audio_stride, device="cuda" if torch.cuda.is_available() else "cpu", sample_rate=sr) - + full_audio_emb=get_audio_embedding_whisper_(audio_encoder, speech_array, fps=save_fps*audio_stride, ) #torch.Size([2142, 5, 1280]) if torch.isnan(full_audio_emb).any(): raise ValueError(f"broken audio embedding with nan values") - # # prepare audio embedding for the first clip - # indices = torch.arange(2 * 2 + 1) - 2 - # audio_start_idx = 0 - # audio_end_idx = audio_start_idx + audio_stride * num_frames - - # center_indices = torch.arange(audio_start_idx, audio_end_idx, audio_stride).unsqueeze(1) + indices.unsqueeze(0) - # center_indices = torch.clamp(center_indices, min=0, max=full_audio_emb.shape[0]-1) - # audio_emb = full_audio_emb[center_indices][None,...].to(device) au_cond={ "full_audio_emb": full_audio_emb, "num_segments": num_segments, @@ -250,8 +234,9 @@ def get_audio_emb(checkpoint_dir,audio,left_audio,audio_type,save_fps,num_segmen return au_cond -def get_audio_vocal(checkpoint_dir,raw_speech_path,left_raw_audio_path,audio_output_dir_temp,): - vocal_separator_path = os.path.join(checkpoint_dir, 'vocal_separator/Kim_Vocal_2.onnx') +def load_audio_vocal(vocal_separator_path,audio_output_dir_temp,checkpoint_dir): + if vocal_separator_path is None: + vocal_separator_path = os.path.join(checkpoint_dir, 'Kim_Vocal_2.onnx') os.makedirs(audio_output_dir_temp, exist_ok=True) audio_output_dir_temp = Path(audio_output_dir_temp) audio_separator_model_path = os.path.dirname(vocal_separator_path) @@ -264,33 +249,23 @@ def get_audio_vocal(checkpoint_dir,raw_speech_path,left_raw_audio_path,audio_out vocal_separator.load_model(audio_separator_model_name) vocal_separator.onnx_execution_provider = ["CUDAExecutionProvider"] + return vocal_separator + +def get_audio_vocal(vocal_separator,raw_speech_path,audio_output_dir_temp,): + vocal_path=replace_to_vocal_suffix(raw_speech_path) - left_audio_path=replace_to_vocal_suffix(left_raw_audio_path) if left_raw_audio_path is not None else None os.makedirs(os.path.dirname(vocal_path), exist_ok=True) - temp_vocal_path = extract_vocal_from_speech(raw_speech_path,vocal_path , vocal_separator, audio_output_dir_temp) - - temp_left_vocal_path,left_audio=None,None - - if left_audio_path is not None: - os.makedirs(os.path.dirname(temp_vocal_path), exist_ok=True) - temp_left_vocal_path = extract_vocal_from_speech(left_raw_audio_path,left_audio_path , vocal_separator, audio_output_dir_temp) - + temp_vocal_path = extract_vocal_from_speech(raw_speech_path,vocal_path , vocal_separator, audio_output_dir_temp) import librosa vocal_array, sr = librosa.load(temp_vocal_path, sr=16000) - if temp_left_vocal_path is not None: - left_vocal_array, sr = librosa.load(temp_left_vocal_path, sr=16000) - left_audio={ - "waveform": torch.from_numpy(left_vocal_array).unsqueeze(0).unsqueeze(0), - "sample_rate": sr, - } #print("vocal_array.shape", vocal_array.shape) audio={ "waveform": torch.from_numpy(vocal_array).unsqueeze(0).unsqueeze(0), "sample_rate": sr, } - return temp_vocal_path,audio,left_audio + return temp_vocal_path,audio def generate(pipe,condition,te_cond,device,seed,stage_1,cond_image,resolution, text_guidance_scale,audio_guidance_scale,num_inference_steps,ref_img_index,mask_frame_range, @@ -434,7 +409,7 @@ def generate(pipe,condition,te_cond,device,seed,stage_1,cond_image,resolution, generator=generator, output_type='both', use_kv_cache=True, - offload_kv_cache=False, + offload_kv_cache=True, enhance_hf=True if not use_distill else False, audio_emb=audio_emb, ref_latent=ref_latent, diff --git a/LongCat_Video/utils.py b/LongCat_Video/utils.py index b81b665..f6bef91 100644 --- a/LongCat_Video/utils.py +++ b/LongCat_Video/utils.py @@ -1,57 +1,12 @@ -from diffusers.quantizers.gguf.utils import dequantize_gguf_tensor -from contextlib import contextmanager -from .layer_streaming import SimpleLayerStreamingWrapper -from collections.abc import Iterator -from typing import TypeVar import gc import torch -# from utils import apply_loras_gguf - -_M = TypeVar("_M", bound=torch.nn.Module) -T = TypeVar("T") - - def cleanup_memory() -> None: gc.collect() torch.cuda.empty_cache() torch.cuda.synchronize() -# LayerStreamingWrapper from https://github.com/Lightricks/LTX-2 - -@contextmanager -def _streaming_model( - model: _M, - layers_attr: str, - target_device: torch.device, - prefetch_count: int, -) -> Iterator[_M]: - """Wrap *model* with :class:`LayerStreamingWrapper`, yield it, then tear down.""" - wrapped = SimpleLayerStreamingWrapper( - model, - layers_attr=layers_attr, - target_device=target_device, - active_count=prefetch_count, - ) - try: - yield wrapped # type: ignore[misc] - finally: - wrapped.to("cpu") - cleanup_memory() - # Flush the host (pinned) memory cache so that freed pinned pages are - # returned to the OS. Without this, sequential streaming models - # (e.g. text encoder then transformer) exhaust host memory because the - # CachingHostAllocator keeps freed blocks cached indefinitely. - torch.cuda.synchronize(device=target_device) - try: - if hasattr(torch._C, "_host_emptyCache"): - torch._C._host_emptyCache() - except Exception: - print("Host empty cache cleanup failed; ignoring.", exc_info=True) - - - def set_gguf2meta_model(meta_model,model_state_dict,dtype,device,lora_sd=None): from diffusers import GGUFQuantizationConfig from diffusers.quantizers.gguf import GGUFQuantizer @@ -153,6 +108,7 @@ def apply_loras_gguf( model_sd, lora_sd, ): + from diffusers.quantizers.gguf.utils import dequantize_gguf_tensor sd = {} for key, weight in model_sd.items(): if weight is None: diff --git a/LongCat_Video_node.py b/LongCat_Video_node.py index c85460a..1a67119 100644 --- a/LongCat_Video_node.py +++ b/LongCat_Video_node.py @@ -8,7 +8,7 @@ import os from comfy_api.latest import io import folder_paths from .node_utils import clear_comfyui_cache,tensor2image,audio2path -from .LongCat_Video.run_demo_avatar_single_audio_to_video import load_longcat_video_model,generate,get_audio_vocal,get_audio_emb +from .LongCat_Video.run_demo_avatar_single_audio_to_video import load_longcat_video_model,generate,get_audio_vocal,get_audio_emb,load_audio_vocal from .LongCat_Video.run_demo_avatar_multi_audio_to_video import generate_multi device = torch.device( "cuda:0") if torch.cuda.is_available() else torch.device( @@ -145,6 +145,7 @@ class LongCat_Video_SM_Audio(io.ComfyNode): display_name="LongCat_Video_SM_Audio", category="LongCat_Video", inputs=[ + io.AudioEncoder.Input("audio_encoder"), io.Audio.Input("audio"), io.Int.Input("save_fps", default=25, min=8, max=1024, step=1), io.Int.Input("num_segments", default=1, min=1, max=1024, step=1), @@ -157,7 +158,7 @@ class LongCat_Video_SM_Audio(io.ComfyNode): ], ) @classmethod - def execute(cls, audio,save_fps,num_segments,audio_type,p_box,left_audio=None) -> io.NodeOutput: + def execute(cls, audio_encoder,audio,save_fps,num_segments,audio_type,p_box,left_audio=None) -> io.NodeOutput: if p_box: import ast # 将类似 "[100, 80, 800, 640], [1001, 80, 800, 640]" 的字符串转为嵌套列表 @@ -165,7 +166,7 @@ class LongCat_Video_SM_Audio(io.ComfyNode): assert isinstance(parsed_p_box, list) and len(parsed_p_box) >= 2 , "p_box must be a list of int ,and must lens >2" else: parsed_p_box = None - au_cond=get_audio_emb(weigths_longcat_current_path,audio,left_audio,audio_type,save_fps,num_segments,device,p_box=parsed_p_box) + au_cond=get_audio_emb(audio_encoder,audio,left_audio,audio_type,save_fps,num_segments,device,p_box=parsed_p_box) clear_comfyui_cache() return io.NodeOutput(au_cond) @@ -177,17 +178,37 @@ class LongCat_Video_SM_Vocal(io.ComfyNode): display_name="LongCat_Video_SM_Vocal", category="LongCat_Video", inputs=[ + io.AudioEncoder.Input("audio_encoder"), io.Audio.Input("audio"), - io.Audio.Input("left_audio",optional=True), ], outputs=[ io.Audio.Output(display_name="audio"), - io.Audio.Output(display_name="left_audio"), io.String.Output(display_name="audio_path"), ], ) @classmethod - def execute(cls, audio, left_audio=None) -> io.NodeOutput: - left_audio_path=audio2path(left_audio) if left_audio is not None else None - audio_path,audio,left_audio=get_audio_vocal(weigths_longcat_current_path,audio2path(audio),left_audio_path,folder_paths.get_output_directory()) - return io.NodeOutput(audio,left_audio,audio_path) \ No newline at end of file + def execute(cls, audio_encoder,audio,) -> io.NodeOutput: + audio_path,audio=get_audio_vocal(audio_encoder,audio2path(audio),folder_paths.get_output_directory()) + return io.NodeOutput(audio,audio_path) + +class LongCat_Video_SM_VocalModel(io.ComfyNode): + @classmethod + def define_schema(cls): + return io.Schema( + node_id="LongCat_Video_SM_VocalModel", + display_name="LongCat_Video_SM_VocalModel", + category="LongCat_Video", + inputs=[ + io.Combo.Input( + "audio_encoder_vocal",options=["none"]+[i for i in folder_paths.get_filename_list("longcat") if i.endswith(".onnx")], + ), + ], + outputs=[ + io.AudioEncoder.Output(), + ], + ) + @classmethod + def execute(cls, audio_encoder_vocal) -> io.NodeOutput: + vocal_separator_path=folder_paths.get_full_path_or_raise("longcat", audio_encoder_vocal) if audio_encoder_vocal!="none" else None + audio_encoder=load_audio_vocal(vocal_separator_path,folder_paths.get_output_directory(),weigths_longcat_current_path) + return io.NodeOutput(audio_encoder) \ No newline at end of file diff --git a/README.md b/README.md index ce4e4ce..92ad512 100644 --- a/README.md +++ b/README.md @@ -3,7 +3,8 @@ Update ---- -* add node +* use comfyUI whisper-large-v3 audio_encoders,offload lora to cpu +* 音频解码改成comfyUI内置单体模型方式,开启lora和cache的卸载,降低显存占用 * 部署节点,目前单人和双人测试通过,部分参数未严谨测试,请自行测试,有bug可以提issues或反馈给我的B站或小红书smthem账号 1.Installation @@ -33,11 +34,11 @@ links: [vae/vocal_separator/whisper-large-v3/lora](https://huggingface.co/meitua | ├── LongCat_Avatar_1.5_vae.safetensors ├── ComfyUI/models/clip/ | ├── umt5_xxl_fp8_e4m3fn_scaled.safetensors -├── ComfyUI/models/longcat/ # 懒得写节点 -| ├── vocal_separator -| ├── 4 files -| ├── whisper-large-v3 -| ├── 13 files ,model.safetensors,config.json,tokenizer.json,vocab.json.. #模型只下载model.safetensors即可,json要下 +├── ComfyUI/models/audio_encoders/ +| ├── whisper-large-v3.safetensors # rename or not +├── ComfyUI/models/longcat/ +| ├── Kim_Vocal_2.onnx # 配套config文件会自动下,可以下了先放进去 + ``` 4 Example diff --git a/__init__.py b/__init__.py index c3b22e0..21777d4 100644 --- a/__init__.py +++ b/__init__.py @@ -1,7 +1,8 @@ from comfy_api.latest import ComfyExtension, io from typing_extensions import override -from .LongCat_Video_node import LongCat_Video_SM_Model, LongCat_Video_SM_Sampler,LongCat_Video_SM_Encode,LongCat_Video_SM_Audio,LongCat_Video_SM_Vocal +from .LongCat_Video_node import LongCat_Video_SM_Model, LongCat_Video_SM_Sampler,LongCat_Video_SM_Encode,LongCat_Video_SM_Audio,LongCat_Video_SM_Vocal,LongCat_Video_SM_VocalModel + class LongCat_Video_SM_Extension(ComfyExtension): @override async def get_node_list(self) -> list[type[io.ComfyNode]]: @@ -11,6 +12,7 @@ class LongCat_Video_SM_Extension(ComfyExtension): LongCat_Video_SM_Encode, LongCat_Video_SM_Audio, LongCat_Video_SM_Vocal, + LongCat_Video_SM_VocalModel ] async def comfy_entrypoint() -> LongCat_Video_SM_Extension: # ComfyUI calls this to load your extension and its nodes. diff --git a/example_workflows/example.png b/example_workflows/example.png index 5bf09b1..9080fba 100644 Binary files a/example_workflows/example.png and b/example_workflows/example.png differ diff --git a/example_workflows/longcat-avatar.json b/example_workflows/longcat-avatar.json index a538dc7..0a85052 100644 --- a/example_workflows/longcat-avatar.json +++ b/example_workflows/longcat-avatar.json @@ -1,22 +1,89 @@ { "id": "24406e80-db5d-4681-95f1-673239e92a0d", "revision": 0, - "last_node_id": 26, - "last_link_id": 32, + "last_node_id": 36, + "last_link_id": 49, "nodes": [ + { + "id": 1, + "type": "LongCat_Video_SM_Model", + "pos": [ + 650.396473073871, + 2991.1738612636673 + ], + "size": [ + 338.5, + 210.078125 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "MODEL", + "links": [ + 1 + ] + } + ], + "properties": { + "Node name for S&R": "LongCat_Video_SM_Model" + }, + "widgets_values": [ + "LongCat-Video-Avatar-1.5-int8.safetensors", + "none", + "LongCat-Video-Avatar-vae.safetensors", + "longcat-avatar-dmd_lora.safetensors" + ] + }, + { + "id": 8, + "type": "CLIPLoader", + "pos": [ + 271.60996506664526, + 3019.074909649015 + ], + "size": [ + 364.5, + 168.140625 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 7 + ] + } + ], + "properties": { + "Node name for S&R": "CLIPLoader" + }, + "widgets_values": [ + "umt5_xxl_fp8_e4m3fn_scaled.safetensors", + "wan", + "default" + ] + }, { "id": 7, "type": "LoadImage", "pos": [ - 1007.2523160561336, - 2976.8889023205434 + 999.5083234413873, + 3006.89688385741 ], "size": [ 270, - 339.453125 + 371.421875 ], "flags": {}, - "order": 0, + "order": 2, "mode": 0, "inputs": [], "outputs": [ @@ -37,23 +104,300 @@ "Node name for S&R": "LoadImage" }, "widgets_values": [ - "man.png", + "微信图片_20260528102159_4662_2556.png", "image" ] }, + { + "id": 27, + "type": "AudioEncoderLoader", + "pos": [ + -510.8617652405235, + 3034.4591464807313 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "AUDIO_ENCODER", + "type": "AUDIO_ENCODER", + "links": [ + 45 + ] + } + ], + "properties": { + "Node name for S&R": "AudioEncoderLoader" + }, + "widgets_values": [ + "whisper-large-v3.safetensors" + ] + }, + { + "id": 36, + "type": "LongCat_Video_SM_Vocal", + "pos": [ + -848.8587082062876, + 3080.4247507745868 + ], + "size": [ + 209.2125, + 46 + ], + "flags": {}, + "order": 13, + "mode": 2, + "inputs": [ + { + "name": "audio_encoder", + "type": "AUDIO_ENCODER", + "link": 47 + }, + { + "name": "audio", + "type": "AUDIO", + "link": 48 + } + ], + "outputs": [ + { + "name": "audio", + "type": "AUDIO", + "links": [ + 49 + ] + }, + { + "name": "audio_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "Node name for S&R": "LongCat_Video_SM_Vocal" + }, + "widgets_values": [] + }, + { + "id": 35, + "type": "LongCat_Video_SM_VocalModel", + "pos": [ + -1196.978482711904, + 2999.740934060109 + ], + "size": [ + 310.378515625, + 58 + ], + "flags": {}, + "order": 4, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "AUDIO_ENCODER", + "type": "AUDIO_ENCODER", + "links": [ + 47 + ] + } + ], + "properties": { + "Node name for S&R": "LongCat_Video_SM_VocalModel" + }, + "widgets_values": [ + "Kim_Vocal_2.onnx" + ] + }, + { + "id": 15, + "type": "LoadAudio", + "pos": [ + -1157.7806655520128, + 3142.772627510884 + ], + "size": [ + 270, + 151.453125 + ], + "flags": {}, + "order": 5, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 48 + ] + } + ], + "properties": { + "Node name for S&R": "LoadAudio" + }, + "widgets_values": [ + "sing_man.WAV", + null, + null + ] + }, + { + "id": 16, + "type": "SaveAudio", + "pos": [ + -830.5966729372642, + 3240.540564736228 + ], + "size": [ + 270, + 112 + ], + "flags": {}, + "order": 15, + "mode": 2, + "inputs": [ + { + "name": "audio", + "type": "AUDIO", + "link": 49 + } + ], + "outputs": [], + "properties": {}, + "widgets_values": [ + "audio/ComfyUI" + ] + }, + { + "id": 24, + "type": "Note", + "pos": [ + -325.3150444390377, + 2807.7094547168954 + ], + "size": [ + 497.1231074057333, + 94.45838764285236 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "note", + "properties": {}, + "widgets_values": [ + "1、连入左侧人物或者动物的声音,开启双人模式\n2、audio type 是声音合成的模式,还未测试\n3 、视频时长由num segments控制,为(93/25)时长的倍数\n4、p box为角色的坐标值,为[100, 80, 800, 640], [1001, 80, 800, 640] ,从左到右,如果有3组,第三组为需要忽略的背景人物(可以理解为mask)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 25, + "type": "Note", + "pos": [ + -1123.0124782852806, + 2815.3497691963007 + ], + "size": [ + 559.5, + 91.453125 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "note", + "properties": {}, + "widgets_values": [ + "如果没有背景音乐,可以跳过分离人声这步\n如果没有,会自动下载" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 5, + "type": "LongCat_Video_SM_Encode", + "pos": [ + 241.78386490317064, + 3271.1579161675004 + ], + "size": [ + 412.0967895507813, + 286.0468688964843 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 7 + } + ], + "outputs": [ + { + "name": "te_cond", + "type": "CONDITIONING", + "links": [ + 4 + ] + } + ], + "properties": { + "Node name for S&R": "LongCat_Video_SM_Encode" + }, + "widgets_values": [ + "Static camera, In a professional recording studio, two people stand facing each other, both wearing large headphones. They are speaking clearly into a large condenser microphone suspended between them. They looked at each other affectionately and occasionally shook their heads according to the rhythm. The soundproofed walls and visible recording equipment create an atmosphere focused on capturing high-quality audio as they interact and communicate.", + "Close-up, Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards." + ] + }, + { + "id": 26, + "type": "Note", + "pos": [ + 241.51565003191646, + 2801.2610872555492 + ], + "size": [ + 515.1837060069965, + 89.03824093286312 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "note", + "properties": {}, + "widgets_values": [ + "1 、gguf的DMD lora目前未支持(反正我也没上传模型)\n2 、stage1只是单角色时有效\n3 、block num 小于36G的显存请设置block num 大于0,看显存的使用量逐步增大\n" + ], + "color": "#322", + "bgcolor": "#533" + }, { "id": 3, "type": "CreateVideo", "pos": [ - 684.4278525325874, - 3733.0244437932192 + 339.14223554650306, + 3606.506829391974 ], "size": [ 270, 103.453125 ], "flags": {}, - "order": 22, + "order": 17, "mode": 0, "inputs": [ { @@ -85,155 +429,47 @@ ] }, { - "id": 9, - "type": "LoadAudio", + "id": 4, + "type": "SaveVideo", "pos": [ - -1143.3925469763471, - 2974.290181342222 + 993.8159507361488, + 3383.12032232679 ], "size": [ - 270, - 151.453125 - ], - "flags": {}, - "order": 1, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 9 - ] - } - ], - "properties": { - "Node name for S&R": "LoadAudio" - }, - "widgets_values": [ - "sing_woman.WAV", - null, - null - ] - }, - { - "id": 10, - "type": "LongCat_Video_SM_Vocal", - "pos": [ - -1147.5284848730753, - 3191.2103243475954 - ], - "size": [ - 225, - 119.453125 - ], - "flags": {}, - "order": 13, - "mode": 2, - "inputs": [ - { - "name": "audio", - "type": "AUDIO", - "link": 9 - }, - { - "name": "left_audio", - "shape": 7, - "type": "AUDIO", - "link": 17 - } - ], - "outputs": [ - { - "name": "audio", - "type": "AUDIO", - "links": [ - 10 - ] - }, - { - "name": "left_audio", - "type": "AUDIO", - "links": [ - 18 - ] - }, - { - "name": "audio_path", - "type": "STRING", - "links": null - } - ], - "properties": { - "Node name for S&R": "LongCat_Video_SM_Vocal" - }, - "widgets_values": [] - }, - { - "id": 11, - "type": "SaveAudio", - "pos": [ - -873.9326912494167, - 3166.8458185553573 - ], - "size": [ - 270, - 112 + 314.7684812377929, + 337.72981218566883 ], "flags": {}, "order": 18, - "mode": 2, + "mode": 0, "inputs": [ { - "name": "audio", - "type": "AUDIO", - "link": 10 + "name": "video", + "type": "VIDEO", + "link": 3 } ], "outputs": [], "properties": {}, "widgets_values": [ - "audio/ComfyUI" + "video/ComfyUI", + "auto", + "auto" ] }, - { - "id": 17, - "type": "Note", - "pos": [ - -505.54372981822735, - 2652.10346618984 - ], - "size": [ - 559.515625, - 202.671875 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [], - "outputs": [], - "title": "single", - "properties": {}, - "widgets_values": [ - "A western man stands on stage under dramatic lighting, holding a microphone close to their mouth. Wearing a vibrant red jacket with gold embroidery, the singer is speaking while smoke swirls around them, creating a dynamic and atmospheric scene." - ], - "color": "#432", - "bgcolor": "#653" - }, { "id": 18, "type": "Note", "pos": [ - -482.5826982631345, - 4259.351800597642 + 239.73268482421932, + 3814.0584339442544 ], "size": [ - 559.515625, - 202.671875 + 543.1021693615721, + 187.39828851074208 ], "flags": {}, - "order": 3, + "order": 9, "mode": 0, "inputs": [], "outputs": [], @@ -246,86 +482,51 @@ "bgcolor": "#653" }, { - "id": 15, - "type": "LoadAudio", + "id": 17, + "type": "Note", "pos": [ - -1279.308641046396, - 3436.7805917595424 + 812.110704844675, + 3813.7163438127627 ], "size": [ - 270, - 151.453125 + 511.4774306384279, + 188.5696042553709 ], "flags": {}, - "order": 4, - "mode": 2, + "order": 10, + "mode": 0, "inputs": [], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 17 - ] - } - ], - "properties": { - "Node name for S&R": "LoadAudio" - }, - "widgets_values": [ - "sing_man.WAV", - null, - null - ] - }, - { - "id": 16, - "type": "SaveAudio", - "pos": [ - -940.5087001284269, - 3467.756562218528 - ], - "size": [ - 270, - 112 - ], - "flags": {}, - "order": 19, - "mode": 2, - "inputs": [ - { - "name": "audio", - "type": "AUDIO", - "link": 18 - } - ], "outputs": [], + "title": "single", "properties": {}, "widgets_values": [ - "audio/ComfyUI" - ] + "A western man stands on stage under dramatic lighting, holding a microphone close to their mouth. Wearing a vibrant red jacket with gold embroidery, the singer is speaking while smoke swirls around them, creating a dynamic and atmospheric scene." + ], + "color": "#432", + "bgcolor": "#653" }, { - "id": 19, + "id": 23, "type": "LoadAudio", "pos": [ - -536.1948856589651, - 3839.271440023964 + -522.4541586730217, + 3187.337899997488 ], "size": [ 270, - 151.453125 + 179.421875 ], "flags": {}, - "order": 5, - "mode": 2, + "order": 11, + "mode": 0, "inputs": [], "outputs": [ { "name": "AUDIO", "type": "AUDIO", "links": [ - 22 + 30, + 32 ] } ], @@ -339,146 +540,44 @@ ] }, { - "id": 13, - "type": "LoadAudio", + "id": 22, + "type": "LongCat_Video_SM_Audio", "pos": [ - -520.1028477226126, - 4054.1885136612545 + -193.11301143584473, + 2998.64437812088 ], "size": [ - 270, - 151.453125 - ], - "flags": {}, - "order": 6, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 14 - ] - } - ], - "properties": { - "Node name for S&R": "LoadAudio" - }, - "widgets_values": [ - "ComfyUI_00004_.flac", - null, - null - ] - }, - { - "id": 14, - "type": "TrimAudioDuration", - "pos": [ - -200.96427362527413, - 4080.4337245283837 - ], - "size": [ - 270, - 111.453125 - ], - "flags": {}, - "order": 15, - "mode": 2, - "inputs": [ - { - "name": "audio", - "type": "AUDIO", - "link": 14 - } - ], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 28 - ] - } - ], - "properties": { - "Node name for S&R": "TrimAudioDuration" - }, - "widgets_values": [ - 5, - 10 - ] - }, - { - "id": 20, - "type": "TrimAudioDuration", - "pos": [ - -247.67058808799393, - 3847.7260747443647 - ], - "size": [ - 270, - 111.453125 + 375.54127001953134, + 188.58220343322773 ], "flags": {}, "order": 14, - "mode": 2, + "mode": 0, "inputs": [ + { + "name": "audio_encoder", + "type": "AUDIO_ENCODER", + "link": 45 + }, { "name": "audio", "type": "AUDIO", - "link": 22 - } - ], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 27 - ] - } - ], - "properties": { - "Node name for S&R": "TrimAudioDuration" - }, - "widgets_values": [ - 5, - 10 - ] - }, - { - "id": 21, - "type": "LongCat_Video_SM_Audio", - "pos": [ - -298.1968153684498, - 3495.2063811242024 - ], - "size": [ - 356.375, - 230.109375 - ], - "flags": {}, - "order": 20, - "mode": 2, - "inputs": [ - { - "name": "audio", - "type": "AUDIO", - "link": 27 + "link": 30 }, { "name": "left_audio", "shape": 7, "type": "AUDIO", - "link": 28 + "link": null } ], "outputs": [ { "name": "au_cond", "type": "CONDITIONING", - "links": [] + "links": [ + 31 + ] } ], "properties": { @@ -486,7 +585,7 @@ }, "widgets_values": [ 25, - 2, + 3, "para", "" ] @@ -495,15 +594,15 @@ "id": 2, "type": "LongCat_Video_SM_Sampler", "pos": [ - 688.3028505442599, - 3226.2618350990106 + 680.5588579295136, + 3256.2698166358773 ], "size": [ 288.5, 436.765625 ], "flags": {}, - "order": 21, + "order": 16, "mode": 0, "inputs": [ { @@ -551,292 +650,6 @@ 3, 1 ] - }, - { - "id": 1, - "type": "LongCat_Video_SM_Model", - "pos": [ - 658.1404656886173, - 2961.1658797268005 - ], - "size": [ - 338.5, - 210.078125 - ], - "flags": {}, - "order": 7, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "model", - "type": "MODEL", - "links": [ - 1 - ] - } - ], - "properties": { - "Node name for S&R": "LongCat_Video_SM_Model" - }, - "widgets_values": [ - "LongCat-Video-Avatar-1.5-int8.safetensors", - "none", - "LongCat-Video-Avatar-vae.safetensors", - "longcat-avatar-dmd_lora.safetensors" - ] - }, - { - "id": 22, - "type": "LongCat_Video_SM_Audio", - "pos": [ - -230.83786816844977, - 3055.1901322259805 - ], - "size": [ - 356.375, - 230.109375 - ], - "flags": {}, - "order": 16, - "mode": 0, - "inputs": [ - { - "name": "audio", - "type": "AUDIO", - "link": 30 - }, - { - "name": "left_audio", - "shape": 7, - "type": "AUDIO", - "link": null - } - ], - "outputs": [ - { - "name": "au_cond", - "type": "CONDITIONING", - "links": [ - 31 - ] - } - ], - "properties": { - "Node name for S&R": "LongCat_Video_SM_Audio" - }, - "widgets_values": [ - 25, - 2, - "para", - "" - ] - }, - { - "id": 25, - "type": "Note", - "pos": [ - -1187.8684265885026, - 3673.965833817273 - ], - "size": [ - 559.5, - 52 - ], - "flags": {}, - "order": 8, - "mode": 0, - "inputs": [], - "outputs": [], - "title": "note", - "properties": {}, - "widgets_values": [ - "如果没有背景音乐,可以跳过分离人声这步" - ], - "color": "#432", - "bgcolor": "#653" - }, - { - "id": 24, - "type": "Note", - "pos": [ - 102.58739661350121, - 4254.3023980824855 - ], - "size": [ - 301.8183213618165, - 88 - ], - "flags": {}, - "order": 9, - "mode": 4, - "inputs": [], - "outputs": [], - "title": "note", - "properties": {}, - "widgets_values": [ - "audio type 是声音合成的模式,还未测试" - ], - "color": "#432", - "bgcolor": "#653" - }, - { - "id": 23, - "type": "LoadAudio", - "pos": [ - -526.943128446215, - 3093.129623329746 - ], - "size": [ - 270, - 151.453125 - ], - "flags": {}, - "order": 10, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 30, - 32 - ] - } - ], - "properties": { - "Node name for S&R": "LoadAudio" - }, - "widgets_values": [ - "ComfyUI_00005_.flac", - null, - null - ] - }, - { - "id": 5, - "type": "LongCat_Video_SM_Encode", - "pos": [ - 249.52785751791677, - 3241.1499346306337 - ], - "size": [ - 419.34375, - 394.0929903362494 - ], - "flags": {}, - "order": 17, - "mode": 0, - "inputs": [ - { - "name": "clip", - "type": "CLIP", - "link": 7 - } - ], - "outputs": [ - { - "name": "te_cond", - "type": "CONDITIONING", - "links": [ - 4 - ] - } - ], - "properties": { - "Node name for S&R": "LongCat_Video_SM_Encode" - }, - "widgets_values": [ - "Static camera, In a professional recording studio, two people stand facing each other, both wearing large headphones. They are speaking clearly into a large condenser microphone suspended between them. They looked at each other affectionately and occasionally shook their heads according to the rhythm. The soundproofed walls and visible recording equipment create an atmosphere focused on capturing high-quality audio as they interact and communicate.", - "Close-up, Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards." - ] - }, - { - "id": 8, - "type": "CLIPLoader", - "pos": [ - 254.7139442536571, - 2986.426921398281 - ], - "size": [ - 364.55508626709, - 168.19122500000003 - ], - "flags": {}, - "order": 11, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "CLIP", - "type": "CLIP", - "links": [ - 7 - ] - } - ], - "properties": { - "Node name for S&R": "CLIPLoader" - }, - "widgets_values": [ - "umt5_xxl_fp8_e4m3fn_scaled.safetensors", - "wan", - "default" - ] - }, - { - "id": 4, - "type": "SaveVideo", - "pos": [ - 1001.5599433508951, - 3353.112340789923 - ], - "size": [ - 306.25, - 549.625 - ], - "flags": {}, - "order": 23, - "mode": 0, - "inputs": [ - { - "name": "video", - "type": "VIDEO", - "link": 3 - } - ], - "outputs": [], - "properties": {}, - "widgets_values": [ - "video/ComfyUI", - "auto", - "auto" - ] - }, - { - "id": 26, - "type": "Note", - "pos": [ - 234.5389908890977, - 2724.4358729710207 - ], - "size": [ - 345.2591069786379, - 152.4753072872163 - ], - "flags": {}, - "order": 12, - "mode": 0, - "inputs": [], - "outputs": [], - "title": "note", - "properties": {}, - "widgets_values": [ - "1 、gguf的DMD lora目前未支持(反正我也没上传模型)\n2 、stage1只是单角色时有效\n3 、视频时长由num segments控制,为(93/25)时长的倍数\n4 、block num 小于36G的显存请设置block num 大于0\n" - ], - "color": "#322", - "bgcolor": "#533" } ], "links": [ @@ -888,76 +701,12 @@ 0, "CLIP" ], - [ - 9, - 9, - 0, - 10, - 0, - "AUDIO" - ], - [ - 10, - 10, - 0, - 11, - 0, - "AUDIO" - ], - [ - 14, - 13, - 0, - 14, - 0, - "AUDIO" - ], - [ - 17, - 15, - 0, - 10, - 1, - "AUDIO" - ], - [ - 18, - 10, - 1, - 16, - 0, - "AUDIO" - ], - [ - 22, - 19, - 0, - 20, - 0, - "AUDIO" - ], - [ - 27, - 20, - 0, - 21, - 0, - "AUDIO" - ], - [ - 28, - 14, - 0, - 21, - 1, - "AUDIO" - ], [ 30, 23, 0, 22, - 0, + 1, "AUDIO" ], [ @@ -975,6 +724,38 @@ 3, 1, "AUDIO" + ], + [ + 45, + 27, + 0, + 22, + 0, + "AUDIO_ENCODER" + ], + [ + 47, + 35, + 0, + 36, + 0, + "AUDIO_ENCODER" + ], + [ + 48, + 15, + 0, + 36, + 1, + "AUDIO" + ], + [ + 49, + 36, + 0, + 16, + 0, + "AUDIO" ] ], "groups": [ @@ -982,10 +763,10 @@ "id": 1, "title": "daul or single", "bounding": [ - 240.9255308735557, - 2903.4858772091006, - 1097.8604703386677, - 1018.9557228256758 + 233.18153825880955, + 2933.4938587459674, + 1096.7956378435506, + 831.5509228256756 ], "color": "#3f789e", "flags": {} @@ -994,34 +775,22 @@ "id": 2, "title": "人声分离", "bounding": [ - -1286.6850870104474, - 2900.563579969892, - 705.979835475769, - 704.766680084228 - ], - "color": "#3f789e", - "flags": {} - }, - { - "id": 3, - "title": "dual 双人", - "bounding": [ - -542.9375931272351, - 3405.198149337393, - 708.3169850550048, - 818.3867383338779 + -1206.978482711904, + 2929.740934060109, + 656.381809774638, + 432.79963067611925 ], "color": "#3f789e", "flags": {} }, { "id": 4, - "title": "single 单人", + "title": "single /dual 单人/双人", "bounding": [ - -543.2920350802378, - 2899.0630689162635, - 709.7340392255003, - 456.98820783187557 + -532.9848890375133, + 2927.4080719481403, + 737.8556166565302, + 466.1775000310031 ], "color": "#3f789e", "flags": {} @@ -1030,10 +799,10 @@ "config": {}, "extra": { "ds": { - "scale": 0.5131581182307067, + "scale": 0.5644739300537773, "offset": [ - 1748.6647862887014, - -2445.5764084706775 + 1478.44028883581, + -2627.7750144615543 ] }, "frontendVersion": "1.43.18"