perf: Optimize Ditto frame→tensor conversion to avoid memory doubling

This commit is contained in:
Hawk Lee
2026-01-22 18:13:00 +08:00
parent 731779be36
commit e6f97e233e
2 changed files with 21 additions and 3 deletions
+20 -2
View File
@@ -771,9 +771,27 @@ class AIIA_DittoSampler:
raise RuntimeError("Ditto generated 0 frames.")
# Convert List[np.array (H,W,C)] -> Batch Tensor (B,H,W,C)
# Note: generated frames are RGB (from writer_queue which usually gets RGB).
# Memory-optimized: pre-allocate tensor and release frames progressively
import torch
video_tensor = torch.from_numpy(np.array(generated)).float() / 255.0
import gc
num_frames = len(generated)
h, w, c = generated[0].shape
# Pre-allocate output tensor
video_tensor = torch.zeros((num_frames, h, w, c), dtype=torch.float32)
# Copy frames one by one and release original
for i in range(num_frames):
video_tensor[i] = torch.from_numpy(generated[i].astype(np.float32) / 255.0)
generated[i] = None # Release original frame
if i > 0 and i % 100 == 0:
gc.collect()
# Final cleanup
del generated
master_sdk.generated_frames = []
gc.collect()
return (video_tensor, audio)
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "aiia"
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
version = "1.9.22"
version = "1.9.23"
license = {file = "LICENSE"}
readme = "README.md"
authors = [