perf: Optimize Ditto frame→tensor conversion to avoid memory doubling
This commit is contained in:
+20
-2
@@ -771,9 +771,27 @@ class AIIA_DittoSampler:
|
||||
raise RuntimeError("Ditto generated 0 frames.")
|
||||
|
||||
# Convert List[np.array (H,W,C)] -> Batch Tensor (B,H,W,C)
|
||||
# Note: generated frames are RGB (from writer_queue which usually gets RGB).
|
||||
# Memory-optimized: pre-allocate tensor and release frames progressively
|
||||
import torch
|
||||
video_tensor = torch.from_numpy(np.array(generated)).float() / 255.0
|
||||
import gc
|
||||
|
||||
num_frames = len(generated)
|
||||
h, w, c = generated[0].shape
|
||||
|
||||
# Pre-allocate output tensor
|
||||
video_tensor = torch.zeros((num_frames, h, w, c), dtype=torch.float32)
|
||||
|
||||
# Copy frames one by one and release original
|
||||
for i in range(num_frames):
|
||||
video_tensor[i] = torch.from_numpy(generated[i].astype(np.float32) / 255.0)
|
||||
generated[i] = None # Release original frame
|
||||
if i > 0 and i % 100 == 0:
|
||||
gc.collect()
|
||||
|
||||
# Final cleanup
|
||||
del generated
|
||||
master_sdk.generated_frames = []
|
||||
gc.collect()
|
||||
|
||||
return (video_tensor, audio)
|
||||
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "aiia"
|
||||
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
|
||||
version = "1.9.22"
|
||||
version = "1.9.23"
|
||||
license = {file = "LICENSE"}
|
||||
readme = "README.md"
|
||||
authors = [
|
||||
|
||||
Reference in New Issue
Block a user