diff --git a/aiia_ditto_nodes.py b/aiia_ditto_nodes.py index b6a6603..6cb9278 100644 --- a/aiia_ditto_nodes.py +++ b/aiia_ditto_nodes.py @@ -771,9 +771,27 @@ class AIIA_DittoSampler: raise RuntimeError("Ditto generated 0 frames.") # Convert List[np.array (H,W,C)] -> Batch Tensor (B,H,W,C) - # Note: generated frames are RGB (from writer_queue which usually gets RGB). + # Memory-optimized: pre-allocate tensor and release frames progressively import torch - video_tensor = torch.from_numpy(np.array(generated)).float() / 255.0 + import gc + + num_frames = len(generated) + h, w, c = generated[0].shape + + # Pre-allocate output tensor + video_tensor = torch.zeros((num_frames, h, w, c), dtype=torch.float32) + + # Copy frames one by one and release original + for i in range(num_frames): + video_tensor[i] = torch.from_numpy(generated[i].astype(np.float32) / 255.0) + generated[i] = None # Release original frame + if i > 0 and i % 100 == 0: + gc.collect() + + # Final cleanup + del generated + master_sdk.generated_frames = [] + gc.collect() return (video_tensor, audio) diff --git a/pyproject.toml b/pyproject.toml index beba489..b3500e8 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "aiia" description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser." -version = "1.9.22" +version = "1.9.23" license = {file = "LICENSE"} readme = "README.md" authors = [