⚡ Optimize video processing with batched tensor conversion
- Implements `process_batched_images` generator in `nodes/video_node.py` to process video frames in batches (default 20), significantly reducing GPU-CPU synchronization overhead. - Optimizes `tensor_to_numpy_uint8` in `shared/media/image_processing.py` to use in-place operations (`.clamp_()`), saving memory allocations for large tensors. - Reduces performance bottlenecks in video encoding pipelines.
This commit is contained in:
@@ -20,3 +20,16 @@
|
||||
- Converting a large PIL Image back to a Numpy array (`np.array(img)`) is surprisingly slow (measured ~500ms for 4K image).
|
||||
- When the PIL Image wraps an existing Numpy array and hasn't been modified (e.g. resized), reusing the original Numpy array avoids this overhead completely.
|
||||
**Action:** Track modifications to PIL images (like resizing) and reuse the source Numpy array for OpenCV operations if no modifications occurred.
|
||||
|
||||
## 2026-01-21 - Batched GPU-CPU Transfer for Video
|
||||
**Learning:**
|
||||
- Processing video frames individually (GPU tensor -> CPU numpy) incurs high synchronization and kernel launch overhead.
|
||||
- Batching frames (e.g., 20 frames/batch) during transfer amortizes this overhead, speeding up processing significantly.
|
||||
- However, batching too many frames at once can lead to OOM on the CPU side due to large intermediate float tensors.
|
||||
**Action:** Use a generator pattern with a safe batch size (e.g., 20) to process video tensors in chunks, balancing speed with memory usage.
|
||||
|
||||
## 2026-01-21 - In-place Tensor Operations
|
||||
**Learning:**
|
||||
- Chained tensor operations like `(tensor * 255).clamp(0, 255)` create new intermediate tensors at each step.
|
||||
- Using in-place operations like `.clamp_(0, 255)` on temporary results avoids allocating full-size tensors, saving memory and allocation time.
|
||||
**Action:** Prefer in-place operations (method names ending in `_`) when modifying temporary tensors that are not referenced elsewhere.
|
||||
|
||||
+33
-5
@@ -85,6 +85,34 @@ except ImportError:
|
||||
def update(self, advance=1):
|
||||
self.current += advance
|
||||
print(f"Progress: {self.current}/{self.total}", end="\r")
|
||||
|
||||
def process_batched_images(image_sequence, batch_size=20):
|
||||
"""
|
||||
Generator that processes images in batches to optimize GPU-CPU transfer.
|
||||
|
||||
Args:
|
||||
image_sequence: A torch.Tensor or list of tensors/images
|
||||
batch_size: Number of frames to process at once for Tensor inputs
|
||||
|
||||
Yields:
|
||||
Numpy array for each frame, contiguous and ready for ffmpeg
|
||||
"""
|
||||
# Optimized path for Tensor input
|
||||
if isinstance(image_sequence, torch.Tensor):
|
||||
total = len(image_sequence)
|
||||
for i in range(0, total, batch_size):
|
||||
# Process a chunk of frames on GPU/CPU together
|
||||
# This amortizes the overhead of kernel launches and synchronization
|
||||
batch = image_sequence[i:i+batch_size]
|
||||
batch_np = tensor_to_numpy_uint8(batch)
|
||||
for frame in batch_np:
|
||||
yield np.ascontiguousarray(frame)
|
||||
else:
|
||||
# Fallback for list input (e.g. pingpong or mixed sources)
|
||||
# We process individually as stacking might be expensive if they are not already contiguous tensors
|
||||
for img in image_sequence:
|
||||
yield np.ascontiguousarray(tensor_to_numpy_uint8(img))
|
||||
|
||||
class DiscordSendSaveVideo:
|
||||
"""
|
||||
A ComfyUI node that can send videos to Discord and save them with advanced options.
|
||||
@@ -566,9 +594,9 @@ class DiscordSendSaveVideo:
|
||||
env.update(video_format["environment"])
|
||||
|
||||
# Convert tensor images to bytes
|
||||
# Optimization: Use tensor_to_numpy_uint8 for faster conversion
|
||||
# Optimization: Use process_batched_images to optimize GPU-CPU transfer
|
||||
# Ensure contiguity to avoid ValueError in subprocess.stdin.write
|
||||
image_chunks = map(lambda x: np.ascontiguousarray(tensor_to_numpy_uint8(x)), image_sequence)
|
||||
image_chunks = process_batched_images(image_sequence)
|
||||
|
||||
# Base ffmpeg arguments
|
||||
args = [
|
||||
@@ -624,9 +652,9 @@ class DiscordSendSaveVideo:
|
||||
else:
|
||||
i_pix_fmt = 'rgb24'
|
||||
|
||||
# Optimization: Use tensor_to_numpy_uint8 for faster conversion
|
||||
# Optimization: Use process_batched_images to optimize GPU-CPU transfer
|
||||
# Ensure contiguity to avoid ValueError in subprocess.stdin.write
|
||||
image_chunks = map(lambda x: np.ascontiguousarray(tensor_to_numpy_uint8(x)), image_sequence)
|
||||
image_chunks = process_batched_images(image_sequence)
|
||||
|
||||
# Set up ffmpeg arguments based on format
|
||||
loop_args = []
|
||||
@@ -1088,4 +1116,4 @@ class DiscordSendSaveVideo:
|
||||
save_cdn_urls=False, github_cdn_update=False, github_repo="", github_token="",
|
||||
github_file_path="cdn_urls.md", prompt=None, extra_pnginfo=None, unique_id=None, **format_properties):
|
||||
# Always return True to ensure execution and proper handling of dynamic format properties
|
||||
return True
|
||||
return True
|
||||
@@ -20,4 +20,5 @@ def tensor_to_numpy_uint8(tensor: torch.Tensor) -> np.ndarray:
|
||||
"""
|
||||
# Optimization: Use torch operations for scaling/clipping/casting to avoid large float64 intermediate arrays on CPU
|
||||
# This is ~70% faster than naive numpy conversion: np.clip(255. * tensor.cpu().numpy(), 0, 255).astype(np.uint8)
|
||||
return (tensor * 255.0).clamp(0, 255).to(dtype=torch.uint8).cpu().numpy()
|
||||
# Further Optimization: Use clamp_ (in-place) to avoid allocating a second float tensor
|
||||
return (tensor * 255.0).clamp_(0, 255).to(dtype=torch.uint8).cpu().numpy()
|
||||
|
||||
Reference in New Issue
Block a user