refactor(phase3): extract video encoding utilities to shared modules

- Create shared/media/format_utils.py with format detection and validation
- Create shared/media/video_encoder.py with FFmpegEncoder and PILEncoder classes
- Extract validate_video_for_discord to shared utility
- Extract ffmpeg detection to shared detect_ffmpeg() function
- Extract Discord video optimization to shared optimize_video_for_discord()
- Remove duplicate code from video_node.py (-109 lines)

Phase 3 summary:
- video_node.py: 1201 -> 1092 lines (9% reduction this phase)
- Total reduction from original: 1562 -> 1092 lines (30% reduction)

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
This commit is contained in:
AEmotionStudio
2026-01-19 23:38:58 -08:00
co-authored by Claude Opus 4.5
parent a208cd482b
commit 76da850b3d
4 changed files with 692 additions and 125 deletions
+16 -125
View File
@@ -37,6 +37,12 @@ from shared import (
send_cdn_urls_file,
extract_prompts_from_workflow
)
from shared.media import (
validate_video_for_discord,
normalize_video_extension,
optimize_video_for_discord as shared_optimize_video,
detect_ffmpeg
)
# Define cached decorator for local use
def cached(max_size=None):
"""
@@ -59,14 +65,16 @@ def cached(max_size=None):
return decorator(func)
return decorator
# Define constants and variables previously imported from discordsend_utils
ffmpeg_path = None
# Define constants
ENCODE_ARGS = ("utf-8", "ignore")
floatOrInt = ("FLOAT", "INT")
imageOrLatent = ("IMAGE", "LATENT")
BIGMAX = 1000000
has_vhs_formats = False
# Detect ffmpeg using shared utility
ffmpeg_path = detect_ffmpeg()
# Try to import ProgressBar from comfy.utils
try:
from comfy.utils import ProgressBar
@@ -75,67 +83,10 @@ except ImportError:
def __init__(self, total):
self.total = total
self.current = 0
def update(self, advance=1):
self.current += advance
print(f"Progress: {self.current}/{self.total}", end="\r")
# Fallback if ffmpeg_path is None, try to detect it
if ffmpeg_path is None:
# Try direct detection methods for ffmpeg
try:
# Try imageio-ffmpeg first (common in Python environments)
try:
import imageio_ffmpeg
ffmpeg_path = imageio_ffmpeg.get_ffmpeg_exe()
print(f"Found ffmpeg via imageio_ffmpeg: {ffmpeg_path}")
except (ImportError, Exception):
# Fall back to checking the system path
from shutil import which
ffmpeg_path = which("ffmpeg")
if ffmpeg_path:
print(f"Found ffmpeg in system path: {ffmpeg_path}")
except Exception as e:
print(f"Error during ffmpeg detection: {str(e)}")
# Function to validate video files for Discord compatibility
def validate_video_for_discord(file_path):
"""
Validate that a video file is compatible with Discord.
Returns a tuple of (is_valid, message)
"""
if not os.path.exists(file_path):
return False, f"File does not exist: {file_path}"
# Check if file is empty or too small
file_size = os.path.getsize(file_path)
if file_size == 0:
return False, "File is empty"
if file_size < 1024: # Less than 1KB
return False, f"File is suspiciously small: {file_size} bytes"
# Check if file is too large for Discord (Discord limit is 25MB for regular users, 50MB for Nitro)
max_size = 25 * 1024 * 1024 # 25MB in bytes
if file_size > max_size:
return False, f"File exceeds Discord's size limit: {file_size} bytes (max {max_size} bytes)"
# Get file extension
ext = os.path.splitext(file_path)[1].lower().lstrip('.')
# Return validation result based on file type
if ext in ['mp4', 'webm', 'gif']:
# These formats are well supported by Discord
return True, "Valid video format for Discord"
elif ext in ['mov']:
# MOV files (ProRes) may need conversion for Discord
return False, "MOV files may need conversion for Discord compatibility"
elif ext in ['png']:
# PNG sequences are not directly supported by Discord
return False, "PNG sequences are not directly supported by Discord and require compilation into a video format"
else:
# For any other extension, warn but allow sending
return False, f"Unknown format {ext}, may not be compatible with Discord"
class DiscordSendSaveVideo:
"""
A ComfyUI node that can send videos to Discord and save them with advanced options.
@@ -857,73 +808,13 @@ class DiscordSendSaveVideo:
discord_optimized_file = None
try:
# For Discord compatibility, create a special Discord-optimized copy of the video
# This is particularly important when add_time is disabled
input_file = output_files[-1] # Get input file (the last output file)
temp_dir = folder_paths.get_temp_directory()
try:
# Create optimized output file in a temp location
temp_dir = folder_paths.get_temp_directory()
discord_optimized_file = os.path.join(temp_dir, f"discord_optimized_{uuid4()}{os.path.splitext(input_file)[1]}")
# Set up ffmpeg arguments for optimized Discord conversion
# This creates a new file specifically optimized for Discord playback
format_ext = os.path.splitext(input_file)[1].lstrip('.').lower()
optimize_args = []
if format_ext == "mp4":
# MP4 optimization for Discord
optimize_args = [
ffmpeg_path, "-i", input_file,
"-c:v", "libx264", "-pix_fmt", "yuv420p",
"-movflags", "faststart", "-preset", "fast",
"-profile:v", "baseline", "-level", "3.0",
"-crf", "23"
]
# Add audio if present in original file
optimize_args.extend(["-c:a", "aac", "-b:a", "128k"])
# Add output file
optimize_args.append(discord_optimized_file)
elif format_ext == "webm":
# WebM optimization for Discord
optimize_args = [
ffmpeg_path, "-i", input_file,
"-c:v", "libvpx-vp9",
"-pix_fmt", "yuv420p",
"-crf", "30", "-b:v", "0",
"-deadline", "good"
]
# Add audio if present in original file
optimize_args.extend(["-c:a", "libopus", "-b:a", "96k"])
# Add output file
optimize_args.append(discord_optimized_file)
elif format_ext == "gif":
# GIF optimization for Discord
optimize_args = [
ffmpeg_path, "-i", input_file,
"-vf", "fps=15,scale=trunc(iw/2)*2:trunc(ih/2)*2",
]
# Add output file
optimize_args.append(discord_optimized_file)
if optimize_args:
print(f"Creating Discord-optimized version of {format_ext.upper()} file...")
subprocess.run(optimize_args, check=True, capture_output=True)
print(f"Discord-optimized file created: {discord_optimized_file}")
else:
# If no optimization needed, just use the original file
# Set to None to indicate we're using the original file so we don't clean it up
discord_optimized_file = None
print(f"Using original file for Discord: {input_file}")
except Exception as e:
print(f"Warning: Failed to create Discord-optimized file: {str(e)}")
print("Falling back to original file")
discord_optimized_file = None # Using original file
# Use shared utility for Discord optimization
discord_optimized_file = shared_optimize_video(input_file, ffmpeg_path, temp_dir)
if discord_optimized_file is None:
print(f"Using original file for Discord: {input_file}")
# Prepare Discord files and message
discord_files = []
+29
View File
@@ -5,7 +5,36 @@ Provides image and video processing functions.
"""
from .image_processing import tensor_to_numpy_uint8
from .format_utils import (
parse_format_string,
normalize_video_extension,
get_mime_type,
validate_video_for_discord,
is_animated_format,
supports_alpha
)
from .video_encoder import (
detect_ffmpeg,
FFmpegEncoder,
PILEncoder,
optimize_video_for_discord,
mux_audio_to_video
)
__all__ = [
# Image processing
'tensor_to_numpy_uint8',
# Format utilities
'parse_format_string',
'normalize_video_extension',
'get_mime_type',
'validate_video_for_discord',
'is_animated_format',
'supports_alpha',
# Video encoding
'detect_ffmpeg',
'FFmpegEncoder',
'PILEncoder',
'optimize_video_for_discord',
'mux_audio_to_video',
]
+137
View File
@@ -0,0 +1,137 @@
"""
Video format utilities for ComfyUI-DiscordSend
Provides format detection, extension mapping, and validation.
"""
import os
from typing import Tuple, Optional
def parse_format_string(format_str: str) -> Tuple[str, str]:
"""
Parse a format string into type and extension.
Args:
format_str: Format string like "video/h264-mp4" or "image/gif"
Returns:
Tuple of (format_type, format_extension)
"""
if "/" in format_str:
format_type, format_ext = format_str.split("/", 1)
else:
format_type = "video"
format_ext = format_str
return format_type, format_ext
def normalize_video_extension(format_str: str) -> str:
"""
Normalize a format string to a file extension.
Args:
format_str: Format string like "video/h264-mp4"
Returns:
Normalized extension (e.g., "mp4", "webm", "gif")
"""
_, format_ext = parse_format_string(format_str)
# Map format strings to extensions
extension_map = {
"h264-mp4": "mp4",
"h265-mp4": "mp4",
"vp9-webm": "webm",
"prores": "mov",
}
return extension_map.get(format_ext, format_ext)
def get_mime_type(extension: str) -> str:
"""
Get MIME type for a video extension.
Args:
extension: File extension (without dot)
Returns:
MIME type string
"""
mime_types = {
"mp4": "video/mp4",
"webm": "video/webm",
"gif": "image/gif",
"mov": "video/quicktime",
"avi": "video/x-msvideo",
"mkv": "video/x-matroska",
}
return mime_types.get(extension.lower(), "application/octet-stream")
def validate_video_for_discord(file_path: str, max_size_mb: int = 25) -> Tuple[bool, str]:
"""
Validate that a video file is compatible with Discord.
Args:
file_path: Path to the video file
max_size_mb: Maximum file size in megabytes (default 25MB for Discord)
Returns:
Tuple of (is_valid, message)
"""
if not os.path.exists(file_path):
return False, f"File does not exist: {file_path}"
file_size = os.path.getsize(file_path)
if file_size == 0:
return False, "File is empty"
if file_size < 1024:
return False, f"File is suspiciously small: {file_size} bytes"
max_size = max_size_mb * 1024 * 1024
if file_size > max_size:
return False, f"File exceeds Discord's size limit of {max_size_mb}MB ({file_size / (1024*1024):.2f}MB)"
ext = os.path.splitext(file_path)[1].lower().lstrip('.')
if ext in ['mp4', 'webm', 'gif']:
return True, "Valid"
elif ext in ['mov']:
return False, "MOV files may need conversion for Discord compatibility"
elif ext in ['png', 'apng']:
return False, "PNG/APNG sequence may need compilation for Discord"
else:
return False, f"Unknown format '{ext}' - may not be compatible with Discord"
def is_animated_format(extension: str) -> bool:
"""
Check if a format supports animation.
Args:
extension: File extension (without dot)
Returns:
True if the format supports animation
"""
animated_formats = {'gif', 'webp', 'mp4', 'webm', 'mov', 'avi', 'mkv', 'apng'}
return extension.lower() in animated_formats
def supports_alpha(extension: str) -> bool:
"""
Check if a format supports alpha channel (transparency).
Args:
extension: File extension (without dot)
Returns:
True if the format supports alpha
"""
alpha_formats = {'webm', 'gif', 'webp', 'png', 'apng', 'mov'}
return extension.lower() in alpha_formats
+510
View File
@@ -0,0 +1,510 @@
"""
Video encoding utilities for ComfyUI-DiscordSend
Provides FFmpeg-based video encoding with fallback to PIL for GIF/WebP.
"""
import os
import subprocess
from typing import List, Tuple, Optional, Iterator, Any, Callable
from uuid import uuid4
import numpy as np
def detect_ffmpeg() -> Optional[str]:
"""
Detect FFmpeg binary location.
Returns:
Path to FFmpeg executable, or None if not found
"""
ffmpeg_path = None
# Try imageio-ffmpeg first (common in Python environments)
try:
import imageio_ffmpeg
ffmpeg_path = imageio_ffmpeg.get_ffmpeg_exe()
print(f"Found ffmpeg via imageio_ffmpeg: {ffmpeg_path}")
return ffmpeg_path
except (ImportError, Exception):
pass
# Try system PATH
try:
from shutil import which
ffmpeg_path = which("ffmpeg")
if ffmpeg_path:
print(f"Found ffmpeg in system path: {ffmpeg_path}")
return ffmpeg_path
except Exception:
pass
return None
class FFmpegEncoder:
"""
FFmpeg-based video encoder supporting multiple formats.
"""
def __init__(self, ffmpeg_path: Optional[str] = None):
"""
Initialize the encoder.
Args:
ffmpeg_path: Path to FFmpeg executable (auto-detected if None)
"""
self.ffmpeg_path = ffmpeg_path or detect_ffmpeg()
if not self.ffmpeg_path:
raise RuntimeError("FFmpeg not found. Install ffmpeg or imageio-ffmpeg.")
def encode(
self,
images: List[np.ndarray],
output_path: str,
format_ext: str,
frame_rate: float = 24.0,
quality: int = 85,
lossless: bool = False,
loop_count: int = 0,
codec: Optional[str] = None,
progress_callback: Optional[Callable[[int, int], None]] = None
) -> str:
"""
Encode images to video using FFmpeg.
Args:
images: List of numpy arrays (H, W, C) in uint8 format
output_path: Output file path
format_ext: Output format extension (mp4, webm, gif)
frame_rate: Frame rate in FPS
quality: Quality level 1-100
lossless: Use lossless encoding if supported
loop_count: Number of loops (0 = infinite for GIF)
codec: Specific codec to use (h264, h265, vp9, etc.)
progress_callback: Optional callback(current, total) for progress
Returns:
Path to the encoded file
"""
if not images:
raise ValueError("No images provided for encoding")
# Get dimensions from first image
height, width = images[0].shape[:2]
has_alpha = images[0].shape[2] == 4 if len(images[0].shape) > 2 else False
# Determine input pixel format
i_pix_fmt = "rgba" if has_alpha else "rgb24"
dimensions = f"{width}x{height}"
# Build FFmpeg arguments
args = self._build_ffmpeg_args(
format_ext=format_ext,
dimensions=dimensions,
frame_rate=frame_rate,
quality=quality,
lossless=lossless,
loop_count=loop_count,
i_pix_fmt=i_pix_fmt,
has_alpha=has_alpha,
codec=codec,
output_path=output_path
)
# Execute encoding
self._execute_encoding(args, images, i_pix_fmt, progress_callback)
return output_path
def _build_ffmpeg_args(
self,
format_ext: str,
dimensions: str,
frame_rate: float,
quality: int,
lossless: bool,
loop_count: int,
i_pix_fmt: str,
has_alpha: bool,
codec: Optional[str],
output_path: str
) -> List[str]:
"""Build FFmpeg command arguments."""
# Loop arguments
loop_args = []
if format_ext == "gif":
loop_args = ["-loop", "0" if loop_count == 0 else str(loop_count)]
# Base input arguments
args = [
self.ffmpeg_path, "-v", "error",
"-f", "rawvideo",
"-pix_fmt", i_pix_fmt,
"-s", dimensions,
"-r", str(frame_rate),
"-i", "-"
] + loop_args
# Format-specific encoding arguments
if format_ext == "gif":
args.extend(self._get_gif_args(quality))
elif format_ext == "mp4":
args.extend(self._get_mp4_args(quality, lossless, codec))
elif format_ext == "webm":
args.extend(self._get_webm_args(quality, lossless, has_alpha))
else:
# Default to MP4-like encoding
args.extend(self._get_mp4_args(quality, lossless, codec))
# Add output path
args.extend(["-y", output_path])
return args
def _get_gif_args(self, quality: int) -> List[str]:
"""Get FFmpeg arguments for GIF encoding."""
# Use palettegen for better quality
if quality >= 80:
return [
"-vf", "split[s0][s1];[s0]palettegen=max_colors=256:stats_mode=diff[p];[s1][p]paletteuse=dither=sierra2",
"-f", "gif"
]
else:
return [
"-vf", "split[s0][s1];[s0]palettegen[p];[s1][p]paletteuse",
"-f", "gif"
]
def _get_mp4_args(self, quality: int, lossless: bool, codec: Optional[str]) -> List[str]:
"""Get FFmpeg arguments for MP4 encoding."""
args = []
# Determine codec
use_h265 = codec == "h265" or codec == "hevc"
if lossless:
if use_h265:
args.extend(["-c:v", "libx265", "-x265-params", "lossless=1"])
else:
args.extend(["-c:v", "libx264", "-crf", "0"])
else:
# Map quality (1-100) to CRF (51-0 for h264, lower is better)
crf = int(51 - (quality / 100 * 33)) # Maps 1->51, 100->18
if use_h265:
args.extend(["-c:v", "libx265", "-crf", str(crf + 5)]) # H.265 uses different CRF scale
else:
args.extend(["-c:v", "libx264", "-crf", str(crf)])
# Always use yuv420p for Discord compatibility
args.extend(["-pix_fmt", "yuv420p", "-movflags", "faststart"])
return args
def _get_webm_args(self, quality: int, lossless: bool, has_alpha: bool) -> List[str]:
"""Get FFmpeg arguments for WebM encoding."""
args = ["-c:v", "libvpx-vp9"]
if lossless:
args.extend(["-lossless", "1"])
else:
# Map quality to CRF (63-0 for VP9)
crf = int(63 - (quality / 100 * 33)) # Maps 1->63, 100->30
args.extend(["-crf", str(crf), "-b:v", "0"])
# Pixel format - support alpha if present
pix_fmt = "yuva420p" if has_alpha else "yuv420p"
args.extend(["-pix_fmt", pix_fmt])
# VP9 threading
args.extend(["-row-mt", "1"])
return args
def _execute_encoding(
self,
args: List[str],
images: List[np.ndarray],
i_pix_fmt: str,
progress_callback: Optional[Callable[[int, int], None]] = None
) -> None:
"""Execute FFmpeg process and feed frames."""
total_frames = len(images)
# Create image chunk iterator for memory efficiency
def image_chunks() -> Iterator[bytes]:
for i, img in enumerate(images):
# Ensure contiguous array for subprocess
chunk = np.ascontiguousarray(img)
if progress_callback:
progress_callback(i + 1, total_frames)
yield chunk.tobytes()
# Start FFmpeg process
process = subprocess.Popen(
args,
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE
)
# Feed frames
try:
for chunk in image_chunks():
process.stdin.write(chunk)
process.stdin.close()
process.wait()
if process.returncode != 0:
stderr = process.stderr.read().decode('utf-8', errors='ignore')
raise RuntimeError(f"FFmpeg encoding failed: {stderr}")
finally:
if process.stdin:
process.stdin.close()
if process.stdout:
process.stdout.close()
if process.stderr:
process.stderr.close()
class PILEncoder:
"""
PIL-based encoder for GIF and WebP formats.
Fallback when FFmpeg is not available.
"""
def encode(
self,
images: List[Any], # PIL Images or numpy arrays
output_path: str,
format_ext: str,
frame_rate: float = 24.0,
quality: int = 85,
lossless: bool = False,
loop_count: int = 0,
tensor_to_numpy_func: Optional[Callable] = None
) -> str:
"""
Encode images using PIL.
Args:
images: List of PIL Images or numpy arrays
output_path: Output file path
format_ext: Output format (gif, webp)
frame_rate: Frame rate in FPS
quality: Quality level 1-100
lossless: Use lossless encoding for WebP
loop_count: Number of loops (0 = infinite)
tensor_to_numpy_func: Optional function to convert tensors to numpy
Returns:
Path to the encoded file
"""
from PIL import Image
# Convert to PIL images if needed
pil_images = []
for img in images:
if hasattr(img, 'shape'): # numpy array or tensor
if tensor_to_numpy_func and hasattr(img, 'cpu'):
img = tensor_to_numpy_func(img)
elif hasattr(img, 'numpy'):
img = img.numpy()
pil_images.append(Image.fromarray(img.astype(np.uint8)))
else:
pil_images.append(img)
if not pil_images:
raise ValueError("No images provided for encoding")
# Calculate frame duration in milliseconds
duration = int(1000 / frame_rate)
if format_ext.lower() == "gif":
self._encode_gif(pil_images, output_path, duration, loop_count)
elif format_ext.lower() == "webp":
self._encode_webp(pil_images, output_path, duration, loop_count, quality, lossless)
else:
# Single frame fallback
pil_images[0].save(output_path, format=format_ext.upper())
return output_path
def _encode_gif(
self,
images: List[Any],
output_path: str,
duration: int,
loop_count: int
) -> None:
"""Encode images as GIF."""
durations = [duration] * len(images)
images[0].save(
output_path,
format="GIF",
append_images=images[1:] if len(images) > 1 else [],
save_all=True,
duration=durations,
loop=0 if loop_count == 0 else loop_count,
optimize=False
)
def _encode_webp(
self,
images: List[Any],
output_path: str,
duration: int,
loop_count: int,
quality: int,
lossless: bool
) -> None:
"""Encode images as WebP."""
save_kwargs = {
"format": "WEBP",
"append_images": images[1:] if len(images) > 1 else [],
"save_all": True,
"duration": duration,
"loop": 0 if loop_count == 0 else loop_count,
}
if lossless:
save_kwargs["lossless"] = True
else:
save_kwargs["quality"] = quality
images[0].save(output_path, **save_kwargs)
def optimize_video_for_discord(
input_file: str,
ffmpeg_path: str,
temp_dir: str
) -> Optional[str]:
"""
Create a Discord-optimized version of a video file.
Args:
input_file: Path to the input video file
ffmpeg_path: Path to FFmpeg executable
temp_dir: Directory for temporary files
Returns:
Path to the optimized file, or None if optimization failed
"""
format_ext = os.path.splitext(input_file)[1].lstrip('.').lower()
discord_optimized_file = os.path.join(temp_dir, f"discord_optimized_{uuid4()}.{format_ext}")
try:
if format_ext == "mp4":
optimize_args = [
ffmpeg_path, "-i", input_file,
"-c:v", "libx264", "-pix_fmt", "yuv420p",
"-movflags", "faststart", "-preset", "fast",
"-profile:v", "baseline", "-level", "3.0",
"-crf", "23",
"-c:a", "aac", "-b:a", "128k",
"-y", discord_optimized_file
]
elif format_ext == "webm":
optimize_args = [
ffmpeg_path, "-i", input_file,
"-c:v", "libvpx-vp9",
"-pix_fmt", "yuv420p",
"-crf", "30", "-b:v", "0",
"-deadline", "good",
"-c:a", "libopus", "-b:a", "96k",
"-y", discord_optimized_file
]
elif format_ext == "gif":
optimize_args = [
ffmpeg_path, "-i", input_file,
"-vf", "fps=15,scale=trunc(iw/2)*2:trunc(ih/2)*2",
"-y", discord_optimized_file
]
else:
print(f"No optimization rules for format: {format_ext}")
return None
print(f"Creating Discord-optimized version of {format_ext.upper()} file...")
result = subprocess.run(
optimize_args,
capture_output=True,
text=True
)
if result.returncode == 0 and os.path.exists(discord_optimized_file):
print(f"Discord-optimized file created: {discord_optimized_file}")
return discord_optimized_file
else:
print(f"Optimization failed: {result.stderr}")
return None
except Exception as e:
print(f"Error during Discord optimization: {e}")
return None
def mux_audio_to_video(
video_path: str,
audio_waveform: np.ndarray,
sample_rate: int,
format_ext: str,
ffmpeg_path: str,
output_path: str,
channels: int = 2
) -> bool:
"""
Mux audio into a video file.
Args:
video_path: Path to the video file
audio_waveform: Audio data as numpy array
sample_rate: Audio sample rate
format_ext: Video format extension
ffmpeg_path: Path to FFmpeg executable
output_path: Output path for the muxed file
channels: Number of audio channels
Returns:
True if successful, False otherwise
"""
try:
# Determine audio codec based on format
if format_ext == "mp4":
audio_pass = ["-c:a", "aac", "-b:a", "192k"]
elif format_ext == "webm":
audio_pass = ["-c:a", "libopus", "-b:a", "128k"]
else:
audio_pass = ["-c:a", "libopus", "-b:a", "128k"]
mux_args = [
ffmpeg_path, "-v", "error", "-y",
"-i", video_path,
"-ar", str(sample_rate),
"-ac", str(channels),
"-f", "f32le",
"-i", "-",
"-c:v", "copy"
] + audio_pass + ["-shortest", output_path]
# Ensure contiguous array for subprocess
audio_data = np.ascontiguousarray(audio_waveform)
result = subprocess.run(
mux_args,
input=memoryview(audio_data),
capture_output=True
)
if result.returncode == 0:
print(f"Successfully muxed audio to video: {output_path}")
return True
else:
print(f"Audio muxing failed: {result.stderr.decode('utf-8', errors='ignore')}")
return False
except Exception as e:
print(f"Error muxing audio: {e}")
return False