From 3a2d5fe563b9e7c1bcd9aa7ce067ed07d0ef5f94 Mon Sep 17 00:00:00 2001 From: Adrien Toupet Date: Sun, 30 Nov 2025 15:47:00 -0500 Subject: [PATCH] Deprecate nightly branch - redirect users to main branch --- README.md | 870 +------------------ inference_cli.py | 1356 +----------------------------- src/interfaces/video_upscaler.py | 14 + 3 files changed, 30 insertions(+), 2210 deletions(-) diff --git a/README.md b/README.md index 22684e3..47a5c31 100644 --- a/README.md +++ b/README.md @@ -1,869 +1,13 @@ -# ComfyUI-SeedVR2_VideoUpscaler +# โš ๏ธ DEPRECATED BRANCH -[![View Code](https://img.shields.io/badge/๐Ÿ“‚_View_Code-GitHub-181717?style=for-the-badge&logo=github)](https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler) +**This nightly branch is no longer supported.** -Official release of [SeedVR2](https://github.com/ByteDance-Seed/SeedVR) for ComfyUI that enables high-quality video and image upscaling. +Please use the **main branch** for the latest SeedVR2 stable version with all features and bug fixes. -Can run as **Multi-GPU standalone CLI** too, see [๐Ÿ–ฅ๏ธ Run as Standalone](#๏ธ-run-as-standalone-cli) section. +## Installation -[![SeedVR2 v2.5 Deep Dive Tutorial](https://img.youtube.com/vi/MBtWYXq_r60/maxresdefault.jpg)](https://youtu.be/MBtWYXq_r60) +Install via [ComfyUI Manager](https://github.com/ltdrdata/ComfyUI-Manager) (recommended) or visit the official repository: -![Usage Example](docs/usage_01.png) +**๐Ÿ“ฆ https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler** -![Usage Example](docs/usage_02.png) - -## ๐Ÿ“‹ Quick Access - -- [๐Ÿ†™ Future Releases](#-future-releases) -- [๐Ÿš€ Updates](#-updates) -- [๐ŸŽฏ Features](#-features) -- [๐Ÿ”ง Requirements](#-requirements) -- [๐Ÿ“ฆ Installation](#-installation) -- [๐Ÿ“– Usage](#-usage) -- [๐Ÿ–ฅ๏ธ Run as Standalone](#๏ธ-run-as-standalone-cli) -- [โš ๏ธ Limitations](#๏ธ-limitations) -- [๐Ÿค Contributing](#-contributing) -- [๐Ÿ™ Credits](#-credits) -- [๐Ÿ“œ License](#-license) - -## ๐Ÿ†™ Future Releases - -We're actively working on improvements and new features. To stay informed: - -- **๐Ÿ“Œ Track Active Development**: Visit [Issues](https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler/issues) to see active development, report bugs, and request new features -- **๐Ÿ’ฌ Join the Community**: Learn from others, share your workflows, and get help in the [Discussions](https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler/discussions) -- **๐Ÿ”ฎ Next Model Survey**: We're looking for community input on the next open-source super-powerful generic restoration model. Share your suggestions in [Issue #164](https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler/issues/164) - -## ๐Ÿš€ Updates - -**2025.11.07 - Version 2.5.0** ๐ŸŽ‰ - -โš ๏ธ **BREAKING CHANGE**: This is a major update requiring workflow recreation. All nodes and CLI parameters have been redesigned for better usability and consistency. Watch the latest video from [AInVFX](https://www.youtube.com/@AInVFX) for a deep dive and check out the [usage](#-usage) section. - -**๐Ÿ“ฆ Official Release**: Now available on main branch with ComfyUI Manager support for easy installation and automatic version tracking. Updated dependencies and local imports prevent conflicts with other ComfyUI custom nodes. - -### ๐ŸŽจ ComfyUI Improvements - -- **Four-Node Modular Architecture**: Split into dedicated nodes for DiT model, VAE model, torch.compile settings, and main upscaler for granular control -- **Global Model Cache**: Models now shared across multiple upscaler instances with automatic config updates - no more redundant loading -- **ComfyUI V3 Migration**: Full compatibility with ComfyUI V3 stateless node design -- **RGBA Support**: Native alpha channel processing with edge-guided upscaling for clean transparency -- **Improved Memory Management**: Streaming architecture prevents VRAM spikes regardless of video length -- **Flexible Resolution Support**: Upscale to any resolution divisible by 2 with lossless padding approach (replaced restrictive cropping) -- **Enhanced Parameters**: Added `uniform_batch_size`, `temporal_overlap`, `prepend_frames`, and `max_resolution` for better control - -### ๐Ÿ–ฅ๏ธ CLI Enhancements - -- **Batch Directory Processing**: Process entire folders of videos/images with model caching for efficiency -- **Single Image Support**: Direct image upscaling without video conversion -- **Smart Output Detection**: Auto-detects output format (MP4/PNG) based on input type -- **Enhanced Multi-GPU**: Improved workload distribution with temporal overlap blending -- **Unified Parameters**: CLI and ComfyUI now use identical parameter names for consistency -- **Better UX**: Auto-display help, validation improvements, progress tracking, and cleaner output - -### โšก Performance & Optimization - -- **torch.compile Support**: 20-40% DiT speedup and 15-25% VAE speedup with full graph compilation -- **Optimized BlockSwap**: Adaptive memory clearing (5% threshold), separate I/O component handling, reduced overhead -- **Enhanced VAE Tiling**: Tensor offload support for accumulation buffers, separate encode/decode configuration -- **Native Dtype Pipeline**: Eliminated unnecessary conversions, maintains bfloat16 precision throughout for speed and quality -- **Optimized Tensor Operations**: Replaced einops rearrange with native PyTorch ops for 2-5x faster transforms - -### ๐ŸŽฏ Quality Improvements - -- **LAB Color Correction**: New perceptual color transfer method with superior color accuracy (now default) -- **Additional Color Methods**: HSV saturation matching, wavelet adaptive, and hybrid approaches -- **Deterministic Generation**: Seed-based reproducibility with phase-specific seeding strategy -- **Better Temporal Consistency**: Hann window blending for smooth transitions between batches - -### ๐Ÿ’พ Memory Management - -- **Smarter Offloading**: Independent device configuration for DiT, VAE, and tensors (CPU/GPU/none) -- **Four-Phase Pipeline**: Completes each phase (encodeโ†’upscaleโ†’decodeโ†’postprocess) for all batches before moving to next, minimizing model swaps -- **Better Cleanup**: Phase-specific resource management with proper tensor memory release -- **Peak VRAM Tracking**: Per-phase memory monitoring with summary display - -### ๐Ÿ”ง Technical Improvements - -- **GGUF Quantization Support**: Added full GGUF support for 4-bit/8-bit inference on low-VRAM systems -- **Improved GGUF Handling**: Fixed VRAM leaks, torch.compile compatibility, non-persistent buffers -- **Apple Silicon Support**: Full MPS (Metal Performance Shaders) support for Apple Silicon Macs -- **AMD ROCm Compatibility**: Conditional FSDP imports for PyTorch ROCm 7+ support -- **Conv3d Memory Workaround**: Fixes PyTorch 2.9+ cuDNN memory bug (3x usage reduction) -- **Flash Attention Optional**: Graceful fallback to SDPA when flash-attn unavailable - -### ๐Ÿ“š Code Quality - -- **Modular Architecture**: Split monolithic files into focused modules (generation_phases, model_configuration, etc.) -- **Comprehensive Documentation**: Extensive docstrings with type hints across all modules -- **Better Error Handling**: Early validation, clear error messages, installation instructions -- **Consistent Logging**: Unified indentation, better categorization, concise messages - -**2025.08.07** - -- ๐ŸŽฏ **Unified Debug System**: New structured logging with categories, timers, and memory tracking. `enable_debug` now available on main node -- โšก **Smart FP8 Optimization**: FP8 models now keep native FP8 storage, converting to BFloat16 only for arithmetic - faster and more memory efficient than FP16 -- ๐Ÿ“ฆ **Model Registry**: Multi-repo support (numz/ & AInVFX/), auto-discovery of user models, added mixed FP8 variants to fix 7B artifacts -- ๐Ÿ’พ **Model Caching**: `cache_model` moved to main node, fixed memory leaks with proper RoPE/wrapper cleanup -- ๐Ÿงน **Code Cleanup**: New modular structure (`constants.py`, `model_registry.py`, `debug.py`), removed legacy code -- ๐Ÿš€ **Performance**: Better memory management with `torch.cuda.ipc_collect()`, improved RoPE handling - -**2025.07.17** - -- ๐Ÿ› ๏ธ Add 7B sharp Models: add 2 new 7B models with sharpen output - -**2025.07.11** - -- ๐ŸŽฌ Complete tutorial released: Adrien from [AInVFX](https://www.youtube.com/@AInVFX) created an in-depth ComfyUI SeedVR2 guide covering everything from basic setup to advanced BlockSwap techniques for running on consumer GPUs. Perfect for understanding memory optimization and upscaling of image sequences with alpha channel! [Watch the tutorial](#-usage) - -**2025.09.07** - -- ๐Ÿ› ๏ธ Blockswap Integration: Big thanks to [Adrien Toupet](https://github.com/adrientoupet) from [AInVFX](https://www.youtube.com/@AInVFX) for this :), useful for low VRAM users (see [usage](#-usage) section) - -**2025.07.03** - -- ๐Ÿ› ๏ธ Can run as **standalone mode** with **Multi GPU** see [๐Ÿ–ฅ๏ธ Run as Standalone](#๏ธ-run-as-standalone-cli) - -**2025.06.30** - -- ๐Ÿš€ Speed Up the process and less VRAM used -- ๐Ÿ› ๏ธ Fixed memory leak on 3B models -- โŒ Can now interrupt process if needed -- โœ… Refactored the code for better sharing with the community, feel free to propose pull requests -- ๐Ÿ› ๏ธ Removed flash attention dependency (thanks to [luke2642](https://github.com/Luke2642) !!) - -**2025.06.24** - -- ๐Ÿš€ Speed up the process until x4 - -**2025.06.22** - -- ๐Ÿ’ช FP8 compatibility ! -- ๐Ÿš€ Speed Up all Process -- ๐Ÿš€ less VRAM consumption (Stay high, batch_size=1 for RTX4090 max, I'm trying to fix that) -- ๐Ÿ› ๏ธ Better benchmark coming soon - -**2025.06.20** - -- ๐Ÿ› ๏ธ Initial push - -## ๐ŸŽฏ Features - -### Core Capabilities -- **High-Quality Diffusion-Based Upscaling**: One-step diffusion model for video and image enhancement -- **Temporal Consistency**: Maintains coherence across video frames with configurable batch processing -- **Multi-Format Support**: Handles RGB and RGBA (alpha channel) for both videos and images -- **Any Video Length**: Suitable for any video length - -### Model Support -- **Multiple Model Variants**: 3B and 7B parameter models with different precision options -- **FP16, FP8, and GGUF Quantization**: Choose between full precision (FP16), mixed precision (FP8), or heavily quantized GGUF models for different VRAM requirements -- **Automatic Model Downloads**: Models are automatically downloaded from HuggingFace on first use - -### Memory Optimization -- **BlockSwap Technology**: Dynamically swap transformer blocks between GPU and CPU memory to run large models on limited VRAM -- **VAE Tiling**: Process large resolutions with tiled encoding/decoding to reduce VRAM usage -- **Intelligent Offloading**: Offload models and intermediate tensors to CPU or secondary GPUs between processing phases -- **GGUF Quantization Support**: Run models with 4-bit or 8-bit quantization for extreme VRAM savings - -### Performance Features -- **torch.compile Integration**: Optional 20-40% DiT speedup and 15-25% VAE speedup with PyTorch 2.0+ compilation -- **Multi-GPU CLI**: Distribute workload across multiple GPUs with automatic temporal overlap blending -- **Model Caching**: Keep models loaded in memory for faster batch processing -- **Flexible Attention Backends**: Choose between PyTorch SDPA (stable, always available) or Flash Attention 2 (faster on supported hardware) - -### Quality Control -- **Advanced Color Correction**: Five methods including LAB (recommended for highest fidelity), wavelet, wavelet adaptive, HSV, and AdaIN -- **Noise Injection Controls**: Fine-tune input and latent noise scales for artifact reduction at high resolutions -- **Configurable Resolution Limits**: Set target and maximum resolutions with automatic aspect ratio preservation - -### Workflow Features -- **ComfyUI Integration**: Four dedicated nodes for complete control over the upscaling pipeline -- **Standalone CLI**: Command-line interface for batch processing and automation -- **Debug Logging**: Comprehensive debug mode with memory tracking, timing information, and processing details -- **Progress Reporting**: Real-time progress updates during processing - -## ๐Ÿ”ง Requirements - -### Hardware - -With the current optimizations (tiling, BlockSwap, GGUF quantization), SeedVR2 can run on a wide range of hardware: - -- **Minimal VRAM** (8GB or less): Use GGUF Q4_K_M models with BlockSwap and VAE tiling enabled -- **Moderate VRAM** (12-16GB): Use FP8 models with BlockSwap or VAE tiling as needed -- **High VRAM** (24GB+): Use FP16 models for best quality and speed without memory optimizations - -### Software - -- **ComfyUI**: Latest version recommended -- **Python**: 3.12+ (Python 3.12 and 3.13 tested and recommended) -- **PyTorch**: 2.0+ for torch.compile support (optional but recommended) -- **Triton**: Required for torch.compile with inductor backend (optional) -- **Flash Attention 2**: Provides faster attention computation on supported hardware (optional, falls back to PyTorch SDPA) - -## ๐Ÿ“ฆ Installation - -### Option 1: ComfyUI Manager (Recommended) - -1. Open ComfyUI Manager in your ComfyUI interface -2. Click "Custom Nodes Manager" -3. Search for "ComfyUI-SeedVR2_VideoUpscaler" -4. Click "Install" and restart ComfyUI - -**Registry Link**: [ComfyUI Registry - SeedVR2 Video Upscaler](https://registry.comfy.org/nodes/seedvr2_videoupscaler) - -### Option 2: Manual Installation - -1. **Clone the repository** into your ComfyUI custom nodes directory: -```bash -cd ComfyUI -git clone https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler.git custom_nodes/seedvr2_videoupscaler -``` - -2. **Install dependencies using standalone Python**: -```bash -# Install requirements (from same ComfyUI directory) -# Windows: -.venv\Scripts\python.exe -m pip install -r custom_nodes\seedvr2_videoupscaler\requirements.txt -# Linux/macOS: -.venv/bin/python -m pip install -r custom_nodes/seedvr2_videoupscaler/requirements.txt -``` - -3. **Restart ComfyUI** - -### Model Installation - -Models will be **automatically downloaded** on first use and saved to `ComfyUI/models/SEEDVR2`. - -You can also manually download models from: -- Main models available at [numz/SeedVR2_comfyUI](https://huggingface.co/numz/SeedVR2_comfyUI/tree/main) and [AInVFX/SeedVR2_comfyUI](https://huggingface.co/AInVFX/SeedVR2_comfyUI/tree/main) -- Additional GGUF models available at [cmeka/SeedVR2-GGUF](https://huggingface.co/cmeka/SeedVR2-GGUF/tree/main) - -## ๐Ÿ“– Usage - -### ๐ŸŽฌ Video Tutorials - -#### Latest Version Deep Dive (Recommended) - -Complete walkthrough of version 2.5 by Adrien from [AInVFX](https://www.youtube.com/@AInVFX), covering the new 4-node architecture, GGUF support, memory optimizations, and production workflows: - -[![SeedVR2 v2.5 Deep Dive Tutorial](https://img.youtube.com/vi/MBtWYXq_r60/maxresdefault.jpg)](https://youtu.be/MBtWYXq_r60) - -This comprehensive tutorial covers: -- Installing v2.5 through ComfyUI Manager and troubleshooting conflicts -- Understanding the new 4-node modular architecture and why we rebuilt it -- Running 7B models on 8GB VRAM with GGUF quantization -- Configuring BlockSwap, VAE tiling, and torch.compile for your hardware -- Image and video upscaling workflows with alpha channel support -- CLI for batch processing and multi-GPU rendering -- Memory optimization strategies for different VRAM levels -- Real production tips and the critical batch_size formula (4n+1) - -#### Previous Version Tutorial - -For reference, here's the original tutorial covering the initial release: - -[![SeedVR2 Deep Dive Tutorial](https://img.youtube.com/vi/I0sl45GMqNg/maxresdefault.jpg)](https://youtu.be/I0sl45GMqNg) - -*Note: This tutorial covers the previous single-node architecture. While the UI has changed significantly in v2.5, the core concepts about BlockSwap and memory management remain valuable.* - -### Node Setup - -SeedVR2 uses a modular node architecture with four specialized nodes: - -#### 1. SeedVR2 (Down)Load DiT Model - -![SeedVR2 (Down)Load DiT Model](docs/dit_model_loader.png) - -Configure the DiT (Diffusion Transformer) model for video upscaling. - -**Parameters:** - -- **model**: Choose your DiT model - - **3B Models**: Faster, lower VRAM requirements - - `seedvr2_ema_3b_fp16.safetensors`: FP16 (best quality) - - `seedvr2_ema_3b_fp8_e4m3fn.safetensors`: FP8 8-bit (good quality) - - `seedvr2_ema_3b-Q4_K_M.gguf`: GGUF 4-bit quantized (acceptable quality) - - `seedvr2_ema_3b-Q8_0.gguf`: GGUF 8-bit quantized (good quality) - - **7B Models**: Higher quality, higher VRAM requirements - - `seedvr2_ema_7b_fp16.safetensors`: FP16 (best quality) - - `seedvr2_ema_7b_fp8_e4m3fn_mixed_block35_fp16.safetensors`: FP8 with last block in FP16 to reduce artifacts (good quality) - - `seedvr2_ema_7b-Q4_K_M.gguf`: GGUF 4-bit quantized (acceptable quality) - - `seedvr2_ema_7b_sharp_*`: Sharp variants for enhanced detail - -- **device**: GPU device for DiT inference (e.g., `cuda:0`) - -- **offload_device**: Device to offload DiT model when not actively processing - - `none`: Keep model on inference device (fastest, highest VRAM) - - `cpu`: Offload to system RAM (reduces VRAM) - - `cuda:X`: Offload to another GPU (good balance if available) - -- **cache_model**: Keep DiT model loaded on offload_device between workflow runs - - Useful for batch processing to avoid repeated loading - - Requires offload_device to be set - -- **blocks_to_swap**: BlockSwap memory optimization - - `0`: Disabled (default) - - `1-32`: Number of transformer blocks to swap for 3B model - - `1-36`: Number of transformer blocks to swap for 7B model - - Higher values = more VRAM savings but slower processing - - Requires offload_device to be set and different from device - -- **swap_io_components**: Offload input/output embeddings and normalization layers - - Additional VRAM savings when combined with blocks_to_swap - - Requires offload_device to be set and different from device - -- **attention_mode**: Attention computation backend - - `sdpa`: PyTorch scaled_dot_product_attention (default, stable, always available) - - `flash_attn`: Flash Attention 2 (faster on supported hardware, requires flash-attn package) - -- **torch_compile_args**: Connect to SeedVR2 Torch Compile Settings node for 20-40% speedup - -**BlockSwap Explained:** - -BlockSwap enables running large models on GPUs with limited VRAM by dynamically swapping transformer blocks between GPU and CPU memory during inference. Here's how it works: - -- **What it does**: Keeps only the currently-needed transformer blocks on the GPU, while storing the rest on CPU or another device -- **When to use it**: When you get OOM (Out of Memory) errors during the upscaling phase -- **How to configure**: - 1. Set `offload_device` to `cpu` or another GPU - 2. Start with `blocks_to_swap=16` (half the blocks) - 3. If still getting OOM, increase to 24 or 32 (3B) / 36 (7B) - 4. Enable `swap_io_components` for maximum VRAM savings - 5. If you have plenty of VRAM, decrease or set to 0 for faster processing - -**Example Configuration for Low VRAM (8GB)**: -- model: `seedvr2_ema_3b-Q8_0.gguf` -- device: `cuda:0` -- offload_device: `cpu` -- blocks_to_swap: `32` -- swap_io_components: `True` - -#### 2. SeedVR2 (Down)Load VAE Model - -![SeedVR2 (Down)Load VAE Model](docs/vae_model_loader.png) - -Configure the VAE (Variational Autoencoder) model for encoding/decoding video frames. - -**Parameters:** - -- **model**: VAE model selection - - `ema_vae_fp16.safetensors`: Default and recommended - -- **device**: GPU device for VAE inference (e.g., `cuda:0`) - -- **offload_device**: Device to offload VAE model when not actively processing - - `none`: Keep model on inference device (default, fastest) - - `cpu`: Offload to system RAM (reduces VRAM) - - `cuda:X`: Offload to another GPU (good balance if available) - -- **cache_model**: Keep VAE model loaded on offload_device between workflow runs - - Requires offload_device to be set - -- **encode_tiled**: Enable tiled encoding to reduce VRAM usage during encoding phase - - Enable if you see OOM errors during the "Encoding" phase in debug logs - -- **encode_tile_size**: Encoding tile size in pixels (default: 1024) - - Applied to both height and width - - Lower values reduce VRAM but may increase processing time - -- **encode_tile_overlap**: Encoding tile overlap in pixels (default: 128) - - Reduces visible seams between tiles - -- **decode_tiled**: Enable tiled decoding to reduce VRAM usage during decoding phase - - Enable if you see OOM errors during the "Decoding" phase in debug logs - -- **decode_tile_size**: Decoding tile size in pixels (default: 1024) - -- **decode_tile_overlap**: Decoding tile overlap in pixels (default: 128) - -- **torch_compile_args**: Connect to SeedVR2 Torch Compile Settings node for 15-25% speedup - -**VAE Tiling Explained:** - -VAE tiling processes large resolutions in smaller tiles to reduce VRAM requirements. Here's how to use it: - -1. **Run without tiling first** and monitor the debug logs (enable `enable_debug` on main node) -2. **If OOM during "Encoding" phase**: - - Enable `encode_tiled` - - If still OOM, reduce `encode_tile_size` (try 768, 512, etc.) -3. **If OOM during "Decoding" phase**: - - Enable `decode_tiled` - - If still OOM, reduce `decode_tile_size` -4. **Adjust overlap** (default 128) if you see visible seams in output (increase it) or processing times are too slow (decrease it). - -**Example Configuration for High Resolution (4K)**: -- encode_tiled: `True` -- encode_tile_size: `1024` -- encode_tile_overlap: `128` -- decode_tiled: `True` -- decode_tile_size: `1024` -- decode_tile_overlap: `128` - -#### 3. SeedVR2 Torch Compile Settings (Optional) - -![SeedVR2 Torch Compile Settings](docs/torch_compile_settings.png) - -Configure torch.compile optimization for 20-40% DiT speedup and 15-25% VAE speedup. - -**Requirements:** -- PyTorch 2.0+ -- Triton (for inductor backend) - -**Parameters:** - -- **backend**: Compilation backend - - `inductor`: Full optimization with Triton kernel generation and fusion (recommended) - - `cudagraphs`: Lightweight wrapper using CUDA graphs, no kernel optimization - -- **mode**: Optimization level (compilation time vs runtime performance) - - `default`: Fast compilation with good speedup (recommended for development) - - `reduce-overhead`: Lower overhead, optimized for smaller models - - `max-autotune`: Slowest compilation, best runtime performance (recommended for production) - - `max-autotune-no-cudagraphs`: Like max-autotune but without CUDA graphs - -- **fullgraph**: Compile entire model as single graph without breaks - - `False`: Allow graph breaks for better compatibility (default, recommended) - - `True`: Enforce no breaks for maximum optimization (may fail with dynamic shapes) - -- **dynamic**: Handle varying input shapes without recompilation - - `False`: Specialize for exact input shapes (default) - - `True`: Create dynamic kernels that adapt to shape variations (enable when processing different resolutions or batch sizes) - -- **dynamo_cache_size_limit**: Max cached compiled versions per function (default: 64) - - Higher = more memory, lower = more recompilation - -- **dynamo_recompile_limit**: Max recompilation attempts before falling back to eager mode (default: 128) - - Safety limit to prevent compilation loops - -**Usage:** -1. Add this node to your workflow -2. Connect its output to the `torch_compile_args` input of DiT and/or VAE loader nodes -3. First run will be slow (compilation), subsequent runs will be much faster - -**When to use:** -- torch.compile only makes sense when processing **multiple batches, long videos, or many tiles** -- For single images or short clips, the compilation time outweighs the speed improvement -- Best suited for batch processing workflows or long videos - -**Recommended Settings:** -- For development/testing: `mode=default`, `backend=inductor`, `fullgraph=False` -- For production: `mode=max-autotune`, `backend=inductor`, `fullgraph=False` - -#### 4. SeedVR2 Video Upscaler (Main Node) - -![SeedVR2 Video Upscaler](docs/video_upscaler.png) - -Main upscaling node that processes video frames using DiT and VAE models. - -**Required Inputs:** - -- **image**: Input video frames as image batch (RGB or RGBA format) -- **dit**: DiT model configuration from SeedVR2 (Down)Load DiT Model node -- **vae**: VAE model configuration from SeedVR2 (Down)Load VAE Model node - -**Parameters:** - -- **seed**: Random seed for reproducible generation (default: 42) - - Same seed with same inputs produces identical output - -- **resolution**: Target resolution for shortest edge in pixels (default: 1080) - - Maintains aspect ratio automatically - -- **max_resolution**: Maximum resolution for any edge (default: 0 = no limit) - - Automatically scales down if exceeded to prevent OOM - -- **batch_size**: Frames per batch (default: 5) - - **CRITICAL REQUIREMENT**: Must follow the **4n+1 formula** (1, 5, 9, 13, 17, 21, 25, ...) - - **Why this matters**: The model uses these frames for temporal consistency calculations - - **Minimum 5 for temporal consistency**: Use 1 only for single images or when temporal consistency isn't needed - - **Match shot length ideally**: For best results, set batch_size to match your shot length (e.g., batch_size=21 for a 20-frame shot) - - **VRAM impact**: Higher batch_size = better quality and speed but requires more VRAM - - **If you get OOM with batch_size=5**: Try optimization techniques first (model offloading, BlockSwap, GGUF models...) before reducing batch_size or input resolution, as these directly impact quality - -**uniform_batch_size** (default: False) - - Pads the final batch to match `batch_size` for uniform processing - - Prevents temporal artifacts when the last batch is significantly smaller than others - - Example: 45 frames with `batch_size=33` creates [33, 33] instead of [33, 12] - - Recommended when using large batch sizes and video length is not a multiple of `batch_size` - - Increases VRAM usage slightly but ensures consistent temporal coherence across all batches - -- **temporal_overlap**: Overlapping frames between batches (default: 0) - - Used for blending between batches to reduce temporal artifacts - - Range: 0-16 frames - -- **prepend_frames**: Frames to prepend (default: 0) - - Prepends reversed frames to reduce artifacts at video start - - Automatically removed after processing - - Range: 0-32 frames - -- **color_correction**: Color correction method (default: "wavelet") - - **`lab`**: Full perceptual color matching with detail preservation (recommended for highest fidelity to original) - - **`wavelet`**: Frequency-based natural colors, preserves details well - - **`wavelet_adaptive`**: Wavelet base + targeted saturation correction - - **`hsv`**: Hue-conditional saturation matching - - **`adain`**: Statistical style transfer - - **`none`**: No color correction - -- **input_noise_scale**: Input noise injection scale 0.0-1.0 (default: 0.0) - - Adds noise to input frames to reduce artifacts at very high resolutions - - Try 0.1-0.3 if you see artifacts with high output resolutions - -- **latent_noise_scale**: Latent space noise scale 0.0-1.0 (default: 0.0) - - Adds noise during diffusion process, can soften excessive detail - - Use if input_noise doesn't help, try 0.05-0.15 - -- **offload_device**: Device for storing intermediate tensors between processing phases (default: "cpu") - - `none`: Keep all tensors on inference device (fastest but highest VRAM) - - `cpu`: Offload to system RAM (recommended for long videos, slower transfers) - - `cuda:X`: Offload to another GPU (good balance if available, faster than CPU) - -- **enable_debug**: Enable detailed debug logging (default: False) - - Shows memory usage, timing information, and processing details - - **Highly recommended** for troubleshooting OOM issues - -**Output:** -- Upscaled video frames with color correction applied -- Format (RGB/RGBA) matches input -- Range [0, 1] normalized for ComfyUI compatibility - -### Typical Workflow Setup - -**Basic Workflow (High VRAM - 24GB+)**: -``` -Load Video Frames - โ†“ -SeedVR2 Load DiT Model - โ”œโ”€ model: seedvr2_ema_3b_fp16.safetensors - โ””โ”€ device: cuda:0 - โ†“ -SeedVR2 Load VAE Model - โ”œโ”€ model: ema_vae_fp16.safetensors - โ””โ”€ device: cuda:0 - โ†“ -SeedVR2 Video Upscaler - โ”œโ”€ batch_size: 21 - โ””โ”€ resolution: 1080 - โ†“ -Save Video/Frames -``` - -**Low VRAM Workflow (8-12GB)**: -``` -Load Video Frames - โ†“ -SeedVR2 Load DiT Model - โ”œโ”€ model: seedvr2_ema_3b-Q8_0.gguf - โ”œโ”€ device: cuda:0 - โ”œโ”€ offload_device: cpu - โ”œโ”€ blocks_to_swap: 32 - โ””โ”€ swap_io_components: True - โ†“ -SeedVR2 Load VAE Model - โ”œโ”€ model: ema_vae_fp16.safetensors - โ”œโ”€ device: cuda:0 - โ”œโ”€ encode_tiled: True - โ””โ”€ decode_tiled: True - โ†“ -SeedVR2 Video Upscaler - โ”œโ”€ batch_size: 5 - โ””โ”€ resolution: 720 - โ†“ -Save Video/Frames -``` - -**High Performance Workflow (24GB+ with torch.compile)**: -``` -Load Video Frames - โ†“ -SeedVR2 Torch Compile Settings - โ”œโ”€ mode: max-autotune - โ””โ”€ backend: inductor - โ†“ -SeedVR2 Load DiT Model - โ”œโ”€ model: seedvr2_ema_7b_sharp_fp16.safetensors - โ”œโ”€ device: cuda:0 - โ””โ”€ torch_compile_args: connected - โ†“ -SeedVR2 Load VAE Model - โ”œโ”€ model: ema_vae_fp16.safetensors - โ”œโ”€ device: cuda:0 - โ””โ”€ torch_compile_args: connected - โ†“ -SeedVR2 Video Upscaler - โ”œโ”€ batch_size: 81 - โ””โ”€ resolution: 1080 - โ†“ -Save Video/Frames -``` - -## ๐Ÿ–ฅ๏ธ Run as Standalone (CLI) - -The standalone CLI provides powerful batch processing capabilities with multi-GPU support and sophisticated optimization options. - -### Prerequisites - -Choose the appropriate setup based on your installation: - -#### Option 1: Already Have ComfyUI with SeedVR2 Installed - -If you've already installed SeedVR2 as part of ComfyUI (via [ComfyUI installation](#-installation)), you can use the CLI directly: - -```bash -# Navigate to your ComfyUI directory -cd ComfyUI - -# Run the CLI using standalone Python (display help message) -# Windows: -.venv\Scripts\python.exe custom_nodes\seedvr2_videoupscaler\inference_cli.py --help -# Linux/macOS: -.venv/bin/python custom_nodes/seedvr2_videoupscaler/inference_cli.py --help -``` - -**Skip to [Command Line Usage](#command-line-usage) below.** - -#### Option 2: Standalone Installation (Without ComfyUI) - -If you want to use the CLI without ComfyUI installation, follow these steps: - -1. **Install [uv](https://docs.astral.sh/uv/getting-started/installation/)** (modern Python package manager): -```bash -# Windows -powershell -ExecutionPolicy ByPass -c "irm https://astral.sh/uv/install.ps1 | iex" - -# macOS and Linux -curl -LsSf https://astral.sh/uv/install.sh | sh -``` - -2. **Clone the repository**: -```bash -git clone https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler.git seedvr2_videoupscaler -cd seedvr2_videoupscaler -``` - -3. **Create virtual environment and install dependencies**: -```bash -# Create virtual environment with Python 3.13 -uv venv --python 3.13 - -# Activate virtual environment -# Windows: -.venv\Scripts\activate -# Linux/macOS: -source .venv/bin/activate - -# Install PyTorch with CUDA support -# Check command line based on your environment: https://pytorch.org/get-started/locally/ -uv pip install --pre torch torchvision torchaudio --index-url https://download.pytorch.org/whl/nightly/cu130 - -# Install SeedVR2 requirements -uv pip install -r requirements.txt - -# Run the CLI (display help message) -# Windows: -.venv\Scripts\python.exe inference_cli.py --help -# Linux/macOS: -.venv/bin/python inference_cli.py --help -``` - -### Command Line Usage - -The CLI provides comprehensive options for single-GPU, multi-GPU, and batch processing workflows. - -**Basic Usage Examples:** - -```bash -# Basic image upscaling -python inference_cli.py image.jpg - -# Basic video video upscaling with temporal consistency -python inference_cli.py video.mp4 --resolution 720 --batch_size 33 - -# Multi-GPU processing with temporal overlap -python inference_cli.py video.mp4 \ - --cuda_device 0,1 \ - --resolution 1080 \ - --batch_size 81 \ - --uniform_batch_size \ - --temporal_overlap 3 \ - --prepend_frames 4 - -# Memory-optimized for low VRAM (8GB) -python inference_cli.py image.png \ - --dit_model seedvr2_ema_3b-Q8_0.gguf \ - --resolution 1080 \ - --blocks_to_swap 32 \ - --swap_io_components \ - --dit_offload_device cpu \ - --vae_offload_device cpu - -# High resolution with VAE tiling -python inference_cli.py video.mp4 \ - --resolution 1440 \ - --batch_size 31 \ - --uniform_batch_size \ - --temporal_overlap 3 \ - --vae_encode_tiled \ - --vae_decode_tiled - -# Batch directory processing with model caching -python inference_cli.py media_folder/ \ - --output processed/ \ - --cuda_device 0 \ - --cache_dit \ - --cache_vae \ - --dit_offload_device cpu \ - --vae_offload_device cpu \ - --resolution 1080 \ - --max_resolution 1920 -``` - -### Command Line Arguments - -**Input/Output:** -- ``: Input file (.mp4, .avi, .png, .jpg, etc.) or directory -- `--output`: Output path (default: auto-generated in 'output/' directory) -- `--output_format`: Output format: 'mp4' (video) or 'png' (image sequence). Default: auto-detect from input type -- `--model_dir`: Model directory (default: ./models/SEEDVR2) - -**Model Selection:** -- `--dit_model`: DiT model to use. Options: 3B/7B with fp16/fp8/GGUF variants (default: 3B FP8) - -**Processing Parameters:** -- `--resolution`: Target short-side resolution in pixels (default: 1080) -- `--max_resolution`: Maximum resolution for any edge. Scales down if exceeded. 0 = no limit (default: 0) -- `--batch_size`: Frames per batch (must follow 4n+1: 1, 5, 9, 13, 17, 21...). Ideally matches shot length for best temporal consistency (default: 5) -- `--seed`: Random seed for reproducibility (default: 42) -- `--skip_first_frames`: Skip N initial frames (default: 0) -- `--load_cap`: Load maximum N frames from video. 0 = load all (default: 0) -- `--prepend_frames`: Prepend N reversed frames to reduce start artifacts (auto-removed) (default: 0) -- `--temporal_overlap`: Frames to overlap between batches/GPUs for smooth blending (default: 0) - -**Quality Control:** -- `--color_correction`: Color correction method: 'lab' (perceptual, recommended), 'wavelet', 'wavelet_adaptive', 'hsv', 'adain', or 'none' (default: lab) -- `--input_noise_scale`: Input noise injection scale (0.0-1.0). Reduces artifacts at high resolutions (default: 0.0) -- `--latent_noise_scale`: Latent space noise scale (0.0-1.0). Softens details if needed (default: 0.0) - -**Memory Management:** -- `--dit_offload_device`: Device to offload DiT model: 'none' (keep on GPU), 'cpu', or 'cuda:X' (default: none) -- `--vae_offload_device`: Device to offload VAE model: 'none', 'cpu', or 'cuda:X' (default: none) -- `--blocks_to_swap`: Number of transformer blocks to swap (0=disabled, 3B: 0-32, 7B: 0-36). Requires dit_offload_device (default: 0) -- `--swap_io_components`: Offload I/O components for additional VRAM savings. Requires dit_offload_device -- `--use_non_blocking`: Use non-blocking memory transfers for BlockSwap (recommended) - -**VAE Tiling:** -- `--vae_encode_tiled`: Enable VAE encode tiling to reduce VRAM during encoding -- `--vae_encode_tile_size`: VAE encode tile size in pixels (default: 1024) -- `--vae_encode_tile_overlap`: VAE encode tile overlap in pixels (default: 128) -- `--vae_decode_tiled`: Enable VAE decode tiling to reduce VRAM during decoding -- `--vae_decode_tile_size`: VAE decode tile size in pixels (default: 1024) -- `--vae_decode_tile_overlap`: VAE decode tile overlap in pixels (default: 128) -- `--tile_debug`: Visualize tiles: 'false' (default), 'encode', or 'decode' - -**Performance Optimization:** -- `--attention_mode`: Attention backend: 'sdpa' (default, stable) or 'flash_attn' (faster, requires package) -- `--compile_dit`: Enable torch.compile for DiT model (20-40% speedup, requires PyTorch 2.0+ and Triton) -- `--compile_vae`: Enable torch.compile for VAE model (15-25% speedup, requires PyTorch 2.0+ and Triton) -- `--compile_backend`: Compilation backend: 'inductor' (full optimization) or 'cudagraphs' (lightweight) (default: inductor) -- `--compile_mode`: Optimization level: 'default', 'reduce-overhead', 'max-autotune', 'max-autotune-no-cudagraphs' (default: default) -- `--compile_fullgraph`: Compile entire model as single graph (faster but less flexible) (default: False) -- `--compile_dynamic`: Handle varying input shapes without recompilation (default: False) -- `--compile_dynamo_cache_size_limit`: Max cached compiled versions per function (default: 64) -- `--compile_dynamo_recompile_limit`: Max recompilation attempts before fallback (default: 128) - -**Model Caching (batch processing):** -- `--cache_dit`: Cache DiT model between files (single GPU only, speeds up directory processing) -- `--cache_vae`: Cache VAE model between files (single GPU only, speeds up directory processing) - -**Multi-GPU:** -- `--cuda_device`: CUDA device id(s). Single id (e.g., '0') or comma-separated list '0,1' for multi-GPU - -**Debugging:** -- `--debug`: Enable verbose debug logging - -### Multi-GPU Processing Explained - -The CLI's multi-GPU mode automatically distributes the workload across multiple GPUs with intelligent temporal overlap handling: - -**How it works:** -1. Video is split into chunks, one per GPU -2. Each GPU processes its chunk independently -3. Chunks overlap by `--temporal_overlap` frames -4. Results are blended together seamlessly using the overlap region - -**Example for 2 GPUs with temporal_overlap=4:** -``` -GPU 0: Frames 0-50 (includes 4 overlap frames at end) -GPU 1: Frames 46-100 (includes 4 overlap frames at beginning) -Result: Frames 0-100 with smooth transition at frame 48 -``` - -**Best practices:** -- Set `--temporal_overlap` to 2-8 frames for smooth blending -- Higher overlap = smoother transitions but more redundant processing -- Use `--prepend_frames` to reduce artifacts at video start -- batch_size should divide evenly into chunk sizes for best results - -## โš ๏ธ Limitations - -### Model Limitations - -**Batch Size Constraint**: The model requires batch_size to follow the **4n+1 formula** (1, 5, 9, 13, 17, 21, 25, ...) due to temporal consistency architecture. All frames in a batch are processed together for temporal coherence, then batches can be blended using temporal_overlap. Ideally, set batch_size to match your shot length for optimal quality. - -### Performance Considerations - -**VAE Bottleneck**: Even with optimized DiT upscaling (BlockSwap, GGUF, torch.compile), the VAE encoding/decoding stages can be the bottleneck, especially for high resolutions. The VAE is slow. Use large batch_size to mitigate this. - -**VRAM Usage**: While the integration now supports low VRAM systems (8GB or less with proper optimization), VRAM usage varies based on: -- Input/output resolution (larger = more VRAM) -- Batch size (higher = more VRAM but better temporal consistency and speed) -- Model choice (FP16 > FP8 > GGUF in VRAM usage) -- Optimization settings (BlockSwap, VAE tiling significantly reduce VRAM) - -**Speed**: Processing speed depends on: -- GPU capabilities (compute performance, VRAM bandwidth, and architecture generation) -- Model size (3B faster than 7B) -- Batch size (larger batch sizes are faster per frame due to better GPU utilization) -- Optimization settings (torch.compile provides significant speedup) -- Resolution (higher resolutions are slower) - -### Best Practices - -1. **Start with debug enabled** to understand where VRAM is being used -2. **For OOM errors during encoding**: Enable VAE encode tiling and reduce tile size -3. **For OOM errors during upscaling**: Enable BlockSwap and increase blocks_to_swap -4. **For OOM errors during decoding**: Enable VAE decode tiling and reduce tile size - - **If still getting OOM after trying all above**: Reduce batch_size or resolution -5. **For best quality**: Use higher batch_size matching your shot length, FP16 models, and LAB color correction -6. **For speed**: Use FP8/GGUF models, enable torch.compile, and use Flash Attention if available -7. **Test settings with a short clip first** before processing long videos - -## ๐Ÿค Contributing - -Contributions are welcome! We value community input and improvements. - -For detailed contribution guidelines, see [CONTRIBUTING.md](CONTRIBUTING.md). - -**Quick Start:** - -1. Fork the repository -2. Create your feature branch (`git checkout -b feature/AmazingFeature`) -3. Commit your changes (`git commit -m 'Add some AmazingFeature'`) -4. Push to the branch (`git push origin feature/AmazingFeature`) -5. Open a Pull Request to **main** branch for stable features or **nightly** branch for experimental features - -**Get Help:** -- YouTube: [AInVFX Channel](https://www.youtube.com/@AInVFX) -- GitHub [Issues](https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler/issues): For bug reports and feature requests -- GitHub [Discussions](https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler/discussions): For questions and community support -- Discord: adrientoupet & NumZ#7184 - -## ๐Ÿ™ Credits - -This ComfyUI implementation is a collaborative project by **[NumZ](https://github.com/numz)** and **[AInVFX](https://www.youtube.com/@AInVFX)** (Adrien Toupet), based on the original [SeedVR2](https://github.com/ByteDance-Seed/SeedVR) by ByteDance Seed Team. - -Special thanks to our community contributors including [benjaminherb](https://github.com/benjaminherb), [cmeka](https://github.com/cmeka), [FurkanGozukara](https://github.com/FurkanGozukara), [JohnAlcatraz](https://github.com/JohnAlcatraz), [lihaoyun6](https://github.com/lihaoyun6), [Luchuanzhao](https://github.com/Luchuanzhao), [Luke2642](https://github.com/Luke2642), [naxci1](https://github.com/naxci1), [q5sys](https://github.com/q5sys), and many others for their improvements, bug fixes, and testing. - -## ๐Ÿ“œ License - -The code in this repository is released under the MIT license as found in the [LICENSE](LICENSE) file. \ No newline at end of file +For installation details and usage instructions, see the main branch README. \ No newline at end of file diff --git a/inference_cli.py b/inference_cli.py index 3abefcf..c4bc2ec 100644 --- a/inference_cli.py +++ b/inference_cli.py @@ -44,1353 +44,15 @@ Model Support: # Standard library imports import sys import os -import argparse -import time -import platform -import multiprocessing as mp -from typing import Dict, Any, List, Optional, Tuple, Literal -from datetime import datetime -from pathlib import Path - -# Set up path before any other imports to fix module resolution -script_dir = os.path.dirname(os.path.abspath(__file__)) -if script_dir not in sys.path: - sys.path.insert(0, script_dir) - -# Set environment variable so all spawned processes can find modules -os.environ['PYTHONPATH'] = script_dir + ':' + os.environ.get('PYTHONPATH', '') - -# Ensure safe CUDA usage with multiprocessing -if mp.get_start_method(allow_none=True) != 'spawn': - mp.set_start_method('spawn', force=True) - -# Configure VRAM management and validate CUDA devices before heavy imports -if platform.system() != "Darwin": - os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync") - - # Pre-parse CUDA device argument for validation and environment setup - _pre_parser = argparse.ArgumentParser(add_help=False) - _pre_parser.add_argument("--cuda_device", type=str, default=None) - _pre_args, _ = _pre_parser.parse_known_args() - - if _pre_args.cuda_device is not None: - device_list_env = [x.strip() for x in _pre_args.cuda_device.split(',') if x.strip()!=''] - - # Temporary torch import for CUDA device validation only - # Must happen before setting CUDA_VISIBLE_DEVICES and before main torch import - import torch as _torch_check - if _torch_check.cuda.is_available(): - available_count = _torch_check.cuda.device_count() - invalid_devices = [d for d in device_list_env if not d.isdigit() or int(d) >= available_count] - if invalid_devices: - print(f"โŒ [ERROR] Invalid CUDA device ID(s): {', '.join(invalid_devices)}. " - f"Available devices: 0-{available_count-1} (total: {available_count})") - sys.exit(1) - else: - print("โŒ [ERROR] CUDA is not available on this system. Cannot use --cuda_device argument.") - sys.exit(1) - - # Set CUDA_VISIBLE_DEVICES for single GPU after validation - if len(device_list_env) == 1: - os.environ["CUDA_VISIBLE_DEVICES"] = device_list_env[0] - -# Heavy dependency imports after environment configuration -import torch -import cv2 -import numpy as np - -# Project imports -from src.utils.downloads import download_weight -from src.utils.model_registry import get_available_dit_models, DEFAULT_DIT, DEFAULT_VAE -from src.utils.constants import SEEDVR2_FOLDER_NAME -from src.core.generation_utils import ( - setup_generation_context, - prepare_runner, - compute_generation_info, - log_generation_start, - blend_overlapping_frames -) -from src.core.generation_phases import ( - encode_all_batches, - upscale_all_batches, - decode_all_batches, - postprocess_all_batches -) -from src.utils.debug import Debug -debug = Debug(enabled=False) # Will be enabled via --debug CLI flag - # ============================================================================= -# Device Management Helpers +# DEPRECATION NOTICE - This branch is no longer supported # ============================================================================= - -def _get_platform_type() -> str: - """Determine the platform device type (cuda/mps/cpu).""" - if platform.system() == "Darwin": - return "mps" - elif torch.cuda.is_available(): - return "cuda" - else: - return "cpu" - - -def _device_id_to_name(device_id: str, platform_type: str = None) -> str: - """ - Convert device ID to full device name. - - Args: - device_id: Device ID ("0", "1") or special value ("cpu", "none") - platform_type: Override platform type ("cuda", "mps", "cpu") - - Returns: - Full device name ("cuda:0", "mps:0", "cpu", "none") - """ - if device_id in ("cpu", "none"): - return device_id - - if platform_type is None: - platform_type = _get_platform_type() - - # MPS typically doesn't use indices - if platform_type == "mps": - return "mps" - - return f"{platform_type}:{device_id}" - - -def _parse_offload_device(offload_arg: str, platform_type: str = None, cache_enabled: bool = False) -> Optional[str]: - """ - Parse offload device argument to full device name. - - Args: - offload_arg: Offload device argument ("none", "cpu", "0", "1", or "cuda:1") - platform_type: Override platform type - cache_enabled: If True and offload_arg is "none", default to "cpu" - - Returns: - Full device name or None - """ - if offload_arg == "none": - # If caching enabled but no offload device specified, default to CPU - return "cpu" if cache_enabled else None - - if offload_arg == "cpu": - return "cpu" - - # If already a full device name (cuda:1, mps:0), return as-is - if ":" in offload_arg: - return offload_arg - - # Otherwise treat as device ID - return _device_id_to_name(offload_arg, platform_type) - - -# ============================================================================= -# Constants -# ============================================================================= - -# Supported file extensions -VIDEO_EXTENSIONS = {'.mp4', '.avi', '.mov', '.mkv', '.webm', '.flv', '.wmv', '.m4v'} -IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.bmp', '.tiff', '.tif', '.webp'} - - -# ============================================================================= -# Video I/O Functions -# ============================================================================= - -def get_media_files(directory: str) -> List[str]: - """ - Get all video and image files from directory, sorted alphabetically. - - Args: - directory: Path to directory to scan - - Returns: - Sorted list of file paths (strings) matching video or image extensions - """ - files = [] - for ext in VIDEO_EXTENSIONS | IMAGE_EXTENSIONS: - files.extend(Path(directory).glob(f'*{ext}')) - files.extend(Path(directory).glob(f'*{ext.upper()}')) - return sorted([str(f) for f in files]) - - -def extract_frames_from_image(image_path: str) -> Tuple[torch.Tensor, float]: - """ - Extract single frame from image file and convert to tensor format. - - Reads image using OpenCV, converts BGR to RGB, normalizes to [0,1] range, - and formats as single-frame video tensor for consistent processing. - - Args: - image_path: Path to input image file - - Returns: - Tuple containing: - - frames_tensor: Single frame as tensor [1, H, W, C], Float16, range [0,1] - - fps: Default FPS value (30.0) for image-to-video conversion - - Raises: - FileNotFoundError: If image file doesn't exist - ValueError: If image cannot be opened - """ - debug.log(f"Loading image: {image_path}", category="file") - - if not os.path.exists(image_path): - raise FileNotFoundError(f"Image file not found: {image_path}") - - # Read image - frame = cv2.imread(image_path) - if frame is None: - raise ValueError(f"Cannot open image file: {image_path}") - - # Convert BGR to RGB - frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) - - # Convert to float32 and normalize - frame = frame.astype(np.float32) / 255.0 - - # Convert to tensor [1, H, W, C] - frames_tensor = torch.from_numpy(frame[None, ...]).to(torch.float16) - - debug.log(f"Image tensor shape: {frames_tensor.shape}, dtype: {frames_tensor.dtype}", category="memory") - - return frames_tensor, 30.0 # Default FPS for images - - -def get_input_type(input_path: str) -> Literal['video', 'image', 'directory', 'unknown']: - """ - Determine input type from file path. - - Args: - input_path: Path to input file or directory - - Returns: - Input type: 'video', 'image', 'directory', or 'unknown' - - Raises: - FileNotFoundError: If input path doesn't exist - """ - path = Path(input_path) - - if not path.exists(): - raise FileNotFoundError(f"Input path not found: {input_path}") - - if path.is_dir(): - return 'directory' - - ext = path.suffix.lower() - if ext in VIDEO_EXTENSIONS: - return "video" - elif ext in IMAGE_EXTENSIONS: - return "image" - else: - return "unknown" - - -def generate_output_path(input_path: str, output_format: str, output_dir: Optional[str] = None, - input_type: Optional[str] = None) -> str: - """ - Generate output path based on input path and format. - - Args: - input_path: Source file path - output_format: "mp4" or "png" - output_dir: Optional output directory - input_type: Optional input type ("image", "video", "directory") - - Returns: - Output path (file for single image/video, directory for sequences) - """ - input_name = Path(input_path).stem - - if output_format == "png": - # Single image โ†’ single PNG file - if input_type == "image": - if output_dir: - return str(Path(output_dir) / f"{input_name}_upscaled.png") - return f"output/{input_name}_upscaled.png" - # Video/sequence โ†’ directory of numbered PNGs - else: - if output_dir: - return str(Path(output_dir) / f"{input_name}_upscaled") - return f"output/{input_name}_upscaled" - else: - # Video format always returns file path - if output_dir: - return str(Path(output_dir) / f"{input_name}_upscaled.mp4") - return f"output/{input_name}_upscaled.mp4" - - -def process_single_file(input_path: str, args: argparse.Namespace, device_list: List[str], - output_path: Optional[str] = None, format_auto_detected: bool = False, - runner_cache: Optional[Dict[str, Any]] = None) -> int: - """ - Process a single video or image file with optional model caching. - - Args: - input_path: Path to input file - args: Command-line arguments with all processing settings - device_list: List of GPU device IDs as strings - output_path: Optional explicit output path (auto-generated if None) - format_auto_detected: Whether output format was auto-detected - runner_cache: Optional cache dict for model reuse across multiple files - - Returns: - Number of frames processed from the input - """ - input_type = get_input_type(input_path) - - if input_type == "unknown": - debug.log(f"Skipping unsupported file: {input_path}", level="WARNING", category="file", force=True) - return 0 - - debug.log(f"Processing {input_type}: {Path(input_path).name}", category="generation", force=True) - - # Extract frames - if input_type == "video": - start_time = time.time() - frames_tensor, original_fps = extract_frames_from_video( - input_path, args.skip_first_frames, args.load_cap - ) - debug.log(f"Frame extraction time: {time.time() - start_time:.2f}s", category="timing") - else: - frames_tensor, original_fps = extract_frames_from_image(input_path) - - # Track frames before processing (for FPS calculation) - input_frame_count = len(frames_tensor) - - # Generate or validate output path - if output_path is None: - output_path = generate_output_path(input_path, args.output_format, input_type=input_type) - elif not Path(output_path).suffix or (args.output_format == "png" and input_type != "image"): - # No extension or PNG sequence โ†’ treat as directory, generate filename - output_path = generate_output_path(input_path, args.output_format, - output_dir=output_path, input_type=input_type) - - # Show format with auto-detection indicator - format_prefix = "Auto-detected" if format_auto_detected else "Requested" - debug.log(f"{format_prefix} output format: {args.output_format}", category="info", force=True, indent_level=1) - - # Process frames - processing_start = time.time() - # Use direct processing if caching enabled - if runner_cache is not None: - # Direct single-GPU processing with model caching - result = _single_gpu_direct_processing(frames_tensor, args, device_list[0], runner_cache) - else: - # Multi-GPU or non-cached processing via worker processes - result = _gpu_processing(frames_tensor, device_list, args) - debug.log(f"Processing time: {time.time() - processing_start:.2f}s", category="timing") - - # Save results - is_png_format = args.output_format == "png" - is_single_image = input_type == "image" - - if is_png_format and is_single_image: - # Single PNG file - os.makedirs(Path(output_path).parent, exist_ok=True) - frame_np = (result[0].cpu().numpy() * 255.0).astype(np.uint8) - frame_bgr = cv2.cvtColor(frame_np, cv2.COLOR_RGB2BGR) - cv2.imwrite(output_path, frame_bgr) - - elif is_png_format: - # PNG sequence (save_frames_to_png creates directory internally) - save_frames_to_png(result, output_path, base_name=Path(input_path).stem) - - else: - # Video file - os.makedirs(Path(output_path).parent, exist_ok=True) - save_frames_to_video(result, output_path, original_fps) - - # Log appropriate save message based on format - if is_png_format and not is_single_image: - debug.log(f"PNG frames saved in directory: {output_path}", category="file", force=True) - else: - debug.log(f"Output saved to: {output_path}", category="file", force=True) - - return input_frame_count - - -def extract_frames_from_video( - video_path: str, - skip_first_frames: int = 0, - load_cap: Optional[int] = None -) -> Tuple[torch.Tensor, float]: - """ - Extract frames from video file and convert to tensor format. - - Reads video using OpenCV, converts BGR to RGB, normalizes to [0,1] range. - Note: Frame prepending is handled later in the processing pipeline via - compute_generation_info(), not in this function. - - Args: - video_path: Path to input video file - skip_first_frames: Number of initial frames to skip (default: 0) - load_cap: Maximum number of frames to load, None loads all (default: None) - - Returns: - Tuple containing: - - frames_tensor: Frames in format [T, H, W, C], Float32, range [0,1] - - fps: Original video frames per second - - Raises: - FileNotFoundError: If video file doesn't exist - ValueError: If video cannot be opened or no frames extracted - """ - debug.log(f"Extracting frames from video: {video_path}", category="file") - - if not os.path.exists(video_path): - raise FileNotFoundError(f"Video file not found: {video_path}") - - # Open video - cap = cv2.VideoCapture(video_path) - if not cap.isOpened(): - raise ValueError(f"Cannot open video file: {video_path}") - - # Get video properties - fps = cap.get(cv2.CAP_PROP_FPS) - frame_count = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) - width = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH)) - height = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT)) - - debug.log(f"Video info: {frame_count} frames, {width}x{height}, {fps:.2f} FPS", category="info") - if skip_first_frames: - debug.log(f"Will skip first {skip_first_frames} frames", category="info") - if load_cap: - debug.log(f"Will load maximum {load_cap} frames", category="info") - - frames = [] - frame_idx = 0 - frames_loaded = 0 - - while True: - ret, frame = cap.read() - if not ret: - break - - # Skip first frame if requested - if frame_idx < skip_first_frames: - frame_idx += 1 - continue - - if skip_first_frames > 0 and frame_idx == skip_first_frames: - debug.log(f"Skipped first {skip_first_frames} frames", category="info") - - # Check load cap - if load_cap is not None and load_cap > 0 and frames_loaded >= load_cap: - debug.log(f"Reached load cap of {load_cap} frames", category="info") - break - - # Convert BGR to RGB - frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) - - # Convert to float32 and normalize to 0-1 - frame = frame.astype(np.float32) / 255.0 - - frames.append(frame) - frame_idx += 1 - frames_loaded += 1 - - if debug.enabled and frames_loaded % 100 == 0: - total_to_load = min(frame_count, load_cap) if load_cap else frame_count - debug.log(f"Extracted {frames_loaded}/{total_to_load} frames", category="file") - - cap.release() - - if len(frames) == 0: - raise ValueError(f"No frames extracted from video: {video_path}") - - debug.log(f"Extracted {len(frames)} frames", category="success") - - # Convert to tensor (will be cast to compute_dtype in worker process) - frames_tensor = torch.from_numpy(np.stack(frames)).to(torch.float32) - - debug.log(f"Frames tensor shape: {frames_tensor.shape}, dtype: {frames_tensor.dtype}", category="memory") - - return frames_tensor, fps - - -def save_frames_to_video( - frames_tensor: torch.Tensor, - output_path: str, - fps: float = 30.0 -) -> None: - """ - Save frames tensor to MP4 video file. - - Converts tensor from Float32 [0,1] to uint8 [0,255], RGB to BGR for OpenCV, - and writes to video file using mp4v codec. - - Args: - frames_tensor: Frames in format [T, H, W, C], Float32, range [0,1] - output_path: Output video file path (will be created if doesn't exist) - fps: Frames per second for output video (default: 30.0) - - Raises: - ValueError: If video writer cannot be initialized - """ - debug.log(f"Saving {frames_tensor.shape[0]} frames to video: {output_path}", category="file") - - # Convert tensor to numpy and denormalize - frames_np = frames_tensor.cpu().numpy() - frames_np = (frames_np * 255.0).astype(np.uint8) - - # Get video properties - T, H, W, C = frames_np.shape - - # Initialize video writer - fourcc = cv2.VideoWriter_fourcc(*'mp4v') - out = cv2.VideoWriter(output_path, fourcc, fps, (W, H)) - - if not out.isOpened(): - raise ValueError(f"Cannot create video writer for: {output_path}") - - # Write frames - for i, frame in enumerate(frames_np): - # Convert RGB to BGR for OpenCV - frame_bgr = cv2.cvtColor(frame, cv2.COLOR_RGB2BGR) - out.write(frame_bgr) - - if debug.enabled and (i + 1) % 100 == 0: - debug.log(f"Saved {i + 1}/{T} frames", category="file") - - out.release() - - debug.log(f"Video saved successfully: {output_path}", category="success") - - -def save_frames_to_png( - frames_tensor: torch.Tensor, - output_dir: str, - base_name: str -) -> None: - """ - Save frames tensor as sequential PNG image files. - - Each frame saved as {base_name}_{index:05d}.png with zero-padded indices. - Converts Float32 [0,1] to uint8 [0,255] and RGB to BGR for OpenCV. - - Args: - frames_tensor: Frames in format [T, H, W, C], Float32, range [0,1] - output_dir: Directory to save PNG files (created if doesn't exist) - base_name: Base name for output files (e.g., "frame" โ†’ "frame_00000.png") - """ - debug.log(f"Saving {frames_tensor.shape[0]} frames as PNGs to directory: {output_dir}", category="file") - - # Ensure output directory exists - os.makedirs(output_dir, exist_ok=True) - - # Convert to numpy uint8 RGB - frames_np = (frames_tensor.cpu().numpy() * 255.0).astype(np.uint8) - total = frames_np.shape[0] - digits = max(5, len(str(total))) # at least 5 digits - - for idx, frame in enumerate(frames_np): - filename = f"{base_name}_{idx:0{digits}d}.png" - file_path = os.path.join(output_dir, filename) - # Convert RGB to BGR for cv2 - frame_bgr = cv2.cvtColor(frame, cv2.COLOR_RGB2BGR) - cv2.imwrite(file_path, frame_bgr) - if debug.enabled and (idx + 1) % 100 == 0: - debug.log(f"Saved {idx + 1}/{total} PNGs", category="file") - - debug.log(f"PNG saving completed: {total} files in '{output_dir}'", category="success") - - -# ============================================================================= -# Core Processing Logic -# ============================================================================= - -def _process_frames_core( - frames_tensor: torch.Tensor, - args: argparse.Namespace, - device_id: str, - debug: Debug, - runner_cache: Optional[Dict[str, Any]] = None -) -> torch.Tensor: - """ - Core frame processing logic shared between worker and direct processing. - - Executes the complete 4-phase pipeline: encode โ†’ upscale โ†’ decode โ†’ postprocess. - Supports both cached (direct) and non-cached (worker) execution modes. - - Args: - frames_tensor: Input frames [T, H, W, C], Float16/Float32, range [0,1] - args: Command-line arguments with all processing settings - device_id: Device ID for inference ("0", "1", etc.) - debug: Debug instance for logging - runner_cache: Optional cache dict for model reuse (direct mode only) - - Returns: - Upscaled frames tensor [T', H', W', C], Float32, range [0,1] - """ - # Determine platform and convert device IDs to full names - platform_type = _get_platform_type() - inference_device = _device_id_to_name(device_id, platform_type) - - # Parse offload devices (with caching defaults) - cache_dit = args.cache_dit if runner_cache is not None else False - cache_vae = args.cache_vae if runner_cache is not None else False - - dit_offload = _parse_offload_device(args.dit_offload_device, platform_type, cache_dit) - vae_offload = _parse_offload_device(args.vae_offload_device, platform_type, cache_vae) - tensor_offload = _parse_offload_device(args.tensor_offload_device, platform_type, False) - - # Setup or reuse generation context - if runner_cache is not None and 'ctx' in runner_cache: - ctx = runner_cache['ctx'] - # Clear previous run data but keep device config - keys_to_keep = {'dit_device', 'vae_device', 'dit_offload_device', - 'vae_offload_device', 'tensor_offload_device', 'compute_dtype'} - for key in list(ctx.keys()): - if key not in keys_to_keep: - del ctx[key] - else: - ctx = setup_generation_context( - dit_device=inference_device, - vae_device=inference_device, - dit_offload_device=dit_offload, - vae_offload_device=vae_offload, - tensor_offload_device=tensor_offload, - debug=debug - ) - if runner_cache is not None: - runner_cache['ctx'] = ctx - - # Build torch compile args - torch_compile_args_dit = None - torch_compile_args_vae = None - if args.compile_dit: - torch_compile_args_dit = { - "backend": args.compile_backend, - "mode": args.compile_mode, - "fullgraph": args.compile_fullgraph, - "dynamic": args.compile_dynamic, - "dynamo_cache_size_limit": args.compile_dynamo_cache_size_limit, - "dynamo_recompile_limit": args.compile_dynamo_recompile_limit, - } - if args.compile_vae: - torch_compile_args_vae = { - "backend": args.compile_backend, - "mode": args.compile_mode, - "fullgraph": args.compile_fullgraph, - "dynamic": args.compile_dynamic, - "dynamo_cache_size_limit": args.compile_dynamo_cache_size_limit, - "dynamo_recompile_limit": args.compile_dynamo_recompile_limit, - } - - # Prepare runner with caching support - model_dir = args.model_dir if args.model_dir is not None else f"./models/{SEEDVR2_FOLDER_NAME}" - - # Use fixed IDs for CLI caching when enabled - dit_id = "cli_dit" if cache_dit else None - vae_id = "cli_vae" if cache_vae else None - - runner, cache_context = prepare_runner( - dit_model=args.dit_model, - vae_model=DEFAULT_VAE, - model_dir=model_dir, - debug=debug, - ctx=ctx, - dit_cache=cache_dit, - vae_cache=cache_vae, - dit_id=dit_id, - vae_id=vae_id, - block_swap_config={ - 'blocks_to_swap': args.blocks_to_swap, - 'swap_io_components': args.swap_io_components, - 'offload_device': dit_offload, - }, - encode_tiled=args.vae_encode_tiled, - encode_tile_size=(args.vae_encode_tile_size, args.vae_encode_tile_size), - encode_tile_overlap=(args.vae_encode_tile_overlap, args.vae_encode_tile_overlap), - decode_tiled=args.vae_decode_tiled, - decode_tile_size=(args.vae_decode_tile_size, args.vae_decode_tile_size), - decode_tile_overlap=(args.vae_decode_tile_overlap, args.vae_decode_tile_overlap), - tile_debug=args.tile_debug.lower() if args.tile_debug else "false", - attention_mode=args.attention_mode, - torch_compile_args_dit=torch_compile_args_dit, - torch_compile_args_vae=torch_compile_args_vae - ) - - ctx['cache_context'] = cache_context - if runner_cache is not None: - runner_cache['runner'] = runner - - # Compute generation info and log start (handles prepending internally) - frames_tensor, gen_info = compute_generation_info( - ctx=ctx, - images=frames_tensor, - resolution=args.resolution, - max_resolution=args.max_resolution, - batch_size=args.batch_size, - uniform_batch_size=args.uniform_batch_size, - seed=args.seed, - prepend_frames=args.prepend_frames, - temporal_overlap=args.temporal_overlap, - debug=debug - ) - log_generation_start(gen_info, debug) - - # Phase 1: Encode - ctx = encode_all_batches( - runner, ctx=ctx, images=frames_tensor, - debug=debug, - batch_size=args.batch_size, - uniform_batch_size=args.uniform_batch_size, - seed=args.seed, - progress_callback=None, - temporal_overlap=args.temporal_overlap, - resolution=args.resolution, - max_resolution=args.max_resolution, - input_noise_scale=args.input_noise_scale, - color_correction=args.color_correction - ) - - # Phase 2: Upscale - ctx = upscale_all_batches( - runner, ctx=ctx, debug=debug, progress_callback=None, - seed=args.seed, - latent_noise_scale=args.latent_noise_scale, - cache_model=cache_dit - ) - - # Phase 3: Decode - ctx = decode_all_batches( - runner, ctx=ctx, debug=debug, progress_callback=None, - cache_model=cache_vae - ) - - # Phase 4: Post-process - ctx = postprocess_all_batches( - ctx=ctx, debug=debug, progress_callback=None, - color_correction=args.color_correction, - prepend_frames=0, # Worker mode handles this in main process - temporal_overlap=args.temporal_overlap, - batch_size=args.batch_size - ) - - result_tensor = ctx['final_video'] - - # Convert to CPU and compatible dtype - if result_tensor.is_cuda or result_tensor.is_mps: - result_tensor = result_tensor.cpu() - if result_tensor.dtype in (torch.bfloat16, torch.float8_e4m3fn, torch.float8_e5m2): - result_tensor = result_tensor.to(torch.float32) - - return result_tensor - - -def _worker_process( - proc_idx: int, - device_id: str, - frames_np: np.ndarray, - shared_args: Dict[str, Any], - return_queue: mp.Queue -) -> None: - """ - Worker process for multi-GPU upscaling. - - Sets up isolated CUDA environment and calls core processing logic. - Results returned via multiprocessing queue as numpy arrays. - """ - if platform.system() != "Darwin": - # Limit CUDA visibility to the chosen GPU BEFORE importing torch - os.environ["CUDA_VISIBLE_DEVICES"] = device_id - os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync") - - import torch - - # Create debug instance for this worker - worker_debug = Debug(enabled=shared_args["debug"]) - - # Convert numpy back to tensor - frames_tensor = torch.from_numpy(frames_np).to(torch.float16) - - # Create args namespace from shared_args - args = argparse.Namespace(**shared_args) - - # Process frames (no caching in worker mode) - result_tensor = _process_frames_core( - frames_tensor=frames_tensor, - args=args, - device_id="0", # Always "0" in worker (CUDA_VISIBLE_DEVICES set) - debug=worker_debug, - runner_cache=None # No caching in multiprocessing mode - ) - - # Send back result as numpy array - return_queue.put((proc_idx, result_tensor.numpy())) - - -def _single_gpu_direct_processing( - frames_tensor: torch.Tensor, - args: argparse.Namespace, - device_id: str, - runner_cache: Dict[str, Any] -) -> torch.Tensor: - """ - Direct single-GPU processing with model caching support. - - Uses main process and shared runner cache for efficient multi-file processing. - """ - return _process_frames_core( - frames_tensor=frames_tensor, - args=args, - device_id=device_id, - debug=debug, - runner_cache=runner_cache - ) - - -def _gpu_processing( - frames_tensor: torch.Tensor, - device_list: List[str], - args: argparse.Namespace -) -> torch.Tensor: - """ - Orchestrate multi-GPU parallel video upscaling with temporal overlap blending. - - Splits input frames across multiple GPUs with optional temporal overlap, - spawns worker processes for parallel processing, and reassembles results - with smooth blending of overlapping regions. - - Processing flow: - 1. Split frames into chunks (with overlap if enabled) - 2. Spawn worker processes on each GPU - 3. Wait for all workers to complete - 4. Blend overlapping regions using Hann window crossfade - 5. Remove prepended frames from final result - - Args: - frames_tensor: Input frames [T, H, W, C], Float32, range [0,1] - device_list: List of device IDs as strings (e.g., ["0", "1"]) - args: Parsed command-line arguments containing all processing settings - - Returns: - Upscaled frames tensor [T', H', W', C], Float32, range [0,1] - where T' may be less than T if prepend_frames were removed - - Note: - - Single GPU: Can use multiprocessing or direct processing - - Multi-GPU with overlap: Chunks sized to multiples of batch_size for - proper temporal blending - - Prepended frames removed after all GPU workers complete (multi-GPU safe) - """ - num_devices = len(device_list) - total_frames = frames_tensor.shape[0] - - # Create overlapping chunks (for multi GPU); ensures every chunk is - # a multiple of batch_size (except last one) to avoid blending issues - if args.temporal_overlap > 0 and num_devices > 1: - chunk_with_overlap = total_frames // num_devices + args.temporal_overlap - if args.batch_size > 1: - chunk_with_overlap = ((chunk_with_overlap + args.batch_size - 1) // args.batch_size) * args.batch_size - base_chunk_size = chunk_with_overlap - args.temporal_overlap - - chunks = [] - for i in range(num_devices): - start_idx = i * base_chunk_size - if i == num_devices - 1: # last chunk/device - end_idx = total_frames - else: - end_idx = min(start_idx + chunk_with_overlap, total_frames) - chunks.append(frames_tensor[start_idx:end_idx]) - else: - chunks = torch.chunk(frames_tensor, num_devices, dim=0) - - # Use direct Queue with explicit unlimited size for large video chunks - return_queue = mp.Queue(maxsize=0) # 0 = unlimited (explicit) - workers = [] - - # Convert args namespace to dict for serialization - shared_args = vars(args).copy() - - # Start all workers - for idx, (device_id, chunk_tensor) in enumerate(zip(device_list, chunks)): - p = mp.Process( - target=_worker_process, - args=(idx, device_id, chunk_tensor.cpu().numpy(), shared_args, return_queue), - ) - p.start() - workers.append(p) - - # Collect results before joining to prevent deadlock - results_np = [None] * num_devices - collected = 0 - while collected < num_devices: - proc_idx, res_np = return_queue.get() - results_np[proc_idx] = res_np - collected += 1 - - # Now safe to join - for p in workers: - p.join() - - # Concatenate results with overlap blending using shared function - if args.temporal_overlap > 0 and num_devices > 1: - overlap = args.temporal_overlap - result_tensor = None - - for idx, res_np in enumerate(results_np): - chunk_tensor = torch.from_numpy(res_np).to(torch.float32) - - if idx == 0: - # First chunk: keep all frames - result_tensor = chunk_tensor - else: - # Subsequent chunks: blend overlapping region with accumulated result - if chunk_tensor.shape[0] > overlap and result_tensor.shape[0] >= overlap: - # Get overlapping regions - prev_tail = result_tensor[-overlap:] # Last N frames from accumulated result - cur_head = chunk_tensor[:overlap] # First N frames from current chunk - - # Blend using shared function - blended = blend_overlapping_frames(prev_tail, cur_head, overlap) - - # Replace tail of result with blended frames, then append rest of chunk - result_tensor = torch.cat([ - result_tensor[:-overlap], # Everything except the tail - blended, # Blended overlapping frames - chunk_tensor[overlap:] # Non-overlapping part of current chunk - ], dim=0) - else: - # Edge case: chunk too small, just append non-overlapping part - if chunk_tensor.shape[0] > overlap: - result_tensor = torch.cat([result_tensor, chunk_tensor[overlap:]], dim=0) - - if result_tensor is None: - result_tensor = torch.from_numpy(results_np[0]).to(torch.float32) - else: - # Simple concatenation without overlap - result_tensor = torch.from_numpy(np.concatenate(results_np, axis=0)).to(torch.float32) - - # Handle prepend_frames removal (multi-GPU safe - done after all workers complete) - if args.prepend_frames > 0: - if args.prepend_frames < result_tensor.shape[0]: - debug.log(f"Removing {args.prepend_frames} prepended frames from output", category="generation") - result_tensor = result_tensor[args.prepend_frames:] - else: - debug.log(f"prepend_frames ({args.prepend_frames}) >= total frames ({result_tensor.shape[0]}), skipping removal", - level="WARNING", category="generation", force=True) - - return result_tensor - - -# ============================================================================= -# Argument Parsing -# ============================================================================= - -def parse_arguments() -> argparse.Namespace: - """ - Parse and validate command-line arguments for SeedVR2 CLI. - - Configures all available options including model selection, processing parameters, - memory optimization settings, and output configuration. - - Returns: - Parsed arguments namespace with all CLI parameters - - Note: - - cuda_device argument only available on non-macOS systems - - Default model directory resolves to "models/SEEDVR2" if not specified - """ - - # Get the actual invocation path for usage examples - invocation = sys.argv[0] - - # Multi-line usage examples for --help - usage_examples = f""" -Examples: - - Basic image upscaling: - python {invocation} image.jpg - - Basic video video upscaling with temporal consistency - python {invocation} video.mp4 --resolution 720 --batch_size 33 - - Multi-GPU processing with temporal overlap: - python {invocation} video.mp4 --cuda_device 0,1 --resolution 1080 --batch_size 81 --uniform_batch_size --temporal_overlap 3 --prepend_frames 4 - - Memory-optimized for low VRAM (8GB): - python {invocation} image.png --dit_model seedvr2_ema_3b-Q8_0.gguf --blocks_to_swap 32 --swap_io_components --dit_offload_device cpu --vae_offload_device cpu - - High resolution with VAE tiling: - python {invocation} video.mp4 --resolution 1440 --batch_size 31 --uniform_batch_size --temporal_overlap 3 --vae_encode_tiled --vae_decode_tiled - - Batch directory processing: - python {invocation} media_folder/ --output processed/ --cuda_device 0 --cache_dit --cache_vae --dit_offload_device cpu --vae_offload_device cpu --resolution 1080 --max_resolution 1920 - -""" - - parser = argparse.ArgumentParser( - description="SeedVR2 Video Upscaler - CLI for high-quality image/video upscaling and batch processing", - epilog=usage_examples, - formatter_class=argparse.RawDescriptionHelpFormatter, - allow_abbrev=False - ) - - # Input/Output - io_group = parser.add_argument_group('Input/Output options') - io_group.add_argument("input", type=str, - help="Input: video file (.mp4, .avi, etc.), image file (.png, .jpg, etc.), or directory") - io_group.add_argument("--output", type=str, default=None, - help="Output path (default: auto-generated in 'output/' directory)") - io_group.add_argument("--output_format", type=str, default=None, choices=["mp4", "png", None], - help="Output format: 'mp4' (video) or 'png' (image sequence). Default: auto-detect from input type") - io_group.add_argument("--model_dir", type=str, default=None, - help=f"Model directory (default: ./models/{SEEDVR2_FOLDER_NAME})") - - # Model Selection - model_group = parser.add_argument_group('Model selection') - model_group.add_argument("--dit_model", type=str, default=DEFAULT_DIT, - choices=get_available_dit_models(), - help="DiT model to use. Options: 3B (fp16/fp8/GGUF) or 7B (fp16/fp8/GGUF). Default: 3B FP8") - - # Processing Parameters - process_group = parser.add_argument_group('Processing parameters') - process_group.add_argument("--resolution", type=int, default=1080, - help="Target short-side resolution in pixels (default: 1080)") - process_group.add_argument("--max_resolution", type=int, default=0, - help="Maximum resolution for any edge. Scales down if exceeded. 0 = no limit (default: 0)") - process_group.add_argument("--batch_size", type=int, default=5, - help="Frames per batch (must follow 4n+1: 1, 5, 9, 13, 17, 21,...). " - "Ideally matches shot length for best temporal consistency. Higher values improve " - "quality and speed but require more VRAM. Default: 5") - process_group.add_argument("--uniform_batch_size", action="store_true", - help="Pad final batch to match batch_size. Prevents temporal artifacts caused by small " - "final batches. Add extra compute but recommended for optimal quality.") - process_group.add_argument("--seed", type=int, default=42, - help="Random seed for reproducibility (default: 42)") - process_group.add_argument("--skip_first_frames", type=int, default=0, - help="Skip N initial frames (default: 0)") - process_group.add_argument("--load_cap", type=int, default=0, - help="Load maximum N frames from video. 0 = load all (default: 0)") - process_group.add_argument("--prepend_frames", type=int, default=0, - help="Prepend N reversed frames to reduce start artifacts (auto-removed). Default: 0") - process_group.add_argument("--temporal_overlap", type=int, default=0, - help="Frames to overlap between batches/GPUs for smooth blending (default: 0)") - - # Quality Control - quality_group = parser.add_argument_group('Quality control') - quality_group.add_argument("--color_correction", type=str, default="lab", - choices=["lab", "wavelet", "wavelet_adaptive", "hsv", "adain", "none"], - help="Color correction method: 'lab' (perceptual color matching, recommended), 'wavelet' (frequency-based), " - "'wavelet_adaptive' (wavelet + saturation correction), 'hsv' (hue-conditional), 'adain' (statistical transfer), " - "'none' (disabled) (default: lab)") - quality_group.add_argument("--input_noise_scale", type=float, default=0.0, - help="Input noise injection scale (0.0-1.0). Adds variation to input images (default: 0.0)") - quality_group.add_argument("--latent_noise_scale", type=float, default=0.0, - help="Latent noise injection scale (0.0-1.0). Adds variation to latent space (default: 0.0)") - - # Device Management - device_group = parser.add_argument_group('Device management') - if platform.system() != "Darwin": - device_group.add_argument("--cuda_device", type=str, default=None, - help="CUDA device(s): single '0' or multi-GPU '0,1,2'. Default: device 0") - device_group.add_argument("--dit_offload_device", type=str, default="none", - help="DiT offload device when idle: 'none' (keep on GPU), 'cpu' (offload to RAM), or GPU ID. " - "Frees VRAM between phases. Required for BlockSwap. Default: none") - device_group.add_argument("--vae_offload_device", type=str, default="none", - help="VAE offload device when idle: 'none', 'cpu', or GPU ID. Frees VRAM between phases. Default: none") - device_group.add_argument("--tensor_offload_device", type=str, default="cpu", - help="Intermediate tensor storage: 'cpu' (recommended), 'none' (keep on GPU), or GPU ID. Default: cpu") - - # Memory Optimization (BlockSwap) - blockswap_group = parser.add_argument_group('Memory optimization (BlockSwap)') - blockswap_group.add_argument("--blocks_to_swap", type=int, default=0, - help="Transformer blocks to swap for VRAM savings. 0-32 (3B) or 0-36 (7B). " - "Requires --dit_offload_device. Default: 0 (disabled)") - blockswap_group.add_argument("--swap_io_components", action="store_true", - help="Offload DiT I/O layers for extra VRAM savings. Requires --dit_offload_device") - - # VAE Tiling - vae_group = parser.add_argument_group('VAE tiling (for high resolution upscale)') - vae_group.add_argument("--vae_encode_tiled", action="store_true", - help="Enable VAE encode tiling to reduce VRAM during encoding") - vae_group.add_argument("--vae_encode_tile_size", type=int, default=1024, - help="VAE encode tile size in pixels (default: 1024). Applied to both height and width. Only used if --vae_encode_tiled is set") - vae_group.add_argument("--vae_encode_tile_overlap", type=int, default=128, - help="VAE encode tile overlap in pixels (default: 128). Reduces visible seams between tiles. Only used if --vae_encode_tiled is set") - vae_group.add_argument("--vae_decode_tiled", action="store_true", - help="Enable VAE decode tiling to reduce VRAM during decoding") - vae_group.add_argument("--vae_decode_tile_size", type=int, default=1024, - help="VAE decode tile size in pixels (default: 1024). Applied to both height and width. Only used if --vae_decode_tiled is set") - vae_group.add_argument("--vae_decode_tile_overlap", type=int, default=128, - help="VAE decode tile overlap in pixels (default: 128). Reduces visible seams between tiles. Only used if --vae_decode_tiled is set") - vae_group.add_argument("--tile_debug", type=str, default="false", choices=["false", "encode", "decode"], - help="Visualize tiles: 'false' (default), 'encode', or 'decode'") - - # Performance - perf_group = parser.add_argument_group('Performance optimization') - perf_group.add_argument("--attention_mode", type=str, default="sdpa", - choices=["sdpa", "flash_attn"], - help="Attention backend: 'sdpa' (default, always available) or 'flash_attn' (faster, requires package)") - perf_group.add_argument("--compile_dit", action="store_true", - help="Enable torch.compile for DiT model (20-40%% speedup, requires PyTorch 2.0+ and Triton)") - perf_group.add_argument("--compile_vae", action="store_true", - help="Enable torch.compile for VAE model (15-25%% speedup, requires PyTorch 2.0+ and Triton)") - perf_group.add_argument("--compile_backend", type=str, default="inductor", choices=["inductor", "cudagraphs"], - help="Compilation backend: 'inductor' (full optimization with Triton) or 'cudagraphs' (lightweight, no kernel optimization) (default: inductor)") - perf_group.add_argument("--compile_mode", type=str, default="default", choices=["default", "reduce-overhead", "max-autotune", "max-autotune-no-cudagraphs"], - help="Optimization level: 'default' (fast compilation), 'reduce-overhead' (lower overhead), 'max-autotune' (best runtime, slow compilation), " - "'max-autotune-no-cudagraphs' (like max-autotune without cudagraphs) (default: default)") - perf_group.add_argument("--compile_fullgraph", action="store_true", - help="Compile entire model as single graph (faster but less flexible). May fail with dynamic shapes (default: False)") - perf_group.add_argument("--compile_dynamic", action="store_true", - help="Handle varying input shapes without recompilation. Useful for different resolutions/batch sizes (default: False)") - perf_group.add_argument("--compile_dynamo_cache_size_limit", type=int, default=64, - help="Max cached compiled versions per function. Increase when using many different input shapes. Higher uses more memory (default: 64)") - perf_group.add_argument("--compile_dynamo_recompile_limit", type=int, default=128, - help="Max recompilation attempts before fallback to eager mode. Safety limit to prevent compilation loops (default: 128)") - - # Model Caching (for batch processing) - cache_group = parser.add_argument_group('Model caching (batch processing)') - cache_group.add_argument("--cache_dit", action="store_true", - help="Cache DiT model between files (single GPU only, speeds up directory processing)") - cache_group.add_argument("--cache_vae", action="store_true", - help="Cache VAE model between files (single GPU only, speeds up directory processing)") - - # Debugging - debug_group = parser.add_argument_group('Debugging') - debug_group.add_argument("--debug", action="store_true", - help="Enable verbose debug logging") - - # Auto-show help if no arguments provided - if len(sys.argv) == 1: - sys.argv.append('--help') - - return parser.parse_args() - - -# ============================================================================= -# Main Entry Point -# ============================================================================= - -def main() -> None: - """ - Main entry point for SeedVR2 Video Upscaler CLI. - - Orchestrates the complete upscaling workflow: - 1. Parse and validate command-line arguments - 2. Extract frames from input video/image(s) - 3. Download required models if not cached - 4. Process frames on single or multiple GPUs - 5. Save results as video or PNG sequence - 6. Report timing and FPS (calculated from total wall-clock time) - - Error handling: - - Validates tile configuration before processing - - Provides detailed error messages with traceback - - Ensures proper cleanup on exit (VRAM automatically freed) - - Raises: - SystemExit: On argument validation failure or processing error - """ - # print header - debug.print_header(cli=True) - - # Parse arguments - args = parse_arguments() - - # Update debug instance with --debug flag - debug.enabled = args.debug - - debug.log("Arguments:", category="setup") - for key, value in vars(args).items(): - debug.log(f"{key}: {value}", category="none", indent_level=1) - - if args.vae_encode_tiled and args.vae_encode_tile_overlap >= args.vae_encode_tile_size: - debug.log(f"VAE encode tile overlap ({args.vae_encode_tile_overlap}) must be smaller than tile size ({args.vae_encode_tile_size})", level="ERROR", category="vae", force=True) - sys.exit(1) - - if args.vae_decode_tiled and args.vae_decode_tile_overlap >= args.vae_decode_tile_size: - debug.log(f"VAE decode tile overlap ({args.vae_decode_tile_overlap}) must be smaller than tile size ({args.vae_decode_tile_size})", level="ERROR", category="vae", force=True) - sys.exit(1) - - # Validate BlockSwap configuration - either blocks_to_swap or swap_io_components requires dit_offload_device - blockswap_enabled = args.blocks_to_swap > 0 or args.swap_io_components - if blockswap_enabled and args.dit_offload_device == "none": - config_details = [] - if args.blocks_to_swap > 0: - config_details.append(f"blocks_to_swap={args.blocks_to_swap}") - if args.swap_io_components: - config_details.append("swap_io_components=True") - - debug.log( - f"BlockSwap enabled ({', '.join(config_details)}) but dit_offload_device='none'. " - "BlockSwap requires dit_offload_device to be set (typically 'cpu'). " - "Either set --dit_offload_device cpu or disable BlockSwap " - "(--blocks_to_swap 0 and do not use --swap_io_components)", - level="ERROR", category="blockswap", force=True - ) - sys.exit(1) - - # Inform about caching defaults - if args.cache_dit and args.dit_offload_device == "none": - offload_target = "system memory (CPU)" if _get_platform_type() != "mps" else "unified memory" - debug.log( - f"DiT caching enabled: Using default {offload_target} for offload. " - "Set --dit_offload_device explicitly to use a different device.", - category="cache", force=True - ) - - if args.cache_vae and args.vae_offload_device == "none": - offload_target = "system memory (CPU)" if _get_platform_type() != "mps" else "unified memory" - debug.log( - f"VAE caching enabled: Using default {offload_target} for offload. " - "Set --vae_offload_device explicitly to use a different device.", - category="cache", force=True - ) - - if args.debug: - if platform.system() == "Darwin": - debug.log("You are running on macOS and will use the MPS backend!", category="info", force=True) - else: - # Show actual CUDA device visibility - debug.log(f"CUDA_VISIBLE_DEVICES: {os.environ.get('CUDA_VISIBLE_DEVICES', 'Not set (all)')}", category="device") - if torch.cuda.is_available(): - debug.log(f"torch.cuda.device_count(): {torch.cuda.device_count()}", category="device") - debug.log(f"Using device index 0 inside script (mapped to selected GPU)", category="device") - - try: - start_time = time.time() - - # Parse GPU list - if platform.system() == "Darwin": - device_list = ["0"] - else: - if args.cuda_device: - device_list = [d.strip() for d in str(args.cuda_device).split(',') if d.strip()] - else: - device_list = ["0"] - if args.debug: - debug.log(f"Using devices: {device_list}", category="device") - - # Download models once before processing - if not download_weight(dit_model=args.dit_model, vae_model=DEFAULT_VAE, model_dir=args.model_dir, debug=debug): - debug.log("Failed to download required models. Check console output above.", level="ERROR", category="download", force=True) - sys.exit(1) - - # Determine input type and process accordingly - input_type = get_input_type(args.input) - - # Track total frames for FPS calculation (time tracked via start_time) - total_frames_processed = 0 - - # Track if output format was user-specified or auto-detected - format_auto_detected = args.output_format is None - - if input_type == 'directory': - media_files = get_media_files(args.input) - if not media_files: - debug.log(f"No video or image files found in directory: {args.input}", - level="ERROR", category="file", force=True) - sys.exit(1) - - debug.log(f"Found {len(media_files)} media files to process", category="file", force=True) - - # Validate caching with multi-GPU (not supported in CLI - would need shared memory) - if (args.cache_dit or args.cache_vae) and len(device_list) > 1: - debug.log( - "Model caching requires single GPU selection (you selected multiple GPUs). " - "Disabling caching for this run.", - level="WARNING", category="cache", force=True - ) - args.cache_dit = False - args.cache_vae = False - - # Initialize runner cache if caching enabled - runner_cache = {} if (args.cache_dit or args.cache_vae) else None - - for idx, file_path in enumerate(media_files, 1): - # Visual separation between files (except before first file) - if idx > 1: - debug.log("", category="none", force=True) - debug.log("โ”" * 60, category="none", force=True) - debug.log("", category="none", force=True) - - debug.log(f"Processing file {idx}/{len(media_files)}", category="generation", force=True) - - # Auto-detect format per file if not user-specified - if format_auto_detected: - file_type = get_input_type(file_path) - file_output_format = "mp4" if file_type == "video" else "png" - else: - file_output_format = args.output_format - - # Temporarily override args.output_format for this file - original_format = args.output_format - args.output_format = file_output_format - - # generate_output_path handles None gracefully with "outputs" default - output_path = generate_output_path(file_path, file_output_format, args.output, - input_type=get_input_type(file_path)) - - # Process with explicit output path and runner cache - frames = process_single_file(file_path, args, device_list, output_path, - format_auto_detected=format_auto_detected, - runner_cache=runner_cache) - total_frames_processed += frames - - # Restore original format - args.output_format = original_format - - elif input_type in ("video", "image"): - # Auto-detect output format for single file if not specified - if format_auto_detected: - args.output_format = "mp4" if input_type == "video" else "png" - - # Validate caching for single file (would provide no benefit but shouldn't error) - if (args.cache_dit or args.cache_vae): - if len(device_list) > 1: - debug.log( - "Model caching requires single GPU selection (you selected multiple GPUs). " - "Disabling caching for this run.", - level="WARNING", category="cache", force=True - ) - args.cache_dit = False - args.cache_vae = False - else: - debug.log( - "Model caching has no benefit for single file processing (only useful for directories). " - "Consider removing --cache_dit/--cache_vae for single files.", - category="tip", force=True - ) - - # No caching for single file (no benefit) - frames = process_single_file(args.input, args, device_list, args.output, - format_auto_detected=format_auto_detected, - runner_cache=None) - total_frames_processed += frames - - else: - debug.log(f"Unsupported input type: {args.input}", level="ERROR", category="file", force=True) - sys.exit(1) - - # Calculate total execution time - total_time = time.time() - start_time - - debug.log("", category="none", force=True) - debug.log(f"All upscaling processes completed successfully in {total_time:.2f}s", category="success", force=True) - - # Calculate and display FPS based on overall wall-clock time - if total_time > 0 and total_frames_processed > 0: - fps = total_frames_processed / total_time - debug.log(f"Average FPS: {fps:.2f} frames/sec", category="timing", force=True) - - except Exception as e: - debug.log(f"Error during processing: {e}", level="ERROR", category="generation", force=True) - import traceback - traceback.print_exc() - sys.exit(1) - - finally: - debug.log(f"Process {os.getpid()} terminating - VRAM will be automatically freed", category="cleanup", force=True) - - # print footer - debug.print_footer() - -if __name__ == "__main__": - main() \ No newline at end of file +print("\n" + "=" * 70) +print("โŒ ERROR: This nightly branch is DEPRECATED and no longer supported.") +print("=" * 70) +print("\nPlease update to the main branch:") +print(" โ€ข Remove the old SeedVR2 folder and reinstall via ComfyUI Manager (recommended)") +print(" โ€ข Or visit: https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler") +print("\n" + "=" * 70 + "\n") +sys.exit(1) \ No newline at end of file diff --git a/src/interfaces/video_upscaler.py b/src/interfaces/video_upscaler.py index 038da12..629b594 100644 --- a/src/interfaces/video_upscaler.py +++ b/src/interfaces/video_upscaler.py @@ -3,6 +3,20 @@ SeedVR2 Video Upscaler Node Main ComfyUI node for high-quality video upscaling using diffusion models """ +# ============================================================================= +# DEPRECATION NOTICE - This branch is no longer supported +# ============================================================================= +raise RuntimeError( + "\n\n" + "======================================================================\n" + "โŒ ERROR: This nightly branch of SeedVR2 is DEPRECATED and no longer supported.\n" + "======================================================================\n\n" + "Please update to the main branch:\n" + " โ€ข Remove the old SeedVR2 folder and reinstall via ComfyUI Manager (recommended)\n" + " โ€ข Or visit: https://github.com/numz/ComfyUI-SeedVR2_VideoUpscaler\n\n" + "======================================================================\n\n" +) + import torch from comfy_api.latest import io from typing import Tuple, Dict, Any, Optional