From 3b530dc98370d3ddc1d284725e270a836e6c9966 Mon Sep 17 00:00:00 2001 From: lihaoyun6 Date: Wed, 6 Aug 2025 18:52:33 +0800 Subject: [PATCH 1/5] Added MPS backend support (for running on macOS) --- inference_cli.py | 56 +++++++++------- src/common/diffusion/samplers/euler.py | 10 ++- src/common/distributed/basic.py | 8 ++- src/core/generation.py | 42 +++++++++--- src/core/infer.py | 15 ++++- src/core/model_manager.py | 11 +++- src/data/image/transforms/area_resize.py | 3 + src/data/image/transforms/na_resize.py | 9 +-- src/data/image/transforms/side_resize.py | 4 +- .../video_vae_v3/modules/attn_video_vae.py | 17 ++++- .../modules/causal_inflation_lib.py | 31 +++++++-- .../modules/attn_video_vae.py | 13 +++- .../modules/causal_inflation_lib.py | 26 ++++++-- src/optimization/blockswap.py | 1 + src/optimization/compatibility.py | 45 ++++++++++--- src/optimization/memory_manager.py | 64 +++++++++++++------ 16 files changed, 267 insertions(+), 88 deletions(-) diff --git a/inference_cli.py b/inference_cli.py index d70f088..e74db59 100644 --- a/inference_cli.py +++ b/inference_cli.py @@ -7,23 +7,25 @@ import sys import os import argparse import time +import platform import multiprocessing as mp # Ensure safe CUDA usage with multiprocessing if mp.get_start_method(allow_none=True) != 'spawn': mp.set_start_method('spawn', force=True) # ------------------------------------------------------------- # 1) Gestion VRAM (cudaMallocAsync) déjà en place -os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync") +if platform.system() != "Darwin": + os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync") -# 2) Pré-parse de la ligne de commande pour récupérer --cuda_device -_pre_parser = argparse.ArgumentParser(add_help=False) -_pre_parser.add_argument("--cuda_device", type=str, default=None) -_pre_args, _ = _pre_parser.parse_known_args() -if _pre_args.cuda_device is not None: - device_list_env = [x.strip() for x in _pre_args.cuda_device.split(',') if x.strip()!=''] - if len(device_list_env) == 1: - # Single GPU: restrict visibility now - os.environ["CUDA_VISIBLE_DEVICES"] = device_list_env[0] + # 2) Pré-parse de la ligne de commande pour récupérer --cuda_device + _pre_parser = argparse.ArgumentParser(add_help=False) + _pre_parser.add_argument("--cuda_device", type=str, default=None) + _pre_args, _ = _pre_parser.parse_known_args() + if _pre_args.cuda_device is not None: + device_list_env = [x.strip() for x in _pre_args.cuda_device.split(',') if x.strip()!=''] + if len(device_list_env) == 1: + # Single GPU: restrict visibility now + os.environ["CUDA_VISIBLE_DEVICES"] = device_list_env[0] # ------------------------------------------------------------- # 3) Imports lourds (torch, etc.) après la configuration env @@ -213,10 +215,11 @@ def save_frames_to_png(frames_tensor, output_dir, base_name, debug=False): def _worker_process(proc_idx, device_id, frames_np, shared_args, return_queue): """Worker process that performs upscaling on a slice of frames using a dedicated GPU.""" - # 1. Limit CUDA visibility to the chosen GPU BEFORE importing torch-heavy deps - os.environ["CUDA_VISIBLE_DEVICES"] = str(device_id) - # Keep same cudaMallocAsync setting - os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync") + if platform.system() != "Darwin": + # 1. Limit CUDA visibility to the chosen GPU BEFORE importing torch-heavy deps + os.environ["CUDA_VISIBLE_DEVICES"] = str(device_id) + # Keep same cudaMallocAsync setting + os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "backend:cudaMallocAsync") import torch # local import inside subprocess from src.core.model_manager import configure_runner @@ -330,7 +333,8 @@ def parse_arguments(): help="Enable VRAM preservation mode") parser.add_argument("--debug", action="store_true", help="Enable debug logging") - parser.add_argument("--cuda_device", type=str, default=None, + if platform.system() != "Darwin": + parser.add_argument("--cuda_device", type=str, default=None, help="CUDA device id(s). Single id (e.g., '0') or comma-separated list '0,1' for multi-GPU") return parser.parse_args() @@ -349,11 +353,14 @@ def main(): print(f" {key}: {value}") if args.debug: - # Show actual CUDA device visibility - print(f"🖥️ CUDA_VISIBLE_DEVICES: {os.environ.get('CUDA_VISIBLE_DEVICES', 'Not set (all)')}") - if torch.cuda.is_available(): - print(f"🖥️ torch.cuda.device_count(): {torch.cuda.device_count()}") - print(f"🖥️ Using device index 0 inside script (mapped to selected GPU)") + if platform.system() == "Darwin": + print("🖥️ You are running on macOS and will use the MPS backend!") + else: + # Show actual CUDA device visibility + print(f"🖥️ CUDA_VISIBLE_DEVICES: {os.environ.get('CUDA_VISIBLE_DEVICES', 'Not set (all)')}") + if torch.cuda.is_available(): + print(f"🖥️ torch.cuda.device_count(): {torch.cuda.device_count()}") + print(f"🖥️ Using device index 0 inside script (mapped to selected GPU)") try: # Ensure --output is a directory when using PNG format @@ -380,7 +387,11 @@ def main(): # print(f"📊 Initial VRAM: {torch.cuda.memory_allocated() / 1024**3:.2f}GB") # may initialize cuda # Parse GPU list - device_list = [d.strip() for d in str(args.cuda_device).split(',') if d.strip()] if args.cuda_device else ["0"] + if platform.system() == "Darwin": + device_list = ["0"] + else: + device_list = [d.strip() for d in str(args.cuda_device).split(',') if d.strip()] if args.cuda_device else ["0"] + if args.debug: print(f"🚀 Using devices: {device_list}") processing_start = time.time() @@ -390,7 +401,8 @@ def main(): if args.debug: print(f"🔄 Generation time: {generation_time:.2f}s") - print(f"📊 Peak VRAM usage: {torch.cuda.max_memory_allocated() / 1024**3:.2f}GB") + if platform.system() != "Darwin": + print(f"📊 Peak VRAM usage: {torch.cuda.max_memory_allocated() / 1024**3:.2f}GB") print(f"📊 Result shape: {result.shape}, dtype: {result.dtype}") # After generation_time calculation, choose saving method diff --git a/src/common/diffusion/samplers/euler.py b/src/common/diffusion/samplers/euler.py index 16abbe4..85a0c36 100644 --- a/src/common/diffusion/samplers/euler.py +++ b/src/common/diffusion/samplers/euler.py @@ -21,6 +21,7 @@ from typing import Callable import torch from einops import rearrange from torch.nn import functional as F +import platform #from ....models.dit_v2 import na @@ -71,8 +72,12 @@ class EulerSampler(Sampler): # Nettoyer les tenseurs temporaires del pred - if torch.cuda.is_available(): - torch.cuda.empty_cache() + if platform.system() == "Darwin": + if torch.mps.is_available(): + torch.mps.empty_cache() + else: + if torch.cuda.is_available(): + torch.cuda.empty_cache() i += 1 progress.update() @@ -125,3 +130,4 @@ class EulerSampler(Sampler): pred_x_s = pred_x_s.where(s >= 0, pred_x_0) pred_x_s = pred_x_s.where(s <= T, pred_x_T) return pred_x_s + \ No newline at end of file diff --git a/src/common/distributed/basic.py b/src/common/distributed/basic.py index f829aec..95570ab 100644 --- a/src/common/distributed/basic.py +++ b/src/common/distributed/basic.py @@ -21,7 +21,7 @@ from datetime import timedelta import torch import torch.distributed as dist from torch.nn.parallel import DistributedDataParallel - +import platform def get_global_rank() -> int: """ @@ -48,7 +48,10 @@ def get_device() -> torch.device: """ Get current rank device. """ - return torch.device("cuda", get_local_rank()) + device = "cuda" + if platform.system() == "Darwin": + device = "mps" + return torch.device(device, get_local_rank()) def barrier_if_distributed(*args, **kwargs): @@ -82,3 +85,4 @@ def convert_to_ddp(module: torch.nn.Module, **kwargs) -> DistributedDataParallel output_device=get_local_rank(), **kwargs, ) + \ No newline at end of file diff --git a/src/core/generation.py b/src/core/generation.py index 3f4baf2..398eca4 100644 --- a/src/core/generation.py +++ b/src/core/generation.py @@ -20,6 +20,7 @@ import os import gc import torch import time +import platform from torchvision.transforms import Compose, Lambda, Normalize @@ -70,6 +71,8 @@ def generation_step(runner, text_embeds_dict, preserve_vram, cond_latents, tempo - Advanced inference optimization """ device = "cuda" if torch.cuda.is_available() else "cpu" + if platform.system() == "Darwin": + device = "mps" # Adaptive dtype detection for optimal performance model_dtype = next(runner.dit.parameters()).dtype @@ -89,12 +92,17 @@ def generation_step(runner, text_embeds_dict, preserve_vram, cond_latents, tempo def _move_to_cuda(x): """Move tensors to CUDA with adaptive optimal dtype""" return [i.to(device, dtype=dtype) for i in x] - + # Memory optimization: Generate noise once and reuse to save VRAM - with torch.cuda.device(device): + if platform.system() == "Darwin": base_noise = torch.randn_like(cond_latents[0], dtype=dtype) noises = [base_noise] aug_noises = [base_noise * 0.1 + torch.randn_like(base_noise) * 0.05] + else: + with torch.cuda.device(device): + base_noise = torch.randn_like(cond_latents[0], dtype=dtype) + noises = [base_noise] + aug_noises = [base_noise * 0.1 + torch.randn_like(base_noise) * 0.05] # Move tensors with adaptive dtype (optimized for FP8/FP16/BFloat16) noises, aug_noises, cond_latents = _move_to_cuda(noises), _move_to_cuda(aug_noises), _move_to_cuda(cond_latents) @@ -126,7 +134,11 @@ def generation_step(runner, text_embeds_dict, preserve_vram, cond_latents, tempo # Use adaptive autocast for optimal performance with torch.no_grad(): - with torch.autocast("cuda", autocast_dtype, enabled=True): + d = "cuda" + if platform.system() == "Darwin": + d = "mps" + + with torch.autocast(d, autocast_dtype, enabled=True): video_tensors = runner.inference( noises=noises, conditions=conditions, @@ -210,6 +222,8 @@ def generation_loop(runner, images, cfg_scale=1.0, seed=666, res_w=720, batch_si - Real-time progress reporting """ device = "cuda" if torch.cuda.is_available() else "cpu" + if platform.system() == "Darwin": + device = "mps" # Log BlockSwap status if block_swap_config: @@ -246,7 +260,7 @@ def generation_loop(runner, images, cfg_scale=1.0, seed=666, res_w=720, batch_si vae_dtype = torch.bfloat16 # Optimization tips for users - if torch.cuda.is_available(): + if torch.cuda.is_available() or torch.mps.is_available(): total_frames = len(images) optimal_batches = [x for x in [i for i in range(1, 200) if i % 4 == 1] if x <= total_frames] if optimal_batches: @@ -379,7 +393,10 @@ def generation_loop(runner, images, cfg_scale=1.0, seed=666, res_w=720, batch_si tps_vae = time.time() if debug: print(f"🔄 VAE dtype: {autocast_dtype}") - with torch.autocast("cuda", autocast_dtype, enabled=True): + d = "cuda" + if platform.system() == "Darwin": + d = "mps" + with torch.autocast(d, autocast_dtype, enabled=True): cond_latents = runner.vae_encode([transformed_video]) if debug: print(f"🔄 VAE encode time: {time.time() - tps_vae} seconds") @@ -436,7 +453,10 @@ def generation_loop(runner, images, cfg_scale=1.0, seed=666, res_w=720, batch_si print(f"🔄 Time batch: {time.time() - tps_loop} seconds") # Clean VRAM after each batch when preserve_vram is active (but not with blockswap) if preserve_vram and not (block_swap_config and block_swap_config.get("blocks_to_swap", 0) > 0): - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() #del transformed_video #clear_vram_cache() # Log memory state at the end of each batch @@ -452,7 +472,10 @@ def generation_loop(runner, images, cfg_scale=1.0, seed=666, res_w=720, batch_si text_neg_embeds = text_neg_embeds.to("cpu") runner.dit.to("cpu") runner.vae.to("cpu") - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() #del text_pos_embeds, text_neg_embeds #clear_vram_cache() @@ -503,7 +526,10 @@ def generation_loop(runner, images, cfg_scale=1.0, seed=666, res_w=720, batch_si # Nettoyage immédiat VRAM del current_block, block_result - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() print(f"✅ Pre-allocation strategy completed: {final_video_images.shape}") else: diff --git a/src/core/infer.py b/src/core/infer.py index 3d6d2dd..5aa5284 100644 --- a/src/core/infer.py +++ b/src/core/infer.py @@ -19,6 +19,7 @@ from einops import rearrange from omegaconf import DictConfig, ListConfig from torch import Tensor from src.optimization.memory_manager import clear_vram_cache +import platform from src.common.diffusion import ( classifier_free_guidance_dispatcher, @@ -170,6 +171,8 @@ class VideoDiffusionInfer(): #t = time.time() device = get_device() dtype = getattr(torch, self.config.vae.dtype) + if platform.system() == "Darwin": + dtype = next(self.vae.parameters()).dtype scale = self.config.vae.scaling_factor shift = self.config.vae.get("shifting_factor", 0.0) @@ -262,6 +265,11 @@ class VideoDiffusionInfer(): def get_vram_usage(self): """Obtenir l'utilisation VRAM actuelle (allouée et réservée)""" + if platform.system() == "Darwin": + allocated = torch.mps.current_allocated_memory() / (1024**3) + reserved = torch.mps.driver_allocated_memory() / (1024**3) + max_allocated = 0 + return allocated, reserved, max_allocated if torch.cuda.is_available(): allocated = torch.cuda.memory_allocated() / (1024**3) reserved = torch.cuda.memory_reserved() / (1024**3) @@ -376,7 +384,11 @@ class VideoDiffusionInfer(): t = time.time() - with torch.autocast("cuda", target_dtype, enabled=True): + d = "cuda" + if platform.system() == "Darwin": + d = "mps" + + with torch.autocast(d, target_dtype, enabled=True): latents = self.sampler.sample( x=latents, f=lambda args: classifier_free_guidance_dispatcher( @@ -467,3 +479,4 @@ class VideoDiffusionInfer(): return samples + \ No newline at end of file diff --git a/src/core/model_manager.py b/src/core/model_manager.py index a2181f1..9025de2 100644 --- a/src/core/model_manager.py +++ b/src/core/model_manager.py @@ -18,6 +18,7 @@ Key Features: import os import time import torch +import platform from omegaconf import DictConfig, OmegaConf # Import SafeTensors with fallback @@ -167,6 +168,8 @@ def configure_runner(model, base_cache_dir, preserve_vram=False, debug=False, bl # Set device device = "cuda" if torch.cuda.is_available() else "cpu" + if platform.system() == "Darwin": + device = "mps" # Configure models checkpoint_path = os.path.join(base_cache_dir, f'./{model}') @@ -376,11 +379,11 @@ def configure_vae_model_inference(runner, device, checkpoint_path, config, prese """ # Create vae model - + if platform.system() == "Darwin": + config.vae.dtype = "bfloat16" dtype = getattr(torch, config.vae.dtype) t = time.time() loading_device = "cpu" if preserve_vram else device - with torch.device(device): runner.vae = create_object(config.vae.model) if debug: @@ -432,6 +435,10 @@ def configure_vae_model_inference(runner, device, checkpoint_path, config, prese print(f"🔄 CONFIG VAE : VAE LOAD TIME: {time.time() - t} seconds") t = time.time() runner.vae.load_state_dict(state) + + if platform.system() == "Darwin": + runner.vae = runner.vae.to(dtype=torch.bfloat16) + if state_loading_device == "cpu": runner.vae.to(device) if 'state' in locals(): diff --git a/src/data/image/transforms/area_resize.py b/src/data/image/transforms/area_resize.py index 9f621da..efc0496 100644 --- a/src/data/image/transforms/area_resize.py +++ b/src/data/image/transforms/area_resize.py @@ -31,6 +31,8 @@ class AreaResize: self.max_area = max_area self.downsample_only = downsample_only self.interpolation = interpolation + if platform.system() == "Darwin": + self.interpolation = InterpolationMode.BILINEAR def __call__(self, image: Union[torch.Tensor, Image.Image]): @@ -133,3 +135,4 @@ class ScaleResize: antialias=antialias, ) return image + \ No newline at end of file diff --git a/src/data/image/transforms/na_resize.py b/src/data/image/transforms/na_resize.py index d230e25..2d56c64 100644 --- a/src/data/image/transforms/na_resize.py +++ b/src/data/image/transforms/na_resize.py @@ -17,7 +17,7 @@ from torchvision.transforms import CenterCrop, Compose, InterpolationMode, Resiz from .area_resize import AreaResize from .side_resize import SideResize - +import platform def NaResize( resolution: int, @@ -25,24 +25,25 @@ def NaResize( downsample_only: bool, interpolation: InterpolationMode = InterpolationMode.BICUBIC, ): + Interpolation = InterpolationMode.BILINEAR if platform.system() == "Darwin" else interpolation if mode == "area": return AreaResize( max_area=resolution**2, downsample_only=downsample_only, - interpolation=interpolation, + interpolation=Interpolation, ) if mode == "side": return SideResize( size=resolution, downsample_only=downsample_only, - interpolation=interpolation, + interpolation=Interpolation, ) if mode == "square": return Compose( [ Resize( size=resolution, - interpolation=interpolation, + interpolation=Interpolation, ), CenterCrop(resolution), ] diff --git a/src/data/image/transforms/side_resize.py b/src/data/image/transforms/side_resize.py index 6e07402..83d7b4e 100644 --- a/src/data/image/transforms/side_resize.py +++ b/src/data/image/transforms/side_resize.py @@ -17,7 +17,7 @@ import torch from PIL import Image from torchvision.transforms import InterpolationMode from torchvision.transforms import functional as TVF - +import platform class SideResize: def __init__( @@ -29,6 +29,8 @@ class SideResize: self.size = size self.downsample_only = downsample_only self.interpolation = interpolation + if platform.system() == "Darwin": + self.interpolation = InterpolationMode.BILINEAR def __call__(self, image: Union[torch.Tensor, Image.Image]): """ diff --git a/src/models/video_vae_v3/modules/attn_video_vae.py b/src/models/video_vae_v3/modules/attn_video_vae.py index 83964fb..ce67fdf 100644 --- a/src/models/video_vae_v3/modules/attn_video_vae.py +++ b/src/models/video_vae_v3/modules/attn_video_vae.py @@ -29,6 +29,7 @@ from diffusers.utils import is_torch_version from diffusers.utils.accelerate_utils import apply_forward_hook from einops import rearrange from ....common.half_precision_fixes import safe_pad_operation, safe_interpolate_operation +import platform from ....common.distributed.advanced import get_sequence_parallel_world_size from ....common.logger import get_logger @@ -134,7 +135,10 @@ class Upsample3D(Upsample2D): hidden_states = [hidden_states] # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() for i in range(len(hidden_states)): hidden_states[i] = self.upscale_conv(hidden_states[i]) hidden_states[i] = rearrange( @@ -153,7 +157,10 @@ class Upsample3D(Upsample2D): hidden_states = hidden_states[0] # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() if self.use_conv: if self.name == "conv": hidden_states = self.conv(hidden_states, memory_state=memory_state, preserve_vram=preserve_vram) @@ -310,7 +317,10 @@ class ResnetBlock3D(ResnetBlock2D): hidden_states = self.nonlinearity(hidden_states) except Exception as e: print("OOM second chance") - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() time.sleep(1) hidden_states = self.nonlinearity(hidden_states) @@ -1362,3 +1372,4 @@ class VideoAutoencoderKLWrapper(VideoAutoencoderKL): for m in self.modules(): if isinstance(m, InflatedCausalConv3d): m.set_memory_limit(conv_max_mem if conv_max_mem is not None else float("inf")) + \ No newline at end of file diff --git a/src/models/video_vae_v3/modules/causal_inflation_lib.py b/src/models/video_vae_v3/modules/causal_inflation_lib.py index 88c8684..06219bb 100644 --- a/src/models/video_vae_v3/modules/causal_inflation_lib.py +++ b/src/models/video_vae_v3/modules/causal_inflation_lib.py @@ -22,6 +22,7 @@ from diffusers.models.normalization import RMSNorm from einops import rearrange from torch import Tensor, nn from torch.nn import Conv3d +import platform from .context_parallel_lib import cache_send_recv, get_cache_size from .global_config import get_norm_limit @@ -120,7 +121,10 @@ class InflatedCausalConv3d(Conv3d): if prev_cache is not None: prev_cache = list(prev_cache.split(split_sizes, dim=split_dim)) if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() # Loop Fwd. cache = None for idx in range(len(x)): @@ -166,14 +170,20 @@ class InflatedCausalConv3d(Conv3d): # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() #print("empty cache 1") #time.sleep(2) try: output = torch.cat(x, split_dim) except Exception as e: print("OOM second chance") - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() time.sleep(2) output = torch.cat(x, split_dim) return output @@ -355,19 +365,28 @@ def causal_norm_wrapper(norm_layer: nn.Module, x: torch.Tensor, preserve_vram: b x[i] = F.group_norm(x[i], num_groups_per_chunk, w, b, norm_layer.eps) except Exception as e: print("OOM Second Chance : Group Norm") - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() time.sleep(2) x[i] = F.group_norm(x[i], num_groups_per_chunk, w, b, norm_layer.eps) x[i] = x[i].to(input_dtype) # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() # ADD BY NUMZ try: x = torch.cat(x, dim=1) except Exception as e: print("OOM Second Chance : Cat") - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() time.sleep(2) x = torch.cat(x, dim=1) else: diff --git a/src/models/video_vae_v3_mine_bad/modules/attn_video_vae.py b/src/models/video_vae_v3_mine_bad/modules/attn_video_vae.py index 9584c83..d9bc9b2 100644 --- a/src/models/video_vae_v3_mine_bad/modules/attn_video_vae.py +++ b/src/models/video_vae_v3_mine_bad/modules/attn_video_vae.py @@ -9,7 +9,7 @@ # # This modified file is released under the same license. - +import platform from contextlib import nullcontext from typing import Literal, Optional, Tuple, Union import diffusers @@ -133,7 +133,10 @@ class Upsample3D(Upsample2D): hidden_states = [hidden_states] # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() for i in range(len(hidden_states)): hidden_states[i] = self.upscale_conv(hidden_states[i]) hidden_states[i] = rearrange( @@ -152,7 +155,10 @@ class Upsample3D(Upsample2D): hidden_states = hidden_states[0] # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() if self.use_conv: if self.name == "conv": hidden_states = self.conv(hidden_states, memory_state=memory_state, preserve_vram=preserve_vram) @@ -1359,3 +1365,4 @@ class VideoAutoencoderKLWrapper(VideoAutoencoderKL): for m in self.modules(): if isinstance(m, InflatedCausalConv3d): m.set_memory_limit(conv_max_mem if conv_max_mem is not None else float("inf")) + \ No newline at end of file diff --git a/src/models/video_vae_v3_mine_bad/modules/causal_inflation_lib.py b/src/models/video_vae_v3_mine_bad/modules/causal_inflation_lib.py index df48682..f819981 100644 --- a/src/models/video_vae_v3_mine_bad/modules/causal_inflation_lib.py +++ b/src/models/video_vae_v3_mine_bad/modules/causal_inflation_lib.py @@ -22,6 +22,7 @@ from diffusers.models.normalization import RMSNorm from einops import rearrange from torch import Tensor, nn from torch.nn import Conv3d +import platform from .context_parallel_lib import cache_send_recv, get_cache_size from .global_config import get_norm_limit @@ -120,7 +121,10 @@ class InflatedCausalConv3d(Conv3d): if prev_cache is not None: prev_cache = list(prev_cache.split(split_sizes, dim=split_dim)) if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() #print("empty cache 0") # Loop Fwd. cache = None @@ -166,14 +170,20 @@ class InflatedCausalConv3d(Conv3d): cache = next_cache # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() #print("empty cache 1") #time.sleep(2) try: output = torch.cat(x, split_dim) except Exception as e: print("OOM second chance") - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() time.sleep(2) output = torch.cat(x, split_dim) return output @@ -355,13 +365,19 @@ def causal_norm_wrapper(norm_layer: nn.Module, x: torch.Tensor, preserve_vram: b x[i] = F.group_norm(x[i], num_groups_per_chunk, w, b, norm_layer.eps) except Exception as e: print("OOM Second Chance : Group Norm") - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() time.sleep(2) x[i] = F.group_norm(x[i], num_groups_per_chunk, w, b, norm_layer.eps) x[i] = x[i].to(input_dtype) # ADD BY NUMZ if preserve_vram: - torch.cuda.empty_cache() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + torch.cuda.empty_cache() x = torch.cat(x, dim=1) else: x = norm_layer(x) diff --git a/src/optimization/blockswap.py b/src/optimization/blockswap.py index 0c43fab..1105dc0 100644 --- a/src/optimization/blockswap.py +++ b/src/optimization/blockswap.py @@ -18,6 +18,7 @@ import torch import weakref import psutil import gc +import platform from typing import Dict, Any, List, Tuple, Optional, Union from src.optimization.memory_manager import get_vram_usage diff --git a/src/optimization/compatibility.py b/src/optimization/compatibility.py index 14699c4..481e4de 100644 --- a/src/optimization/compatibility.py +++ b/src/optimization/compatibility.py @@ -7,6 +7,7 @@ Extracted from: seedvr2.py (lines 1045-1630) import time import torch +import platform from typing import List, Tuple, Union, Any, Optional @@ -97,13 +98,21 @@ class FP8CompatibleDiT(torch.nn.Module): # Convert all parameters of this module to BFloat16 for param_name, param in module.named_parameters(): if param.dtype != torch.bfloat16: - param.data = param.data.to(torch.bfloat16) + if param.device.type == "mps": + param_data = param.data.to("cpu").to(torch.bfloat16).to("mps") + else: + param_data = param.data.to(torch.bfloat16) + param.data = param_data rope_count += 1 # Also convert buffers (non-trainable parameters) for buffer_name, buffer in module.named_buffers(): if buffer.dtype != torch.bfloat16: - buffer.data = buffer.data.to(torch.bfloat16) + if param.device.type == "mps": + buffer_data = buffer.data.to("cpu").to(torch.bfloat16).to("mps") + else: + buffer_data = buffer.data.to(torch.bfloat16) + buffer.data = buffer_data rope_count += 1 def _force_nadit_bfloat16(self) -> None: @@ -118,13 +127,21 @@ class FP8CompatibleDiT(torch.nn.Module): if original_dtype is None: original_dtype = param.dtype if param.dtype != torch.bfloat16: - param.data = param.data.to(torch.bfloat16) + if param.device.type == "mps": + param_data = param.data.to("cpu").to(torch.bfloat16).to("mps") + else: + param_data = param.data.to(torch.bfloat16) + param.data = param_data converted_count += 1 # Also convert buffers for name, buffer in self.dit_model.named_buffers(): if buffer.dtype != torch.bfloat16: - buffer.data = buffer.data.to(torch.bfloat16) + if param.device.type == "mps": + buffer_data = buffer.data.to("cpu").to(torch.bfloat16).to("mps") + else: + buffer_data = buffer.data.to(torch.bfloat16) + buffer.data = buffer_data converted_count += 1 print(f" ✅ Converted {converted_count} parameters/buffers from {original_dtype} to BFloat16") @@ -301,17 +318,24 @@ class FP8CompatibleDiT(torch.nn.Module): k = k.view(batch_size, seq_len, num_heads, head_dim).transpose(1, 2) v = v.view(batch_size, seq_len, num_heads, head_dim).transpose(1, 2) - # Use optimized SDPA - with torch.backends.cuda.sdp_kernel( - enable_flash=True, - enable_math=True, - enable_mem_efficient=True - ): + if platform.system() == "Darwin": attn_output = torch.nn.functional.scaled_dot_product_attention( q, k, v, dropout_p=0.0, is_causal=False ) + else: + # Use optimized SDPA + with torch.backends.cuda.sdp_kernel( + enable_flash=True, + enable_math=True, + enable_mem_efficient=True + ): + attn_output = torch.nn.functional.scaled_dot_product_attention( + q, k, v, + dropout_p=0.0, + is_causal=False + ) # Reshape back attn_output = attn_output.transpose(1, 2).contiguous().view( @@ -483,3 +507,4 @@ def remove_compatibility_hooks(hooks: List[Tuple[str, Any]]) -> None: print(f"🧹 Removed {removed_count}/{len(hooks)} compatibility hooks") + \ No newline at end of file diff --git a/src/optimization/memory_manager.py b/src/optimization/memory_manager.py index 834187b..2e4cce5 100644 --- a/src/optimization/memory_manager.py +++ b/src/optimization/memory_manager.py @@ -9,6 +9,8 @@ import os import torch import gc import time +import platform +import psutil from typing import Tuple, Optional from src.common.cache import Cache from src.models.dit_v2.rope import RotaryEmbeddingBase @@ -21,12 +23,16 @@ except: pass def get_basic_vram_info(): - """🔍 Méthode basique avec PyTorch natif""" - if not torch.cuda.is_available(): - return {"error": "CUDA not available"} - - # Mémoire libre et totale (en bytes) - free_memory, total_memory = torch.cuda.mem_get_info() + if platform.system() == "Darwin": + mem = psutil.virtual_memory() + free_memory = mem.total - mem.used + total_memory = mem.total + else: + """🔍 Méthode basique avec PyTorch natif""" + if not torch.cuda.is_available(): + return {"error": "CUDA not available"} + # Mémoire libre et totale (en bytes) + free_memory, total_memory = torch.cuda.mem_get_info() # Conversion en GB free_gb = free_memory / (1024**3) @@ -39,7 +45,10 @@ def get_basic_vram_info(): # Utilisation vram_info = get_basic_vram_info() -print(f"VRAM libre: {vram_info['free_gb']:.2f} GB") +if "error" not in vram_info: + print(f"📊 Initial VRAM status: {vram_info['free_gb']:.2f}GB free / {vram_info['total_gb']:.2f}GB total") +else: + print(f"⚠️ VRAM check: {vram_info['error']} - SeedVR2 requires an NVIDIA GPU") def get_vram_usage() -> Tuple[float, float, float]: """ @@ -49,6 +58,11 @@ def get_vram_usage() -> Tuple[float, float, float]: tuple: (allocated_gb, reserved_gb, max_allocated_gb) Returns (0, 0, 0) if CUDA not available """ + if platform.system() == "Darwin": + allocated = torch.mps.current_allocated_memory() / (1024**3) + reserved = torch.mps.driver_allocated_memory() / (1024**3) + max_allocated = 0 + return allocated, reserved, max_allocated if torch.cuda.is_available(): allocated = torch.cuda.memory_allocated() / (1024**3) reserved = torch.cuda.memory_reserved() / (1024**3) @@ -62,9 +76,12 @@ def clear_vram_cache() -> None: Clear VRAM cache and run garbage collection """ print("🧹 Clearing VRAM cache...") - if torch.cuda.is_available(): - torch.cuda.empty_cache() - gc.collect() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + if torch.cuda.is_available(): + torch.cuda.empty_cache() + gc.collect() def reset_vram_peak() -> None: @@ -245,10 +262,13 @@ def fast_ram_cleanup(): # Garbage collection gc.collect() - # Clear CUDA cache - if torch.cuda.is_available(): - torch.cuda.empty_cache() - torch.cuda.reset_peak_memory_stats() + if platform.system() == "Darwin": + torch.mps.empty_cache() + else: + # Clear CUDA cache + if torch.cuda.is_available(): + torch.cuda.empty_cache() + torch.cuda.reset_peak_memory_stats() # Clear PyTorch internal caches try: @@ -293,7 +313,7 @@ def clear_all_caches(runner, debugger=None) -> int: for key, value in list(runner.cache.cache.items()): if torch.is_tensor(value): # Force deallocation of tensor storage - if value.is_cuda: + if value.is_cuda or value.is_mps: value.data = value.data.cpu() value.grad = None if value.numel() > 0: @@ -301,7 +321,7 @@ def clear_all_caches(runner, debugger=None) -> int: elif isinstance(value, (list, tuple)): for item in value: if torch.is_tensor(item): - if item.is_cuda: + if item.is_cuda or item.is_mps: item.data = item.data.cpu() item.grad = None if item.numel() > 0: @@ -388,10 +408,16 @@ def clear_all_caches(runner, debugger=None) -> int: # Force garbage collection gc.collect(2) # Collect all generations - # Clear CUDA cache - if torch.cuda.is_available(): - torch.cuda.empty_cache() + if platform.system() == "Darwin": + # Clear MPS cache + torch.mps.empty_cache() if COMFYUI_AVAILABLE: mm.soft_empty_cache() + else: + # Clear CUDA cache + if torch.cuda.is_available(): + torch.cuda.empty_cache() + if COMFYUI_AVAILABLE: + mm.soft_empty_cache() return cleaned_items From 3ae18bc84556ea392de285d02c418e055e2fdf5a Mon Sep 17 00:00:00 2001 From: lihaoyun6 Date: Wed, 6 Aug 2025 23:54:57 +0800 Subject: [PATCH 2/5] Fixed BlockSwap on MPS backend --- src/interfaces/comfyui_node.py | 10 ++++++---- src/optimization/blockswap.py | 33 +++++++++++++++++++++++++-------- 2 files changed, 31 insertions(+), 12 deletions(-) diff --git a/src/interfaces/comfyui_node.py b/src/interfaces/comfyui_node.py index 70d7352..8275c33 100644 --- a/src/interfaces/comfyui_node.py +++ b/src/interfaces/comfyui_node.py @@ -5,6 +5,7 @@ import os import time import torch +import platform from typing import Tuple, Dict, Any from src.utils.downloads import download_weight, get_base_cache_dir @@ -155,7 +156,7 @@ class SeedVR2: cleanup_blockswap(self.runner, keep_state_for_cache=True) # Clear all caches - if self.runner: + if self.runner: clear_all_caches(self.runner, debugger) else: @@ -335,7 +336,7 @@ class SeedVR2BlockSwap: "BOOLEAN", { "default": True, - "tooltip": "Use non-blocking GPU transfers for better performance.", + "tooltip": "Use non-blocking GPU transfers for better performance.\n(This will always False on macOS to prevent Nan tensors)", }, ), "offload_io_components": ( @@ -399,11 +400,12 @@ The actual memory savings depend on your specific model architecture and will be cache_model, enable_debug, ): + Use_non_blocking = False if platform.system() == "Darwin" else use_non_blocking if blocks_to_swap > 0 or offload_io_components: configs = [] if blocks_to_swap > 0: configs.append(f"{blocks_to_swap} blocks") - if use_non_blocking: + if Use_non_blocking: configs.append("non blocking") if offload_io_components: configs.append("I/O components") @@ -413,7 +415,7 @@ The actual memory savings depend on your specific model architecture and will be return ( { "blocks_to_swap": blocks_to_swap, - "use_non_blocking": use_non_blocking, + "use_non_blocking": Use_non_blocking, "offload_io_components": offload_io_components, "cache_model": cache_model, "enable_debug": enable_debug, diff --git a/src/optimization/blockswap.py b/src/optimization/blockswap.py index 1105dc0..3536322 100644 --- a/src/optimization/blockswap.py +++ b/src/optimization/blockswap.py @@ -19,6 +19,7 @@ import weakref import psutil import gc import platform +import psutil from typing import Dict, Any, List, Tuple, Optional, Union from src.optimization.memory_manager import get_vram_usage @@ -105,7 +106,7 @@ class BlockSwapDebugger: """Log current memory state for debugging.""" if self.enabled: # GPU Memory - if torch.cuda.is_available(): + if torch.cuda.is_available() or torch.mps.is_available(): allocated_gb, reserved_gb, peak_gb = get_vram_usage() vram_info = f"VRAM: {allocated_gb:.2f}/{reserved_gb:.2f}GB (peak: {peak_gb:.2f}GB)" self.vram_history.append(allocated_gb) @@ -180,6 +181,8 @@ def apply_block_swap_to_dit(runner, block_swap_config: Dict[str, Any]) -> None: # Determine devices device = "cuda" if torch.cuda.is_available() else "cpu" + if platform.system() == "Darwin": + device = "mps" offload_device = str(mm.unet_offload_device()) use_non_blocking = block_swap_config.get("use_non_blocking", True) @@ -366,7 +369,7 @@ def _wrap_block_forward(block: torch.nn.Module, block_idx: int, model: torch.nn. self.to(model.main_device, non_blocking=model.use_non_blocking) # Synchronize if needed - if hasattr(model, 'use_non_blocking') and not model.use_non_blocking: + if hasattr(model, 'use_non_blocking') and not model.use_non_blocking and platform.system() != "Darwin": torch.cuda.synchronize() # Execute forward pass with OOM protection @@ -385,8 +388,13 @@ def _wrap_block_forward(block: torch.nn.Module, block_idx: int, model: torch.nn. ) # Only clear cache under memory pressure - if torch.cuda.memory_allocated() > torch.cuda.get_device_properties(0).total_memory * 0.9: - mm.soft_empty_cache() + if platform.system() == "Darwin": + mem = psutil.virtual_memory() + if torch.mps.current_allocated_memory() > mem.total * 0.9: + mm.soft_empty_cache() + else: + if torch.cuda.memory_allocated() > torch.cuda.get_device_properties(0).total_memory * 0.9: + mm.soft_empty_cache() else: output = original_forward(*args, **kwargs) @@ -437,7 +445,10 @@ def _wrap_io_forward(module: torch.nn.Module, module_name: str, model: torch.nn. # Synchronize if not using non-blocking transfers if hasattr(model, 'use_non_blocking') and not model.use_non_blocking: - torch.cuda.synchronize() + if platform.system() == "Darwin": + torch.mps.synchronize() + else: + torch.cuda.synchronize() # Execute forward pass output = self._original_forward(*args, **kwargs) @@ -455,8 +466,13 @@ def _wrap_io_forward(module: torch.nn.Module, module_name: str, model: torch.nn. ) # Only clear cache under memory pressure - if torch.cuda.memory_allocated() > torch.cuda.get_device_properties(0).total_memory * 0.9: - mm.soft_empty_cache() + if platform.system() == "Darwin": + mem = psutil.virtual_memory() + if torch.mps.current_allocated_memory() > mem.total * 0.9: + mm.soft_empty_cache() + else: + if torch.cuda.memory_allocated() > torch.cuda.get_device_properties(0).total_memory * 0.9: + mm.soft_empty_cache() return output @@ -493,7 +509,7 @@ def _patch_rope_for_blockswap(model, debugger: BlockSwapDebugger) -> None: debugger.log(f"RoPE issue for {name}: {e}") # Get current device from parameters - current_device = "cuda" + current_device = "mps" if platform.system() == "Darwin" else "cuda" if list(self.parameters()): current_device = next(self.parameters()).device @@ -740,3 +756,4 @@ def cleanup_blockswap(runner, keep_state_for_cache: bool = False) -> None: + \ No newline at end of file From 7534ab5b87760cbb5071a85c4a98d7500a21ba49 Mon Sep 17 00:00:00 2001 From: lihaoyun6 Date: Thu, 7 Aug 2025 12:19:39 +0800 Subject: [PATCH 3/5] Fixed dtype mangling on MPS backend --- src/core/infer.py | 2 -- src/core/model_manager.py | 11 +++++++---- 2 files changed, 7 insertions(+), 6 deletions(-) diff --git a/src/core/infer.py b/src/core/infer.py index 5aa5284..9dcecbf 100644 --- a/src/core/infer.py +++ b/src/core/infer.py @@ -171,8 +171,6 @@ class VideoDiffusionInfer(): #t = time.time() device = get_device() dtype = getattr(torch, self.config.vae.dtype) - if platform.system() == "Darwin": - dtype = next(self.vae.parameters()).dtype scale = self.config.vae.scaling_factor shift = self.config.vae.get("shifting_factor", 0.0) diff --git a/src/core/model_manager.py b/src/core/model_manager.py index 9025de2..9becd1b 100644 --- a/src/core/model_manager.py +++ b/src/core/model_manager.py @@ -249,7 +249,7 @@ def load_quantized_state_dict(checkpoint_path, device="cpu", keep_native_fp8=Tru if hasattr(tensor, 'dtype') and tensor.dtype in fp8_types: fp8_detected = True break - + if fp8_detected: if keep_native_fp8: # Keep native FP8 format for optimal performance @@ -380,7 +380,10 @@ def configure_vae_model_inference(runner, device, checkpoint_path, config, prese # Create vae model if platform.system() == "Darwin": - config.vae.dtype = "bfloat16" + config.vae.dtype = "float16" + if "fp8_e4m3fn" in runner._model_name: + config.vae.dtype = "bfloat16" + dtype = getattr(torch, config.vae.dtype) t = time.time() loading_device = "cpu" if preserve_vram else device @@ -435,9 +438,9 @@ def configure_vae_model_inference(runner, device, checkpoint_path, config, prese print(f"🔄 CONFIG VAE : VAE LOAD TIME: {time.time() - t} seconds") t = time.time() runner.vae.load_state_dict(state) - + if platform.system() == "Darwin": - runner.vae = runner.vae.to(dtype=torch.bfloat16) + runner.vae = runner.vae.to(dtype=getattr(torch, config.vae.dtype)) if state_loading_device == "cpu": runner.vae.to(device) From e328545582e52ad8201ae7ceae5037be91cd3238 Mon Sep 17 00:00:00 2001 From: lihaoyun6 Date: Tue, 12 Aug 2025 22:31:40 +0800 Subject: [PATCH 4/5] Delete .DS_Store file --- .DS_Store | Bin 6148 -> 0 bytes src/.DS_Store | Bin 8196 -> 0 bytes src/common/.DS_Store | Bin 6148 -> 0 bytes src/common/diffusion/.DS_Store | Bin 6148 -> 0 bytes src/models/.DS_Store | Bin 6148 -> 0 bytes src/models/video_vae_v3/.DS_Store | Bin 6148 -> 0 bytes 6 files changed, 0 insertions(+), 0 deletions(-) delete mode 100644 .DS_Store delete mode 100644 src/.DS_Store delete mode 100644 src/common/.DS_Store delete mode 100644 src/common/diffusion/.DS_Store delete mode 100644 src/models/.DS_Store delete mode 100644 src/models/video_vae_v3/.DS_Store diff --git a/.DS_Store b/.DS_Store deleted file mode 100644 index 5e7839596486c2ae597b4b2081a854eda7160515..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 6148 zcmeHK%}T>S5Z-O8Z7D(y3VK`cS}-kAC|*LWFJMFuDm5`hgE3o@)Er77_xeJ<8H&#u#@O;eauVG1h^G$Wf^gbk~L&CK-|A7}-3CWdPPkFgLNk z4*2a&ma&vQ2F3U9kE1NN?N8olwsv;gAiegz_bdx9_wz;O`spoNS5n47rTf8kG%Kdo z{<%zYKT2k)Du}`vq}<&^Nhk|fE|M@)wVn=iL8jKxayb}|M}09E*{i--j)!(%oQ{X9 zRkw3+d~$XDgpj*kF!3y3%G0S=L(j=D26L_lZDvOX9AO?s5VqmKoFvmit zx794r;)wxbppF6D9|SZ+$6%>ZZ5`0z^%>(WL=@2RErBQuItELP5CP%36i}CP^Tgn~ z9Q?xMIR;COx}0$}GmK+qt{yL3%?^H{(iwL&QcnyJ1DgyqwQ1q`e*wQt?IVA)ge+o! z82D!l@YckexUeX5w*FWip0xtn12h!OD^URfed7`U2JRz=DyZWEb;xrJmKt#s^s90} Ox(Fyjs3Qh`fq^ga4N1HJ diff --git a/src/.DS_Store b/src/.DS_Store deleted file mode 100644 index 40aa478ea6a96cfe15cc46ddec37dac2dc4482ef..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 8196 zcmeHM&yUhT6n?XdEbD6Ipb6fZ7;hv3F`E!C74aXiMh|L`(z@GNI;1UP-ul_xL^J5Xp;%Q@8e2>h0!@T*Xecz;0-atg6GY*rw1~nCPn(RSXs<0`BFm(7GH5}+#oEtQB5{6E~ zURl@-Md+)8=jh=iT!UP*0<6Hg0xEYe(*T+K5TDij{i{EUqbzSUeu=&P(zWsd?C9W) zcPb}d-p|Kbr=Px3-=0eu`{(Vx{~`(}z3S~JGRgZ<5~i9E1tEsKeiY?lK2=5}f-|9`sq{C~S{ zn(u}cU`9Z*CCK>$x*K5et)l=X%t&oA9 diff --git a/src/common/.DS_Store b/src/common/.DS_Store deleted file mode 100644 index 048a9a817c8bfe723e37319c4fbefc803ccf11de..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 6148 zcmeHKPfNov6i>G4R))}ng5CmN2X;dS!%M023s}*E%53S>Vr^t?-C+!R*DvH3@$>jz zl8R#!Jc+pX%F8c#e;V>j$!i#6+$*9kV>V-~1&Ua+pjjd4M_rPdu^@8wj)G+@W|F1h zbSat)e~|&)JBvlIYq#)x*}v>5Ok@b&x9};6v)pmsd9B{q*ldEdo7eumocei?7nv7K zE^%}wWfE5XAUu!9(~-S>EYm!Q)A2+l#L*Z+t}fyX_Tp4Upp+2k-ayYb$f$; zM|Atnyd!3Vp3@PBgWh~@wRZLoj!s7p$zv*CG@l$kJtbQPD|m&@ilsgKlQfa(9eAtE zDua+1AO?tmHDSOUf35bKte2Ka3=ji9X8_L!0gC7vEH$d50~-8&M1KPj1^RfGKokaD zgQZ6BfN-4(s8hMQVsM=fc46XNgQZ5D&bXQx-eYFw=7z%6>|hrvoN-qpwZs51u*g7N z4{fafC*R-y7n5j23=jkViUHmj`a=(vWNPcu;;_~V&>K(`j4L&Mra(uPVu;03ybG!X Z>;gN0uEA0xSU~7UK+!-AG4QJld;szLO~U{H diff --git a/src/common/diffusion/.DS_Store b/src/common/diffusion/.DS_Store deleted file mode 100644 index a0210ecd7c22ca2e543e0b1e520e221054136ffa..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 6148 zcmeHK%}T>S5T0$TO({YT3VK`cS}-kAC|*LXFJMFuDm5WRgK4%jtv!@N?)pN$h|lB9 z?nW$C@FZeqVD_7xpDg<&>|_Bz^kz{DpaB35Dq*RH!xutv(lser524V{NMHyP$ijIx zUdv|3Uu1yZodX#JFoH3BdVgV}LX5qJ<0Q_euKOlR<;vD}6{TLi^&iySp9a%e)(a-r zG`mnL2@5+2FXQ2S(AYgy=`@Jb;Y1h2(GXK^uHrOO^PZZeQKoBs6R4pK8V8F-tKIE1 zWvk;Zn{v@@yG?oAZ7-KKXYcUn03@`&5#DF~lo%#mtnYYReFatkhfX)YrO6Xb44eFx<2l{=ac!iJzZMsVkN{gPw z+#p6!gegTdrNUk@gegbAw0WMz+@L83p;yM|*p-F7p$NS?`lU_>;ThzS8DIuh87P}+ zgZls3_xJzRBAzh=%)q~5KvepEzlU40y>)GK)N3W`9V!Xsd2m>Wb73jYXb8hBs^ewBfD5Is$7 diff --git a/src/models/.DS_Store b/src/models/.DS_Store deleted file mode 100644 index 966a94f808816b074659bc504373a3be0a909b08..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 6148 zcmeH~K}*9h6vvZox{WFHprE&a*MZ%X!SGV%`~p_=pi)~}v{;+5cH3bLde<-H7xDA> zUXqGq?&3jkyazA8G@a3A#v0(rN)w774ExbntYTV_T)jgwk4>4+ zA|Tm+Pyp9%2i7cRA*|Z>^_#MjV3v)7LW(IuXC2Px zoo>I^7M-5GXp4EjYq!N|zq?p88~aDcXP4e{@{-C|tuluXP}#0wi8rXMo#50Rr-@7- zA*ReP^N6GXDL@KrumWbwYqU0ajx;q=fE4(h0=Pd2aHL~cXsWFP3jBS<{u&Yu>Uftx z3PZ=R(1ac^u2X?Jm76OD*XhtMOq^p_XzFyv)yVK3Gcq?f6s|^xcA>%S5T3QwwiF=;1-&hJEtnQ56fdFH7cim+mD-S?!8BW%#vV!`cYPsW#OHBl zcLP>?@FZeqVE3DypWVy{*&hH9{XA>~H~>(`Mkq*GBV=Cd+OolfLeG&v1X(c8CPA`h zqQ7XOZ?C}^6oin&r|(w*dftLb5@$2leHUAm>h?|zrCGc4AJyESjb?e)A5Cv)bg5Jl zl=>*Rilh0^**jC|Y!s)_R42q?gdw-raT=<5U*&0->0IA98Yn~OaIt82dfk?6cim-6 zE_xleB~N;t<+4%VKRQ0W7(ONARJ|B31=6l%*J25;sC+5w*`KC~N*~ZuW|h&5%m6dM z3@`&5#eh8q>dlR+nu}!yn1P=#K>LHlM(9~A4C<`|JGwrTze-4gI=v+brA5zTVGuni z!lWXaRAHYO!la{J+BnZ*VbG+5&@1CScIEQ%BJ}ELmpUATXOMelfEie2pkjs&o&V?f z%dCCmucq*b8DIwf83Up^@CSV?%AT#?%A>PZV!OviLU9EtD5$So0&qb4$bohmza$;w YJd1@vnuY8-9g!~rnh@@ofnQ+Y1O69G=Kufz From 3db50c3849352ded78a9b2f32d4795a3c6e3ca5b Mon Sep 17 00:00:00 2001 From: lihaoyun6 Date: Tue, 12 Aug 2025 22:32:36 +0800 Subject: [PATCH 5/5] update .gitignore --- .gitignore | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.gitignore b/.gitignore index 28b31d2..7b478a8 100644 --- a/.gitignore +++ b/.gitignore @@ -22,4 +22,5 @@ src/core/isolated_generation.py src/core/subprocess_runner.py models/video_vae_v3_mine_bad/ src/processing/ -TILE_VAE* \ No newline at end of file +TILE_VAE* +.DS_Store \ No newline at end of file