diff --git a/nodes.py b/nodes.py index c90b62f..875665a 100644 --- a/nodes.py +++ b/nodes.py @@ -10,6 +10,15 @@ import comfy.model_management as mm import comfy.utils from contextlib import nullcontext from .lvdm.models.samplers.ddim import DDIMSampler +from .lvdm.modules.networks.openaimodel3d import ControlNet +from .utils.enhanced_clip_vision import encode_image_masked + +from contextlib import nullcontext +try: + from accelerate import init_empty_weights + is_accelerate_available = True +except: + pass def split_and_trim(input_string): # Split the string into an array using '|' as a separator @@ -37,12 +46,15 @@ class DownloadAndLoadDynamiCrafterModel: def INPUT_TYPES(s): return {"required": { "model": ( - [ 'tooncrafter_512_interp-fp16.safetensors', - 'dynamicrafter_512_interp_v1_bf16.safetensors', - 'dynamicrafter_1024_v1_bf16.safetensors' + [ 'tooncrafter_512_interp-pruned-fp16.safetensors', + 'dynamicrafter_512_fp16_pruned.safetensors', + 'dynamicrafter_512_interp_fp16_pruned.safetensors', + 'dynamicrafter_1024_fp16_pruned.safetensors', + 'dynamicrafter-CIL-512-no-watermark-fixed-pruned-fp16.safetensors', + 'dynamicrafter-CIL-1024-no-watermark-pruned-fp16.safetensors' ], { - "default": 'tooncrafter_512_interp-fp16.safetensors' + "default": 'tooncrafter_512_interp-pruned-fp16.safetensors' }), "dtype": ( [ @@ -63,6 +75,7 @@ class DownloadAndLoadDynamiCrafterModel: CATEGORY = "DynamiCrafterWrapper" def loadmodel(self, dtype, model, fp8_unet=False): + device = mm.get_torch_device() mm.soft_empty_cache() custom_config = { 'dtype': dtype, @@ -103,25 +116,156 @@ class DownloadAndLoadDynamiCrafterModel: model_config = config.pop("model", OmegaConf.create()) model_config['params']['unet_config']['params']['use_checkpoint']=False - self.model = instantiate_from_config(model_config) - self.model = load_model_checkpoint(self.model, model_path) - self.model.eval() + if dtype == "auto": try: if mm.should_use_fp16(): - self.model.to(convert_dtype('fp16')) + precision = (convert_dtype('fp16')) elif mm.should_use_bf16(): - self.model.to(convert_dtype('bf16')) + precision = (convert_dtype('bf16')) else: - self.model.to(convert_dtype('fp32')) + precision = (convert_dtype('fp32')) except: raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.") else: - self.model.to(convert_dtype(dtype)) + precision = (convert_dtype(dtype)) + + with (init_empty_weights() if is_accelerate_available else nullcontext()): + self.model = instantiate_from_config(model_config) + self.model = load_model_checkpoint(self.model, model_path, precision, device) + self.model.to(precision).to(device).eval() + if fp8_unet: self.model.model.diffusion_model = self.model.model.diffusion_model.to(torch.float8_e4m3fn) print(f"Model using dtype: {self.model.dtype}") - return (self.model,) + + dcmodel = { + 'model': self.model, + 'model_name': model, + 'dtype': precision + } + return (dcmodel,) + +class DownloadAndLoadDynamiCrafterCNModel: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "model": ( + [ + 'sketch_encoder-fp16.safetensors', + ], + ), + }, + } + + RETURN_TYPES = ("DC_CN_MODEL",) + RETURN_NAMES = ("DynCraft_CN_model",) + FUNCTION = "loadmodel" + CATEGORY = "DynamiCrafterWrapper" + + def loadmodel(self, model): + custom_config = { + 'ckpt_name': model, + } + if not hasattr(self, 'cn_model') or self.cn_model == None or custom_config != self.current_config: + + download_path = os.path.join(folder_paths.models_dir, "checkpoints", "dynamicrafter", "controlnet") + cn_model_path = os.path.join(download_path, model) + + if not os.path.exists(cn_model_path): + print(f"Downloading model to: {cn_model_path}") + from huggingface_hub import snapshot_download + snapshot_download(repo_id="Kijai/DynamiCrafter_pruned", + allow_patterns=[f"*{model}*"], + local_dir=download_path, + local_dir_use_symlinks=False) + cn_config = { + "use_checkpoint": True, + "image_size": 32, # unused + "in_channels": 4, + "hint_channels": 3, + "model_channels": 320, + "attention_resolutions": [4, 2, 1], + "num_res_blocks": 2, + "channel_mult": [1, 2, 4, 4], + "num_head_channels": 64, # need to fix for flash-attn + "use_spatial_transformer": True, + "use_linear_in_transformer": True, + "transformer_depth": 1, + "context_dim": 1024, + "legacy": False + } + if "sketch_encoder" in model: + cn_config["hint_channels"] = 1 + + self.cn_model = ControlNet(**cn_config) + print("Loading ControlNet") + cn_sd = comfy.utils.load_torch_file(cn_model_path) + self.cn_model.load_state_dict(cn_sd, strict=True) + print("ControlNet loaded") + + controlnet = { + 'model': self.cn_model, + 'config': cn_config, + } + + return (controlnet,) + +class DynamiCrafterCNLoader: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "ckpt_name": (folder_paths.get_filename_list("controlnet"), ), + }, + } + + RETURN_TYPES = ("DC_CN_MODEL",) + RETURN_NAMES = ("DynCraft_CN_model",) + FUNCTION = "loadmodel" + CATEGORY = "DynamiCrafterWrapper" + + def loadmodel(self, ckpt_name): + custom_config = { + 'ckpt_name': ckpt_name, + } + if not hasattr(self, 'cn_model') or self.cn_model == None or custom_config != self.current_config: + self.current_config = custom_config + + model_path = folder_paths.get_full_path("controlnet", ckpt_name) + print(f"Loading ControlNet from: {model_path}") + + cn_config = { + "use_checkpoint": True, + "image_size": 32, # unused + "in_channels": 4, + "hint_channels": 3, + "model_channels": 320, + "attention_resolutions": [4, 2, 1], + "num_res_blocks": 2, + "channel_mult": [1, 2, 4, 4], + "num_head_channels": 64, # need to fix for flash-attn + "use_spatial_transformer": True, + "use_linear_in_transformer": True, + "transformer_depth": 1, + "context_dim": 1024, + "legacy": False + } + if "sketch_encoder" in ckpt_name: + cn_config["hint_channels"] = 1 + + self.cn_model = ControlNet(**cn_config) + print("Loading ControlNet") + cn_sd = comfy.utils.load_torch_file(model_path) + self.cn_model.load_state_dict(cn_sd, strict=True) + del cn_sd + print("ControlNet loaded") + + controlnet = { + 'model': self.cn_model, + 'config': cn_config, + } + + return (controlnet,) class DownloadAndLoadCLIPModel: @classmethod @@ -241,6 +385,7 @@ class DynamiCrafterModelLoader: CATEGORY = "DynamiCrafterWrapper" def loadmodel(self, dtype, ckpt_name, fp8_unet=False): + device = mm.get_torch_device() mm.soft_empty_cache() custom_config = { 'dtype': dtype, @@ -250,7 +395,9 @@ class DynamiCrafterModelLoader: if not hasattr(self, 'model') or self.model == None or custom_config != self.current_config: self.current_config = custom_config model_path = folder_paths.get_full_path("checkpoints", ckpt_name) - ckpt_base_name = os.path.basename(ckpt_name) + ckpt_base_name = os.path.basename(model_path) + print(f"Loading model from: {model_path}") + base_name, _ = os.path.splitext(ckpt_base_name) if 'toon' in base_name and '512' in base_name: config_file=os.path.join(script_directory, "configs", "tooncrafter_512_interp.yaml") @@ -268,25 +415,34 @@ class DynamiCrafterModelLoader: model_config = config.pop("model", OmegaConf.create()) model_config['params']['unet_config']['params']['use_checkpoint']=False - self.model = instantiate_from_config(model_config) - self.model = load_model_checkpoint(self.model, model_path) - self.model.eval() + if dtype == "auto": try: if mm.should_use_fp16(): - self.model.to(convert_dtype('fp16')) + precision = (convert_dtype('fp16')) elif mm.should_use_bf16(): - self.model.to(convert_dtype('bf16')) + precision = (convert_dtype('bf16')) else: - self.model.to(convert_dtype('fp32')) + precision = (convert_dtype('fp32')) except: raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.") else: - self.model.to(convert_dtype(dtype)) + precision = (convert_dtype(dtype)) + + with (init_empty_weights() if is_accelerate_available else nullcontext()): + self.model = instantiate_from_config(model_config) + self.model = load_model_checkpoint(self.model, model_path, precision, device) + self.model.to(precision).to(device).eval() + if fp8_unet: self.model.model.diffusion_model = self.model.model.diffusion_model.to(torch.float8_e4m3fn) print(f"Model using dtype: {self.model.dtype}") - return (self.model,) + + dcmodel = { + 'model': self.model, + 'model_name': ckpt_name, + } + return (dcmodel,) class DynamiCrafterI2V: @classmethod @@ -320,6 +476,8 @@ class DynamiCrafterI2V: "mask": ("MASK",), "frame_window_size": ("INT", {"default": 16, "min": 1, "max": 200, "step": 1}), "frame_window_stride": ("INT", {"default": 4, "min": 1, "max": 200, "step": 1}), + "augmentation_level": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 10.0, "step": 0.0001}), + "init_noise": ("DCNOISE",), } } @@ -328,32 +486,37 @@ class DynamiCrafterI2V: FUNCTION = "process" CATEGORY = "DynamiCrafterWrapper" - def process(self, model, image, clip_vision, positive, negative, cfg, steps, eta, seed, fs, keep_model_loaded, frames, vae_dtype, frame_window_size=16, frame_window_stride=4, mask=None, image2=None): + def process(self, model, image, clip_vision, positive, negative, cfg, steps, eta, seed, fs, keep_model_loaded, + frames, vae_dtype, frame_window_size=16, frame_window_stride=4, mask=None, image2=None, augmentation_level=0, init_noise=None): device = mm.get_torch_device() offload_device = mm.unet_offload_device() mm.unload_all_models() mm.soft_empty_cache() + self.model = model['model'] + torch.manual_seed(seed) - dtype = model.dtype + dtype = self.model.dtype if vae_dtype == "auto": try: if mm.should_use_bf16(): - model.first_stage_model.to(convert_dtype('bf16')) + self.model.first_stage_model.to(convert_dtype('bf16')) else: - model.first_stage_model.to(convert_dtype('fp32')) + self.model.first_stage_model.to(convert_dtype('fp32')) except: raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.") else: - model.first_stage_model.to(convert_dtype(vae_dtype)) - print(f"VAE using dtype: {model.first_stage_model.dtype}") + self.model.first_stage_model.to(convert_dtype(vae_dtype)) + print(f"VAE using dtype: {self.model.first_stage_model.dtype}") - self.model = model + self.model.to(device) autocast_condition = (dtype != torch.float32) and not comfy.model_management.is_device_mps(device) with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=dtype) if autocast_condition else nullcontext(): - image = image * 2 - 1 + image = image.permute(0, 3, 1, 2).to(dtype).to(device) + if augmentation_level > 0: + image += torch.randn_like(image) * augmentation_level B, C, H, W = image.shape orig_H, orig_W = H, W @@ -362,21 +525,26 @@ class DynamiCrafterI2V: if H % 64 != 0: H = H - (H % 64) if orig_H % 64 != 0 or orig_W % 64 != 0: - image = F.interpolate(image, size=(H, W), mode="bicubic") + image = F.interpolate(image, size=(H, W), mode="bilinear") B, C, H, W = image.shape noise_shape = [B, self.model.model.diffusion_model.out_channels, frames, H // 8, W // 8] self.model.first_stage_model.to(device) - - z = get_latent_z(self.model, image.unsqueeze(2)) #bc,1,hw + encode_pixels = image.unsqueeze(2) * 2 - 1 + z = get_latent_z(self.model, encode_pixels) #bc,1,hw if image2 is not None: - image2 = image2 * 2 - 1 image2 = image2.permute(0, 3, 1, 2).to(dtype).to(device) + + if augmentation_level > 0: + image2 += torch.randn_like(image2) * augmentation_level + if image2.shape != image.shape: - image2 = F.interpolate(image, size=(H, W), mode="bicubic") - z2 = get_latent_z(self.model, image2.unsqueeze(2)) #bc,1,hw + image2 = F.interpolate(image, size=(H, W), mode="bilinear") + + encode_pixels = image2.unsqueeze(2) * 2 - 1 + z2 = get_latent_z(self.model, encode_pixels) #bc,1,hw img_tensor_repeat = repeat(z, 'b c t h w -> b c (repeat t) h w', repeat=frames) img_tensor_repeat = torch.zeros_like(img_tensor_repeat) img_tensor_repeat[:,:,:1,:,:] = z @@ -389,15 +557,20 @@ class DynamiCrafterI2V: self.model.image_proj_model.to(device) text_emb = positive[0][0].to(device) - cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device) + #cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device) + cond_images = encode_image_masked(clip_vision, image.permute(0, 2, 3, 1), batch_size=0, tiles=4, ratio=0.1).last_hidden_state.to(device) + #cond_images = torch.sum(cond_images, dim=0).unsqueeze(0) + cond_images = torch.mean(cond_images, dim=0).unsqueeze(0) img_emb = self.model.image_proj_model(cond_images) imtext_cond = torch.cat([text_emb, img_emb], dim=1) - del cond_images, img_emb, text_emb + del cond_images, img_emb, text_emb, encode_pixels fs = torch.tensor([fs], dtype=torch.long, device=self.model.device) - cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat]} + cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat], "control_cond": None} + + self.model.control_model = None if noise_shape[-1] == 32: timestep_spacing = "uniform" @@ -411,7 +584,7 @@ class DynamiCrafterI2V: uc_emb = negative[0][0].to(device) ## process image embedding token if hasattr(self.model, 'embedder'): - uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device) + uc_img = torch.rand(noise_shape[0],3,224,224).to(self.model.device) ## img: b c h w >> b l c uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to(self.model.device) uc_img = self.model.image_proj_model(uc_img) @@ -440,9 +613,30 @@ class DynamiCrafterI2V: mask = mask.permute(0, 2, 1, 3, 4) mask = torch.where(mask < 1.0, torch.tensor(0.0, device=device, dtype=dtype), torch.tensor(1.0, device=device, dtype=dtype)) + if init_noise is not None: + if init_noise['analytic_init']: + eps=torch.randn_like(init_noise['mu_p']) + sigma_p = init_noise['sigma_p'] + init = (init_noise['mu_p'] + sigma_p*eps).to(dtype).to(device) + if noise_shape[2] % init.shape[2] == 0: + init = init.repeat(1, 1, noise_shape[2] // init.shape[2], 1, 1) + else: + raise ValueError("The target dimension size is not an integral multiple of the original dimension size.") + else: + init = None + timestep_spacing = "uniform_trailing" + guidance_rescale = 0.7 + ddpm_from = init_noise['M'] + + + else: + init = None + ddpm_from = 1000 + #inference ddim_sampler = DDIMSampler(self.model) - samples, _ = ddim_sampler.sample(S=steps, + samples, _ = ddim_sampler.sample( + S=steps, conditioning=cond, batch_size=noise_shape[0], shape=noise_shape[1:], @@ -452,7 +646,7 @@ class DynamiCrafterI2V: eta=eta, temporal_length=noise_shape[2], conditional_guidance_scale_temporal=None, - x_T=None, + x_T=init, fs=fs, timestep_spacing=timestep_spacing, guidance_rescale=guidance_rescale, @@ -460,7 +654,8 @@ class DynamiCrafterI2V: mask=mask, x0=img_tensor_repeat.clone() if mask is not None else None, frame_window_size = frame_window_size, - frame_window_stride = frame_window_stride + frame_window_stride = frame_window_stride, + ddpm_from=ddpm_from ) assert not torch.isnan(samples).any().item(), "Resulting tensor containts NaNs. I'm unsure why this happens, changing step count and/or image dimensions might help." @@ -488,6 +683,54 @@ class DynamiCrafterI2V: video = F.interpolate(video.permute(0, 3, 1, 2), size=(final_H, final_W), mode="bicubic").permute(0, 2, 3, 1) last_image = video[-1].unsqueeze(0) return (video, last_image) + +class DynamiCrafterLoadInitNoise: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "model": ("DCMODEL",), + "M": ("INT", {"default": 1000, "min": 1, "max": 1000, "step": 1}), + "analytic_init": ("BOOLEAN", {"default": True}), + }, + } + + RETURN_TYPES = ("DCNOISE", "INT", "INT",) + RETURN_NAMES = ("init_noise", "width", "height",) + FUNCTION = "load" + CATEGORY = "DynamiCrafterWrapper" + + def load(self, model, M, analytic_init): + device = mm.get_torch_device() + + model_name = model['model_name'] + if '512' in model_name: + analytic_noise = "initial_noise_512.safetensors" + elif '1024' in model_name: + analytic_noise = "initial_noise_1024.safetensors" + else: + print("Can't find matching init_noise for model: ", model_name) + model_path = os.path.join(script_directory, 'init_noises', analytic_noise) + + # Analytic-Init:load initial noise + dic = comfy.utils.load_torch_file(model_path) + expectation_X_0=dic["Expectation_X0"].to(device) + tr_Cov_d=dic["Tr_Cov_d"].to(device) + + sqrt_alpha_t=model['model'].get_sqrt_alpha_t_bar(expectation_X_0,torch.tensor([M-1]).to(device)) + mu_p=sqrt_alpha_t*expectation_X_0 + alpha_t=sqrt_alpha_t**2 + sigma_p=torch.sqrt(1-alpha_t + alpha_t*tr_Cov_d) + + init_noise = { + "sigma_p": sigma_p, + "mu_p": mu_p, + "M": M, + "analytic_init": analytic_init + } + width = mu_p.shape[4] * 8 + height = mu_p.shape[3] * 8 + + return (init_noise, width, height) class ToonCrafterInterpolation: @classmethod @@ -516,6 +759,10 @@ class ToonCrafterInterpolation: }, "optional": { "image_embed_ratio": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "augmentation_level": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 10.0, "step": 0.0001}), + "optional_latents": ("LATENT",), + "ddpm_from": ("INT", {"default": 1000, "min": 1, "max": 1000, "step": 1}), + "controlnet": ("DC_CONTROL",), } } @@ -524,27 +771,35 @@ class ToonCrafterInterpolation: FUNCTION = "process" CATEGORY = "DynamiCrafterWrapper" - def process(self, model, clip_vision, images, positive, negative, cfg, steps, eta, seed, fs, frames, vae_dtype, image_embed_ratio=1.0): + def process(self, model, clip_vision, images, positive, negative, cfg, steps, eta, seed, fs, frames, + vae_dtype, image_embed_ratio=1.0, augmentation_level=0, optional_latents=None, ddpm_from=1000, controlnet=None): device = mm.get_torch_device() offload_device = mm.unet_offload_device() mm.unload_all_models() mm.soft_empty_cache() torch.manual_seed(seed) - dtype = model.dtype + + self.model = model['model'] + + if controlnet is not None: + self.model.control_model = controlnet["model"] + else: + self.model.control_model = None + + dtype = self.model.dtype if vae_dtype == "auto": try: if mm.should_use_bf16(): - model.first_stage_model.to(convert_dtype('bf16')) + self.model.first_stage_model.to(convert_dtype('bf16')) else: - model.first_stage_model.to(convert_dtype('fp32')) + self.model.first_stage_model.to(convert_dtype('fp32')) except: raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.") else: - model.first_stage_model.to(convert_dtype(vae_dtype)) - print(f"VAE using dtype: {model.first_stage_model.dtype}") + self.model.first_stage_model.to(convert_dtype(vae_dtype)) + print(f"VAE using dtype: {self.model.first_stage_model.dtype}") - images = images * 2 - 1 images = images.permute(0, 3, 1, 2).to(dtype).to(device) B, C, H, W = images.shape @@ -556,55 +811,88 @@ class ToonCrafterInterpolation: if orig_H % 64 != 0 or orig_W % 64 != 0: images = F.interpolate(images, size=(H, W), mode="bicubic") - self.model = model self.model.to(device) out = [] hidden_states = [] - + pbar = comfy.utils.ProgressBar(len(images) - 1) autocast_condition = (dtype != torch.float32) and not comfy.model_management.is_device_mps(device) with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=dtype) if autocast_condition else nullcontext(): - for i in range(len(images) - 1): + for i in range(len(images) - 1) if len(images) > 1 else range(len(images)): videos, videos2 = None, None + mm.soft_empty_cache() image = images[i].unsqueeze(0) - image2 = images[i+1].unsqueeze(0) + if len(images) !=1: + image2 = images[i+1].unsqueeze(0) B, C, H, W = image.shape noise_shape = [B, self.model.model.diffusion_model.out_channels, frames, H // 8, W // 8] self.model.first_stage_model.to(device) - videos = image.unsqueeze(2) # bc1hw - videos = repeat(videos, 'b c t h w -> b c (repeat t) h w', repeat=frames//2) - videos2 = image2.unsqueeze(2) # bc1hw - videos2 = repeat(videos2, 'b c t h w -> b c (repeat t) h w', repeat=frames//2) - videos = torch.cat([videos, videos2], dim=2) + if augmentation_level > 0: + image += torch.randn_like(image) * augmentation_level + image2 += torch.randn_like(image) * augmentation_level - z, hs = get_latent_z_with_hidden_states(self.model, videos) - hidden_states.append(hs) + encode_pixels = image.unsqueeze(2) * 2 - 1 + videos = encode_pixels # bc1hw + videos = repeat(videos, 'b c t h w -> b c (repeat t) h w', repeat=frames // 2) + + if len(images) == 1: + videos = torch.cat([videos, videos], dim=2) + else: + encode_pixels = image2.unsqueeze(2) * 2 - 1 + videos2 = encode_pixels # bc1hw + videos2 = repeat(videos2, 'b c t h w -> b c (repeat t) h w', repeat=frames // 2) + videos = torch.cat([videos, videos2], dim=2) + + try: + z, hs = get_latent_z_with_hidden_states(self.model, videos) + hs = [t.to("cpu") for t in hs] + hidden_states.append(hs) + except: + z = get_latent_z(self.model, videos) + hidden_states = None img_tensor_repeat = torch.zeros_like(z) img_tensor_repeat[:,:,:1,:,:] = z[:,:,:1,:,:] - img_tensor_repeat[:,:,-1:,:,:] = z[:,:,-1:,:,:] + if len(images) !=1: + img_tensor_repeat[:,:,-1:,:,:] = z[:,:,-1:,:,:] self.model.first_stage_model.to(offload_device) text_emb = positive[0][0].to(device) - - cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device) - cond_images2 = clip_vision.encode_image(image2.permute(0, 2, 3, 1))['last_hidden_state'].to(device) + + #cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))["last_hidden_state"].to(device) + #cond_images2 = clip_vision.encode_image(image2.permute(0, 2, 3, 1))["last_hidden_state"].to(device) self.model.image_proj_model.to(device) - + cond_images = encode_image_masked(clip_vision, image.permute(0, 2, 3, 1), batch_size=0, tiles=4, ratio=1).last_hidden_state.to(device) + cond_images = torch.sum(cond_images, dim=0).unsqueeze(0) img_emb = self.model.image_proj_model(cond_images) - img_emb2 = self.model.image_proj_model(cond_images2) - img_embeds = img_emb * image_embed_ratio + img_emb2 * (1.0 - image_embed_ratio) + if len(images) !=1: + cond_images2 = encode_image_masked(clip_vision, image2.permute(0, 2, 3, 1), batch_size=0, tiles=4, ratio=1).last_hidden_state.to(device) + cond_images2 = torch.sum(cond_images2, dim=0).unsqueeze(0) + img_emb2 = self.model.image_proj_model(cond_images2) + img_embeds = img_emb * image_embed_ratio + img_emb2 * (1.0 - image_embed_ratio) + else: + img_embeds = img_emb imtext_cond = torch.cat([text_emb, img_embeds], dim=1) - del cond_images, img_emb, img_emb2, text_emb + del cond_images, img_emb, text_emb - fs = torch.tensor([fs], dtype=torch.long, device=self.model.device) - cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat]} + if comfy.model_management.is_device_mps(device): + fs = torch.tensor([fs], dtype=torch.float32, device=self.model.device) + else: + fs = torch.tensor([fs], dtype=torch.float64, device=self.model.device) + + if controlnet is not None: + cn_videos = controlnet["cn_videos"] + cn_videos = cn_videos.to(dtype).to(device) + else: + cn_videos = None + + cond = {"c_crossattn": [imtext_cond], "fs": fs, "c_concat": [img_tensor_repeat], "control_cond": cn_videos} if noise_shape[-1] == 32: timestep_spacing = "uniform" @@ -618,7 +906,7 @@ class ToonCrafterInterpolation: uc_emb = negative[0][0].to(device) ## process image embedding token if hasattr(self.model, 'embedder'): - uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device) + uc_img = torch.rand(noise_shape[0],3,224,224).to(self.model.device) ## img: b c h w >> b l c uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to(self.model.device) uc_img = self.model.image_proj_model(uc_img) @@ -634,6 +922,16 @@ class ToonCrafterInterpolation: self.model.image_proj_model.to(offload_device) #inference + if optional_latents is not None: + samples_in = optional_latents['samples'].clone().to(device) + samples_in = samples_in * 0.18215 + samples_in = samples_in.unsqueeze(0).permute(0, 2, 1, 3, 4) + noise = torch.randn(noise_shape, device=device) + samples_in[:, :, 0, :, :] = noise[:, :, 0, :, :] + samples_in[:, :, -1, :, :] = noise[:, :, -1, :, :] + samples_in = samples_in.to(dtype).to(device) + else: + samples_in = None self.model.model.diffusion_model.to(device) ddim_sampler = DDIMSampler(self.model) @@ -647,7 +945,7 @@ class ToonCrafterInterpolation: eta=eta, temporal_length=noise_shape[2], conditional_guidance_scale_temporal=None, - x_T=None, + x_T=samples_in, fs=fs, timestep_spacing=timestep_spacing, guidance_rescale=guidance_rescale, @@ -656,11 +954,13 @@ class ToonCrafterInterpolation: x0=None, frame_window_size = 16, frame_window_stride = 4, + ddpm_from=ddpm_from ) - + print(f"Sampled {i+1} out of {(len(images) - 1)}") assert not torch.isnan(samples).any().item(), "Resulting tensor containts NaNs. I'm unsure why this happens, changing step count and/or image dimensions might help." - samples = samples.squeeze(0).permute(1, 0, 2, 3) + samples = samples.squeeze(0).permute(1, 0, 2, 3).cpu().to(self.model.first_stage_model.dtype) out.append(samples) + pbar.update(1) self.model.to(offload_device) mm.soft_empty_cache() @@ -672,7 +972,43 @@ class ToonCrafterInterpolation: "samples": samples, "hidden_states": hidden_states, } + return (latent,) + +class DynamiCrafterControlnetApply: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "controlnet": ("DC_CN_MODEL",), + "images": ("IMAGE",), + "control_scale": ("FLOAT", {"default": 0.6, "min": 0.0, "max": 1.0, "step": 0.01}), + } + } + + RETURN_TYPES = ("DC_CONTROL",) + RETURN_NAMES = ("controlnet",) + FUNCTION = "process" + CATEGORY = "DynamiCrafterWrapper" + + def process(self, controlnet, images, control_scale): + + controlnet['model'].control_scale = control_scale + + #images = images * 2.0 - 1.0 + + cn_tensor = images.permute(3, 0, 1, 2).unsqueeze(0) + print("control frame: ", cn_tensor.shape) # b c t h w + print(controlnet["config"]) + if controlnet["config"]["hint_channels"] == 1: + cn_tensor = cn_tensor[:, :1, :, :, :] + print("control frame: ", cn_tensor.shape) # b c t h w + + controlnet = { + "model": controlnet['model'], + "cn_videos": cn_tensor, + } + + return (controlnet,) class ToonCrafterDecode: @classmethod @@ -702,52 +1038,63 @@ class ToonCrafterDecode: def process(self, model, latent, vae_dtype, prune_last_frame=False): device = mm.get_torch_device() + offload_device = mm.unet_offload_device() + mm.unload_all_models() mm.soft_empty_cache() - + self.model = model['model'] samples = latent["samples"] + num_samples = samples.shape[0] samples = samples * 0.18215 + self.model.first_stage_model.to(device) + #samples = samples.to(model.first_stage_model.device) + hs = latent["hidden_states"] - model.en_and_decode_n_samples_a_time = 16 + self.model.en_and_decode_n_samples_a_time = 16 if vae_dtype == "auto": try: if mm.should_use_bf16(): - model.first_stage_model.to(convert_dtype('bf16')) + self.model.first_stage_model.to(convert_dtype('bf16')) else: - model.first_stage_model.to(convert_dtype('fp32')) + self.model.first_stage_model.to(convert_dtype('fp32')) except: raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.") else: - model.first_stage_model.to(convert_dtype(vae_dtype)) - print(f"VAE using dtype: {model.first_stage_model.dtype}") + self.model.first_stage_model.to(convert_dtype(vae_dtype)) + print(f"VAE using dtype: {self.model.first_stage_model.dtype}") out = [] iteration_counter = 0 - for i in range(0, samples.shape[0], 16): - + pbar = comfy.utils.ProgressBar(num_samples // 16) + autocast_condition = (self.model.first_stage_model.dtype != torch.float32) and not comfy.model_management.is_device_mps(device) + for i in range(0, num_samples, 16): batch_start = i - batch_end = min(i + 16, samples.shape[0]) # Ensure we don't go beyond the tensor's size - batch_samples = samples[batch_start:batch_end] - model.first_stage_model.to(device) - if mm.XFORMERS_IS_AVAILABLE: - print("Using xformers") - additional_decode_kwargs = {'ref_context': hs[iteration_counter]} - decoded_images = model.decode_first_stage(batch_samples, **additional_decode_kwargs) #b c t h w - else: - raise Exception("XFormers not available, it is required for ToonCrafter decoder. Alternatively you can use a standard VAE Decode -node instead, but this has a negative effect on the image quality though.") - #print("xformers not available, ToonCrafter does not work well without it.") - #decoded_images = model.decode_first_stage(batch_samples) #b c t h w - - video = decoded_images.detach().cpu() - video = torch.clamp(video.float(), -1., 1.) - video = (video + 1.0) / 2.0 - video = video.squeeze(0).permute(0, 2, 3, 1) - iteration_counter += 1 - out.append(video) - del decoded_images - mm.soft_empty_cache() + batch_end = min(i + 16, num_samples) # Ensure we don't go beyond the tensor's size + batch_samples = samples[batch_start:batch_end].to(self.model.first_stage_model.device) + with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=self.model.first_stage_model.dtype) if autocast_condition else nullcontext(): + #if mm.XFORMERS_IS_AVAILABLE: + print(f"Decoding frames {iteration_counter * 16} - {16 + iteration_counter * 16} out of {num_samples} using xformers") + if hs is not None: + hs_ = hs[iteration_counter] + hs_ = [t.to(self.model.first_stage_model.device) for t in hs_] + additional_decode_kwargs = {'ref_context': hs_} + decoded_images = self.model.decode_first_stage(batch_samples, **additional_decode_kwargs) #b c t h w + else: + decoded_images = self.model.decode_first_stage(batch_samples) #b c t h w + #else: + # raise Exception("XFormers not available, it is required for ToonCrafter decoder. Alternatively you can use a standard VAE Decode -node instead, but this has a negative effect on the image quality though.") + + video = decoded_images.detach().cpu() + video = torch.clamp(video.float(), -1., 1.) + video = (video + 1.0) / 2.0 + video = video.squeeze(0).permute(0, 2, 3, 1) + iteration_counter += 1 + pbar.update(1) + out.append(video) + del decoded_images + mm.soft_empty_cache() + self.model.first_stage_model.to(offload_device) video_out = torch.cat(out, dim=0) - model.first_stage_model.to('cpu') if prune_last_frame: video_out = video_out[torch.arange(video_out.shape[0]) % 16!= 0] @@ -758,12 +1105,14 @@ class DynamiCrafterBatchInterpolation: def INPUT_TYPES(s): return {"required": { "model": ("DCMODEL",), + "clip_vision": ("CLIP_VISION",), + "positive": ("CONDITIONING",), + "negative": ("CONDITIONING",), "images": ("IMAGE",), "steps": ("INT", {"default": 50, "min": 1, "max": 200, "step": 1}), "cfg": ("FLOAT", {"default": 7.0, "min": 0.0, "max": 20.0, "step": 0.01}), "eta": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 20.0, "step": 0.01}), "frames": ("INT", {"default": 16, "min": 1, "max": 100, "step": 1}), - "prompt": ("STRING", {"multiline": True, "default": "",}), "seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff}), "fs": ("INT", {"default": 10, "min": 2, "max": 100, "step": 1}), "keep_model_loaded": ("BOOLEAN", {"default": True}), @@ -785,30 +1134,31 @@ class DynamiCrafterBatchInterpolation: FUNCTION = "process" CATEGORY = "DynamiCrafterWrapper" - def process(self, model, images, prompt, cfg, steps, eta, seed, fs, keep_model_loaded, frames, vae_dtype, cut_near_keyframes): + def process(self, model, images, clip_vision, positive, negative, cfg, steps, eta, seed, fs, keep_model_loaded, + frames, vae_dtype, cut_near_keyframes): assert images.shape[0] > 1, "DynamiCrafterBatchInterpolation needs at least 2 images" device = mm.get_torch_device() mm.unload_all_models() mm.soft_empty_cache() torch.manual_seed(seed) - dtype = model.dtype + + self.model = model['model'] + dtype = self.model.dtype if vae_dtype == "auto": try: if mm.should_use_bf16(): - model.first_stage_model.to(convert_dtype('bf16')) + self.model.first_stage_model.to(convert_dtype('bf16')) else: - model.first_stage_model.to(convert_dtype('fp32')) + self.model.first_stage_model.to(convert_dtype('fp32')) except: raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.") else: - model.first_stage_model.to(convert_dtype(vae_dtype)) - print(f"VAE using dtype: {model.first_stage_model.dtype}") - - self.model = model + self.model.first_stage_model.to(convert_dtype(vae_dtype)) + print(f"VAE using dtype: {self.model.first_stage_model.dtype}") + self.model.to(device) - images = images * 2 - 1 images = images.permute(0, 3, 1, 2).to(dtype).to(device) B, C, H, W = images.shape orig_H, orig_W = H, W @@ -819,7 +1169,6 @@ class DynamiCrafterBatchInterpolation: if orig_H % 64 != 0 or orig_W % 64 != 0: images = F.interpolate(images, size=(H, W), mode="bicubic") - split_prompt = split_and_trim(prompt) out = [] autocast_condition = (dtype != torch.float32) and not comfy.model_management.is_device_mps(device) with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=dtype) if autocast_condition else nullcontext(): @@ -832,8 +1181,11 @@ class DynamiCrafterBatchInterpolation: self.model.first_stage_model.to(device) - z = get_latent_z(self.model, image.unsqueeze(2)) #bc,1,hw - z2 = get_latent_z(self.model, image2.unsqueeze(2)) #bc,1,hw + encode_pixels1 = image * 2 - 1 + encode_pixels2 = image2 * 2 - 1 + + z = get_latent_z(self.model, encode_pixels1.unsqueeze(2)) #bc,1,hw + z2 = get_latent_z(self.model, encode_pixels2.unsqueeze(2)) #bc,1,hw img_tensor_repeat = repeat(z, 'b c t h w -> b c (repeat t) h w', repeat=frames) img_tensor_repeat = torch.zeros_like(img_tensor_repeat) img_tensor_repeat[:,:,:1,:,:] = z @@ -841,23 +1193,18 @@ class DynamiCrafterBatchInterpolation: self.model.first_stage_model.to('cpu') - self.model.cond_stage_model.to(device) self.model.embedder.to(device) self.model.image_proj_model.to(device) - - try: - text_emb = self.model.get_learned_conditioning([split_prompt[i]]) - print("Prompt: ", split_prompt[i]) - except: - text_emb = self.model.get_learned_conditioning([split_prompt[0]]) - print("Prompt: ", split_prompt[0]) + text_emb = positive[0][0].to(device) + cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device) - cond_images = self.model.embedder(image) img_emb = self.model.image_proj_model(cond_images) imtext_cond = torch.cat([text_emb, img_emb], dim=1) fs = torch.tensor([fs], dtype=torch.long, device=self.model.device) - cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat]} + cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat], "control_cond": None} + + self.model.control_model = None if noise_shape[-1] == 32: timestep_spacing = "uniform" @@ -867,13 +1214,14 @@ class DynamiCrafterBatchInterpolation: guidance_rescale = 0.7 ## construct unconditional guidance - if cfg != 1.0: - uc_emb = self.model.get_learned_conditioning([""]) + if cfg != 1.0: + uc_emb = negative[0][0].to(device) ## process image embedding token if hasattr(self.model, 'embedder'): - uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device) + uc_img = torch.rand(noise_shape[0], 3, 224, 224).to(self.model.device) ## img: b c h w >> b l c - uc_img = self.model.embedder(uc_img) + uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to( + self.model.device) uc_img = self.model.image_proj_model(uc_img) uc_emb = torch.cat([uc_emb, uc_img], dim=1) if isinstance(cond, dict): @@ -883,8 +1231,6 @@ class DynamiCrafterBatchInterpolation: uc = uc_emb else: uc = None - - self.model.cond_stage_model.to('cpu') self.model.embedder.to('cpu') self.model.image_proj_model.to('cpu') @@ -965,16 +1311,24 @@ NODE_CLASS_MAPPINGS = { "ToonCrafterDecode": ToonCrafterDecode, "DownloadAndLoadDynamiCrafterModel": DownloadAndLoadDynamiCrafterModel, "DownloadAndLoadCLIPModel": DownloadAndLoadCLIPModel, - "DownloadAndLoadCLIPVisionModel": DownloadAndLoadCLIPVisionModel + "DownloadAndLoadCLIPVisionModel": DownloadAndLoadCLIPVisionModel, + "DynamiCrafterLoadInitNoise": DynamiCrafterLoadInitNoise, + "DownloadAndLoadDynamiCrafterCNModel": DownloadAndLoadDynamiCrafterCNModel, + "DynamiCrafterControlnetApply": DynamiCrafterControlnetApply, + "DynamiCrafterCNLoader": DynamiCrafterCNLoader } NODE_DISPLAY_NAME_MAPPINGS = { "DynamiCrafterI2V": "DynamiCrafterI2V", - "DynamiCrafterModelLoader": "DynamiCrafterModelLoader", - "DynamiCrafterBatchInterpolation": "DynamiCrafterBatchInterpolation", - "ToonCrafterInterpolation": "ToonCrafterInterpolation", - "ToonCrafterDecode": "ToonCrafterDecode", - "DownloadAndLoadDynamiCrafterModel": "DownloadAndLoadDynamiCrafterModel", - "DownloadAndLoadCLIPModel": "DownloadAndLoadCLIPModel", - "DownloadAndLoadCLIPVisionModel": "DownloadAndLoadCLIPVisionModel" + "DynamiCrafterModelLoader": "DynamiCrafter ModelLoader", + "DynamiCrafterBatchInterpolation": "DynamiCrafter BatchInterpolation", + "ToonCrafterInterpolation": "ToonCrafter Interpolation", + "ToonCrafterDecode": "ToonCrafter Decode", + "DownloadAndLoadDynamiCrafterModel": "(Down)Load DynamiCrafterModel", + "DownloadAndLoadCLIPModel": "(Down)Load CLIPModel", + "DownloadAndLoadCLIPVisionModel": "(Down)Load CLIPVisionModel", + "DynamiCrafterLoadInitNoise": "DynamiCrafter LoadInitNoise", + "DownloadAndLoadDynamiCrafterCNModel": "(Down)Load DynamiCrafter CNModel", + "DynamiCrafterControlnetApply": "DynamiCrafter ControlnetApply", + "DynamiCrafterCNLoader": "DynamiCrafter CNLoader" }