Update nodes.py
This commit is contained in:
@@ -10,6 +10,15 @@ import comfy.model_management as mm
|
||||
import comfy.utils
|
||||
from contextlib import nullcontext
|
||||
from .lvdm.models.samplers.ddim import DDIMSampler
|
||||
from .lvdm.modules.networks.openaimodel3d import ControlNet
|
||||
from .utils.enhanced_clip_vision import encode_image_masked
|
||||
|
||||
from contextlib import nullcontext
|
||||
try:
|
||||
from accelerate import init_empty_weights
|
||||
is_accelerate_available = True
|
||||
except:
|
||||
pass
|
||||
|
||||
def split_and_trim(input_string):
|
||||
# Split the string into an array using '|' as a separator
|
||||
@@ -37,12 +46,15 @@ class DownloadAndLoadDynamiCrafterModel:
|
||||
def INPUT_TYPES(s):
|
||||
return {"required": {
|
||||
"model": (
|
||||
[ 'tooncrafter_512_interp-fp16.safetensors',
|
||||
'dynamicrafter_512_interp_v1_bf16.safetensors',
|
||||
'dynamicrafter_1024_v1_bf16.safetensors'
|
||||
[ 'tooncrafter_512_interp-pruned-fp16.safetensors',
|
||||
'dynamicrafter_512_fp16_pruned.safetensors',
|
||||
'dynamicrafter_512_interp_fp16_pruned.safetensors',
|
||||
'dynamicrafter_1024_fp16_pruned.safetensors',
|
||||
'dynamicrafter-CIL-512-no-watermark-fixed-pruned-fp16.safetensors',
|
||||
'dynamicrafter-CIL-1024-no-watermark-pruned-fp16.safetensors'
|
||||
],
|
||||
{
|
||||
"default": 'tooncrafter_512_interp-fp16.safetensors'
|
||||
"default": 'tooncrafter_512_interp-pruned-fp16.safetensors'
|
||||
}),
|
||||
"dtype": (
|
||||
[
|
||||
@@ -63,6 +75,7 @@ class DownloadAndLoadDynamiCrafterModel:
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def loadmodel(self, dtype, model, fp8_unet=False):
|
||||
device = mm.get_torch_device()
|
||||
mm.soft_empty_cache()
|
||||
custom_config = {
|
||||
'dtype': dtype,
|
||||
@@ -103,25 +116,156 @@ class DownloadAndLoadDynamiCrafterModel:
|
||||
|
||||
model_config = config.pop("model", OmegaConf.create())
|
||||
model_config['params']['unet_config']['params']['use_checkpoint']=False
|
||||
self.model = instantiate_from_config(model_config)
|
||||
self.model = load_model_checkpoint(self.model, model_path)
|
||||
self.model.eval()
|
||||
|
||||
if dtype == "auto":
|
||||
try:
|
||||
if mm.should_use_fp16():
|
||||
self.model.to(convert_dtype('fp16'))
|
||||
precision = (convert_dtype('fp16'))
|
||||
elif mm.should_use_bf16():
|
||||
self.model.to(convert_dtype('bf16'))
|
||||
precision = (convert_dtype('bf16'))
|
||||
else:
|
||||
self.model.to(convert_dtype('fp32'))
|
||||
precision = (convert_dtype('fp32'))
|
||||
except:
|
||||
raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.")
|
||||
else:
|
||||
self.model.to(convert_dtype(dtype))
|
||||
precision = (convert_dtype(dtype))
|
||||
|
||||
with (init_empty_weights() if is_accelerate_available else nullcontext()):
|
||||
self.model = instantiate_from_config(model_config)
|
||||
self.model = load_model_checkpoint(self.model, model_path, precision, device)
|
||||
self.model.to(precision).to(device).eval()
|
||||
|
||||
if fp8_unet:
|
||||
self.model.model.diffusion_model = self.model.model.diffusion_model.to(torch.float8_e4m3fn)
|
||||
print(f"Model using dtype: {self.model.dtype}")
|
||||
return (self.model,)
|
||||
|
||||
dcmodel = {
|
||||
'model': self.model,
|
||||
'model_name': model,
|
||||
'dtype': precision
|
||||
}
|
||||
return (dcmodel,)
|
||||
|
||||
class DownloadAndLoadDynamiCrafterCNModel:
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
return {"required": {
|
||||
"model": (
|
||||
[
|
||||
'sketch_encoder-fp16.safetensors',
|
||||
],
|
||||
),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("DC_CN_MODEL",)
|
||||
RETURN_NAMES = ("DynCraft_CN_model",)
|
||||
FUNCTION = "loadmodel"
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def loadmodel(self, model):
|
||||
custom_config = {
|
||||
'ckpt_name': model,
|
||||
}
|
||||
if not hasattr(self, 'cn_model') or self.cn_model == None or custom_config != self.current_config:
|
||||
|
||||
download_path = os.path.join(folder_paths.models_dir, "checkpoints", "dynamicrafter", "controlnet")
|
||||
cn_model_path = os.path.join(download_path, model)
|
||||
|
||||
if not os.path.exists(cn_model_path):
|
||||
print(f"Downloading model to: {cn_model_path}")
|
||||
from huggingface_hub import snapshot_download
|
||||
snapshot_download(repo_id="Kijai/DynamiCrafter_pruned",
|
||||
allow_patterns=[f"*{model}*"],
|
||||
local_dir=download_path,
|
||||
local_dir_use_symlinks=False)
|
||||
cn_config = {
|
||||
"use_checkpoint": True,
|
||||
"image_size": 32, # unused
|
||||
"in_channels": 4,
|
||||
"hint_channels": 3,
|
||||
"model_channels": 320,
|
||||
"attention_resolutions": [4, 2, 1],
|
||||
"num_res_blocks": 2,
|
||||
"channel_mult": [1, 2, 4, 4],
|
||||
"num_head_channels": 64, # need to fix for flash-attn
|
||||
"use_spatial_transformer": True,
|
||||
"use_linear_in_transformer": True,
|
||||
"transformer_depth": 1,
|
||||
"context_dim": 1024,
|
||||
"legacy": False
|
||||
}
|
||||
if "sketch_encoder" in model:
|
||||
cn_config["hint_channels"] = 1
|
||||
|
||||
self.cn_model = ControlNet(**cn_config)
|
||||
print("Loading ControlNet")
|
||||
cn_sd = comfy.utils.load_torch_file(cn_model_path)
|
||||
self.cn_model.load_state_dict(cn_sd, strict=True)
|
||||
print("ControlNet loaded")
|
||||
|
||||
controlnet = {
|
||||
'model': self.cn_model,
|
||||
'config': cn_config,
|
||||
}
|
||||
|
||||
return (controlnet,)
|
||||
|
||||
class DynamiCrafterCNLoader:
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
return {"required": {
|
||||
"ckpt_name": (folder_paths.get_filename_list("controlnet"), ),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("DC_CN_MODEL",)
|
||||
RETURN_NAMES = ("DynCraft_CN_model",)
|
||||
FUNCTION = "loadmodel"
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def loadmodel(self, ckpt_name):
|
||||
custom_config = {
|
||||
'ckpt_name': ckpt_name,
|
||||
}
|
||||
if not hasattr(self, 'cn_model') or self.cn_model == None or custom_config != self.current_config:
|
||||
self.current_config = custom_config
|
||||
|
||||
model_path = folder_paths.get_full_path("controlnet", ckpt_name)
|
||||
print(f"Loading ControlNet from: {model_path}")
|
||||
|
||||
cn_config = {
|
||||
"use_checkpoint": True,
|
||||
"image_size": 32, # unused
|
||||
"in_channels": 4,
|
||||
"hint_channels": 3,
|
||||
"model_channels": 320,
|
||||
"attention_resolutions": [4, 2, 1],
|
||||
"num_res_blocks": 2,
|
||||
"channel_mult": [1, 2, 4, 4],
|
||||
"num_head_channels": 64, # need to fix for flash-attn
|
||||
"use_spatial_transformer": True,
|
||||
"use_linear_in_transformer": True,
|
||||
"transformer_depth": 1,
|
||||
"context_dim": 1024,
|
||||
"legacy": False
|
||||
}
|
||||
if "sketch_encoder" in ckpt_name:
|
||||
cn_config["hint_channels"] = 1
|
||||
|
||||
self.cn_model = ControlNet(**cn_config)
|
||||
print("Loading ControlNet")
|
||||
cn_sd = comfy.utils.load_torch_file(model_path)
|
||||
self.cn_model.load_state_dict(cn_sd, strict=True)
|
||||
del cn_sd
|
||||
print("ControlNet loaded")
|
||||
|
||||
controlnet = {
|
||||
'model': self.cn_model,
|
||||
'config': cn_config,
|
||||
}
|
||||
|
||||
return (controlnet,)
|
||||
|
||||
class DownloadAndLoadCLIPModel:
|
||||
@classmethod
|
||||
@@ -241,6 +385,7 @@ class DynamiCrafterModelLoader:
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def loadmodel(self, dtype, ckpt_name, fp8_unet=False):
|
||||
device = mm.get_torch_device()
|
||||
mm.soft_empty_cache()
|
||||
custom_config = {
|
||||
'dtype': dtype,
|
||||
@@ -250,7 +395,9 @@ class DynamiCrafterModelLoader:
|
||||
if not hasattr(self, 'model') or self.model == None or custom_config != self.current_config:
|
||||
self.current_config = custom_config
|
||||
model_path = folder_paths.get_full_path("checkpoints", ckpt_name)
|
||||
ckpt_base_name = os.path.basename(ckpt_name)
|
||||
ckpt_base_name = os.path.basename(model_path)
|
||||
print(f"Loading model from: {model_path}")
|
||||
|
||||
base_name, _ = os.path.splitext(ckpt_base_name)
|
||||
if 'toon' in base_name and '512' in base_name:
|
||||
config_file=os.path.join(script_directory, "configs", "tooncrafter_512_interp.yaml")
|
||||
@@ -268,25 +415,34 @@ class DynamiCrafterModelLoader:
|
||||
|
||||
model_config = config.pop("model", OmegaConf.create())
|
||||
model_config['params']['unet_config']['params']['use_checkpoint']=False
|
||||
self.model = instantiate_from_config(model_config)
|
||||
self.model = load_model_checkpoint(self.model, model_path)
|
||||
self.model.eval()
|
||||
|
||||
if dtype == "auto":
|
||||
try:
|
||||
if mm.should_use_fp16():
|
||||
self.model.to(convert_dtype('fp16'))
|
||||
precision = (convert_dtype('fp16'))
|
||||
elif mm.should_use_bf16():
|
||||
self.model.to(convert_dtype('bf16'))
|
||||
precision = (convert_dtype('bf16'))
|
||||
else:
|
||||
self.model.to(convert_dtype('fp32'))
|
||||
precision = (convert_dtype('fp32'))
|
||||
except:
|
||||
raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.")
|
||||
else:
|
||||
self.model.to(convert_dtype(dtype))
|
||||
precision = (convert_dtype(dtype))
|
||||
|
||||
with (init_empty_weights() if is_accelerate_available else nullcontext()):
|
||||
self.model = instantiate_from_config(model_config)
|
||||
self.model = load_model_checkpoint(self.model, model_path, precision, device)
|
||||
self.model.to(precision).to(device).eval()
|
||||
|
||||
if fp8_unet:
|
||||
self.model.model.diffusion_model = self.model.model.diffusion_model.to(torch.float8_e4m3fn)
|
||||
print(f"Model using dtype: {self.model.dtype}")
|
||||
return (self.model,)
|
||||
|
||||
dcmodel = {
|
||||
'model': self.model,
|
||||
'model_name': ckpt_name,
|
||||
}
|
||||
return (dcmodel,)
|
||||
|
||||
class DynamiCrafterI2V:
|
||||
@classmethod
|
||||
@@ -320,6 +476,8 @@ class DynamiCrafterI2V:
|
||||
"mask": ("MASK",),
|
||||
"frame_window_size": ("INT", {"default": 16, "min": 1, "max": 200, "step": 1}),
|
||||
"frame_window_stride": ("INT", {"default": 4, "min": 1, "max": 200, "step": 1}),
|
||||
"augmentation_level": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 10.0, "step": 0.0001}),
|
||||
"init_noise": ("DCNOISE",),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -328,32 +486,37 @@ class DynamiCrafterI2V:
|
||||
FUNCTION = "process"
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def process(self, model, image, clip_vision, positive, negative, cfg, steps, eta, seed, fs, keep_model_loaded, frames, vae_dtype, frame_window_size=16, frame_window_stride=4, mask=None, image2=None):
|
||||
def process(self, model, image, clip_vision, positive, negative, cfg, steps, eta, seed, fs, keep_model_loaded,
|
||||
frames, vae_dtype, frame_window_size=16, frame_window_stride=4, mask=None, image2=None, augmentation_level=0, init_noise=None):
|
||||
device = mm.get_torch_device()
|
||||
offload_device = mm.unet_offload_device()
|
||||
mm.unload_all_models()
|
||||
mm.soft_empty_cache()
|
||||
|
||||
self.model = model['model']
|
||||
|
||||
torch.manual_seed(seed)
|
||||
dtype = model.dtype
|
||||
dtype = self.model.dtype
|
||||
if vae_dtype == "auto":
|
||||
try:
|
||||
if mm.should_use_bf16():
|
||||
model.first_stage_model.to(convert_dtype('bf16'))
|
||||
self.model.first_stage_model.to(convert_dtype('bf16'))
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype('fp32'))
|
||||
self.model.first_stage_model.to(convert_dtype('fp32'))
|
||||
except:
|
||||
raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.")
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {model.first_stage_model.dtype}")
|
||||
self.model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {self.model.first_stage_model.dtype}")
|
||||
|
||||
self.model = model
|
||||
|
||||
self.model.to(device)
|
||||
autocast_condition = (dtype != torch.float32) and not comfy.model_management.is_device_mps(device)
|
||||
with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=dtype) if autocast_condition else nullcontext():
|
||||
image = image * 2 - 1
|
||||
|
||||
image = image.permute(0, 3, 1, 2).to(dtype).to(device)
|
||||
if augmentation_level > 0:
|
||||
image += torch.randn_like(image) * augmentation_level
|
||||
|
||||
B, C, H, W = image.shape
|
||||
orig_H, orig_W = H, W
|
||||
@@ -362,21 +525,26 @@ class DynamiCrafterI2V:
|
||||
if H % 64 != 0:
|
||||
H = H - (H % 64)
|
||||
if orig_H % 64 != 0 or orig_W % 64 != 0:
|
||||
image = F.interpolate(image, size=(H, W), mode="bicubic")
|
||||
image = F.interpolate(image, size=(H, W), mode="bilinear")
|
||||
|
||||
B, C, H, W = image.shape
|
||||
noise_shape = [B, self.model.model.diffusion_model.out_channels, frames, H // 8, W // 8]
|
||||
|
||||
self.model.first_stage_model.to(device)
|
||||
|
||||
z = get_latent_z(self.model, image.unsqueeze(2)) #bc,1,hw
|
||||
encode_pixels = image.unsqueeze(2) * 2 - 1
|
||||
z = get_latent_z(self.model, encode_pixels) #bc,1,hw
|
||||
|
||||
if image2 is not None:
|
||||
image2 = image2 * 2 - 1
|
||||
image2 = image2.permute(0, 3, 1, 2).to(dtype).to(device)
|
||||
|
||||
if augmentation_level > 0:
|
||||
image2 += torch.randn_like(image2) * augmentation_level
|
||||
|
||||
if image2.shape != image.shape:
|
||||
image2 = F.interpolate(image, size=(H, W), mode="bicubic")
|
||||
z2 = get_latent_z(self.model, image2.unsqueeze(2)) #bc,1,hw
|
||||
image2 = F.interpolate(image, size=(H, W), mode="bilinear")
|
||||
|
||||
encode_pixels = image2.unsqueeze(2) * 2 - 1
|
||||
z2 = get_latent_z(self.model, encode_pixels) #bc,1,hw
|
||||
img_tensor_repeat = repeat(z, 'b c t h w -> b c (repeat t) h w', repeat=frames)
|
||||
img_tensor_repeat = torch.zeros_like(img_tensor_repeat)
|
||||
img_tensor_repeat[:,:,:1,:,:] = z
|
||||
@@ -389,15 +557,20 @@ class DynamiCrafterI2V:
|
||||
self.model.image_proj_model.to(device)
|
||||
text_emb = positive[0][0].to(device)
|
||||
|
||||
cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
|
||||
#cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
|
||||
cond_images = encode_image_masked(clip_vision, image.permute(0, 2, 3, 1), batch_size=0, tiles=4, ratio=0.1).last_hidden_state.to(device)
|
||||
#cond_images = torch.sum(cond_images, dim=0).unsqueeze(0)
|
||||
cond_images = torch.mean(cond_images, dim=0).unsqueeze(0)
|
||||
|
||||
img_emb = self.model.image_proj_model(cond_images)
|
||||
|
||||
imtext_cond = torch.cat([text_emb, img_emb], dim=1)
|
||||
del cond_images, img_emb, text_emb
|
||||
del cond_images, img_emb, text_emb, encode_pixels
|
||||
|
||||
fs = torch.tensor([fs], dtype=torch.long, device=self.model.device)
|
||||
cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat]}
|
||||
cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat], "control_cond": None}
|
||||
|
||||
self.model.control_model = None
|
||||
|
||||
if noise_shape[-1] == 32:
|
||||
timestep_spacing = "uniform"
|
||||
@@ -411,7 +584,7 @@ class DynamiCrafterI2V:
|
||||
uc_emb = negative[0][0].to(device)
|
||||
## process image embedding token
|
||||
if hasattr(self.model, 'embedder'):
|
||||
uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device)
|
||||
uc_img = torch.rand(noise_shape[0],3,224,224).to(self.model.device)
|
||||
## img: b c h w >> b l c
|
||||
uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to(self.model.device)
|
||||
uc_img = self.model.image_proj_model(uc_img)
|
||||
@@ -440,9 +613,30 @@ class DynamiCrafterI2V:
|
||||
mask = mask.permute(0, 2, 1, 3, 4)
|
||||
mask = torch.where(mask < 1.0, torch.tensor(0.0, device=device, dtype=dtype), torch.tensor(1.0, device=device, dtype=dtype))
|
||||
|
||||
if init_noise is not None:
|
||||
if init_noise['analytic_init']:
|
||||
eps=torch.randn_like(init_noise['mu_p'])
|
||||
sigma_p = init_noise['sigma_p']
|
||||
init = (init_noise['mu_p'] + sigma_p*eps).to(dtype).to(device)
|
||||
if noise_shape[2] % init.shape[2] == 0:
|
||||
init = init.repeat(1, 1, noise_shape[2] // init.shape[2], 1, 1)
|
||||
else:
|
||||
raise ValueError("The target dimension size is not an integral multiple of the original dimension size.")
|
||||
else:
|
||||
init = None
|
||||
timestep_spacing = "uniform_trailing"
|
||||
guidance_rescale = 0.7
|
||||
ddpm_from = init_noise['M']
|
||||
|
||||
|
||||
else:
|
||||
init = None
|
||||
ddpm_from = 1000
|
||||
|
||||
#inference
|
||||
ddim_sampler = DDIMSampler(self.model)
|
||||
samples, _ = ddim_sampler.sample(S=steps,
|
||||
samples, _ = ddim_sampler.sample(
|
||||
S=steps,
|
||||
conditioning=cond,
|
||||
batch_size=noise_shape[0],
|
||||
shape=noise_shape[1:],
|
||||
@@ -452,7 +646,7 @@ class DynamiCrafterI2V:
|
||||
eta=eta,
|
||||
temporal_length=noise_shape[2],
|
||||
conditional_guidance_scale_temporal=None,
|
||||
x_T=None,
|
||||
x_T=init,
|
||||
fs=fs,
|
||||
timestep_spacing=timestep_spacing,
|
||||
guidance_rescale=guidance_rescale,
|
||||
@@ -460,7 +654,8 @@ class DynamiCrafterI2V:
|
||||
mask=mask,
|
||||
x0=img_tensor_repeat.clone() if mask is not None else None,
|
||||
frame_window_size = frame_window_size,
|
||||
frame_window_stride = frame_window_stride
|
||||
frame_window_stride = frame_window_stride,
|
||||
ddpm_from=ddpm_from
|
||||
)
|
||||
|
||||
assert not torch.isnan(samples).any().item(), "Resulting tensor containts NaNs. I'm unsure why this happens, changing step count and/or image dimensions might help."
|
||||
@@ -488,6 +683,54 @@ class DynamiCrafterI2V:
|
||||
video = F.interpolate(video.permute(0, 3, 1, 2), size=(final_H, final_W), mode="bicubic").permute(0, 2, 3, 1)
|
||||
last_image = video[-1].unsqueeze(0)
|
||||
return (video, last_image)
|
||||
|
||||
class DynamiCrafterLoadInitNoise:
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
return {"required": {
|
||||
"model": ("DCMODEL",),
|
||||
"M": ("INT", {"default": 1000, "min": 1, "max": 1000, "step": 1}),
|
||||
"analytic_init": ("BOOLEAN", {"default": True}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("DCNOISE", "INT", "INT",)
|
||||
RETURN_NAMES = ("init_noise", "width", "height",)
|
||||
FUNCTION = "load"
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def load(self, model, M, analytic_init):
|
||||
device = mm.get_torch_device()
|
||||
|
||||
model_name = model['model_name']
|
||||
if '512' in model_name:
|
||||
analytic_noise = "initial_noise_512.safetensors"
|
||||
elif '1024' in model_name:
|
||||
analytic_noise = "initial_noise_1024.safetensors"
|
||||
else:
|
||||
print("Can't find matching init_noise for model: ", model_name)
|
||||
model_path = os.path.join(script_directory, 'init_noises', analytic_noise)
|
||||
|
||||
# Analytic-Init:load initial noise
|
||||
dic = comfy.utils.load_torch_file(model_path)
|
||||
expectation_X_0=dic["Expectation_X0"].to(device)
|
||||
tr_Cov_d=dic["Tr_Cov_d"].to(device)
|
||||
|
||||
sqrt_alpha_t=model['model'].get_sqrt_alpha_t_bar(expectation_X_0,torch.tensor([M-1]).to(device))
|
||||
mu_p=sqrt_alpha_t*expectation_X_0
|
||||
alpha_t=sqrt_alpha_t**2
|
||||
sigma_p=torch.sqrt(1-alpha_t + alpha_t*tr_Cov_d)
|
||||
|
||||
init_noise = {
|
||||
"sigma_p": sigma_p,
|
||||
"mu_p": mu_p,
|
||||
"M": M,
|
||||
"analytic_init": analytic_init
|
||||
}
|
||||
width = mu_p.shape[4] * 8
|
||||
height = mu_p.shape[3] * 8
|
||||
|
||||
return (init_noise, width, height)
|
||||
|
||||
class ToonCrafterInterpolation:
|
||||
@classmethod
|
||||
@@ -516,6 +759,10 @@ class ToonCrafterInterpolation:
|
||||
},
|
||||
"optional": {
|
||||
"image_embed_ratio": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
|
||||
"augmentation_level": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 10.0, "step": 0.0001}),
|
||||
"optional_latents": ("LATENT",),
|
||||
"ddpm_from": ("INT", {"default": 1000, "min": 1, "max": 1000, "step": 1}),
|
||||
"controlnet": ("DC_CONTROL",),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -524,27 +771,35 @@ class ToonCrafterInterpolation:
|
||||
FUNCTION = "process"
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def process(self, model, clip_vision, images, positive, negative, cfg, steps, eta, seed, fs, frames, vae_dtype, image_embed_ratio=1.0):
|
||||
def process(self, model, clip_vision, images, positive, negative, cfg, steps, eta, seed, fs, frames,
|
||||
vae_dtype, image_embed_ratio=1.0, augmentation_level=0, optional_latents=None, ddpm_from=1000, controlnet=None):
|
||||
device = mm.get_torch_device()
|
||||
offload_device = mm.unet_offload_device()
|
||||
mm.unload_all_models()
|
||||
mm.soft_empty_cache()
|
||||
|
||||
torch.manual_seed(seed)
|
||||
dtype = model.dtype
|
||||
|
||||
self.model = model['model']
|
||||
|
||||
if controlnet is not None:
|
||||
self.model.control_model = controlnet["model"]
|
||||
else:
|
||||
self.model.control_model = None
|
||||
|
||||
dtype = self.model.dtype
|
||||
if vae_dtype == "auto":
|
||||
try:
|
||||
if mm.should_use_bf16():
|
||||
model.first_stage_model.to(convert_dtype('bf16'))
|
||||
self.model.first_stage_model.to(convert_dtype('bf16'))
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype('fp32'))
|
||||
self.model.first_stage_model.to(convert_dtype('fp32'))
|
||||
except:
|
||||
raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.")
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {model.first_stage_model.dtype}")
|
||||
self.model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {self.model.first_stage_model.dtype}")
|
||||
|
||||
images = images * 2 - 1
|
||||
images = images.permute(0, 3, 1, 2).to(dtype).to(device)
|
||||
|
||||
B, C, H, W = images.shape
|
||||
@@ -556,55 +811,88 @@ class ToonCrafterInterpolation:
|
||||
if orig_H % 64 != 0 or orig_W % 64 != 0:
|
||||
images = F.interpolate(images, size=(H, W), mode="bicubic")
|
||||
|
||||
self.model = model
|
||||
self.model.to(device)
|
||||
|
||||
out = []
|
||||
hidden_states = []
|
||||
|
||||
pbar = comfy.utils.ProgressBar(len(images) - 1)
|
||||
autocast_condition = (dtype != torch.float32) and not comfy.model_management.is_device_mps(device)
|
||||
with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=dtype) if autocast_condition else nullcontext():
|
||||
for i in range(len(images) - 1):
|
||||
for i in range(len(images) - 1) if len(images) > 1 else range(len(images)):
|
||||
videos, videos2 = None, None
|
||||
mm.soft_empty_cache()
|
||||
image = images[i].unsqueeze(0)
|
||||
image2 = images[i+1].unsqueeze(0)
|
||||
if len(images) !=1:
|
||||
image2 = images[i+1].unsqueeze(0)
|
||||
|
||||
B, C, H, W = image.shape
|
||||
noise_shape = [B, self.model.model.diffusion_model.out_channels, frames, H // 8, W // 8]
|
||||
|
||||
self.model.first_stage_model.to(device)
|
||||
|
||||
videos = image.unsqueeze(2) # bc1hw
|
||||
videos = repeat(videos, 'b c t h w -> b c (repeat t) h w', repeat=frames//2)
|
||||
videos2 = image2.unsqueeze(2) # bc1hw
|
||||
videos2 = repeat(videos2, 'b c t h w -> b c (repeat t) h w', repeat=frames//2)
|
||||
videos = torch.cat([videos, videos2], dim=2)
|
||||
if augmentation_level > 0:
|
||||
image += torch.randn_like(image) * augmentation_level
|
||||
image2 += torch.randn_like(image) * augmentation_level
|
||||
|
||||
z, hs = get_latent_z_with_hidden_states(self.model, videos)
|
||||
hidden_states.append(hs)
|
||||
encode_pixels = image.unsqueeze(2) * 2 - 1
|
||||
videos = encode_pixels # bc1hw
|
||||
videos = repeat(videos, 'b c t h w -> b c (repeat t) h w', repeat=frames // 2)
|
||||
|
||||
if len(images) == 1:
|
||||
videos = torch.cat([videos, videos], dim=2)
|
||||
else:
|
||||
encode_pixels = image2.unsqueeze(2) * 2 - 1
|
||||
videos2 = encode_pixels # bc1hw
|
||||
videos2 = repeat(videos2, 'b c t h w -> b c (repeat t) h w', repeat=frames // 2)
|
||||
videos = torch.cat([videos, videos2], dim=2)
|
||||
|
||||
try:
|
||||
z, hs = get_latent_z_with_hidden_states(self.model, videos)
|
||||
hs = [t.to("cpu") for t in hs]
|
||||
hidden_states.append(hs)
|
||||
except:
|
||||
z = get_latent_z(self.model, videos)
|
||||
hidden_states = None
|
||||
|
||||
img_tensor_repeat = torch.zeros_like(z)
|
||||
img_tensor_repeat[:,:,:1,:,:] = z[:,:,:1,:,:]
|
||||
img_tensor_repeat[:,:,-1:,:,:] = z[:,:,-1:,:,:]
|
||||
if len(images) !=1:
|
||||
img_tensor_repeat[:,:,-1:,:,:] = z[:,:,-1:,:,:]
|
||||
|
||||
self.model.first_stage_model.to(offload_device)
|
||||
|
||||
text_emb = positive[0][0].to(device)
|
||||
|
||||
cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
|
||||
cond_images2 = clip_vision.encode_image(image2.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
|
||||
|
||||
#cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))["last_hidden_state"].to(device)
|
||||
#cond_images2 = clip_vision.encode_image(image2.permute(0, 2, 3, 1))["last_hidden_state"].to(device)
|
||||
|
||||
self.model.image_proj_model.to(device)
|
||||
|
||||
cond_images = encode_image_masked(clip_vision, image.permute(0, 2, 3, 1), batch_size=0, tiles=4, ratio=1).last_hidden_state.to(device)
|
||||
cond_images = torch.sum(cond_images, dim=0).unsqueeze(0)
|
||||
img_emb = self.model.image_proj_model(cond_images)
|
||||
img_emb2 = self.model.image_proj_model(cond_images2)
|
||||
img_embeds = img_emb * image_embed_ratio + img_emb2 * (1.0 - image_embed_ratio)
|
||||
if len(images) !=1:
|
||||
cond_images2 = encode_image_masked(clip_vision, image2.permute(0, 2, 3, 1), batch_size=0, tiles=4, ratio=1).last_hidden_state.to(device)
|
||||
cond_images2 = torch.sum(cond_images2, dim=0).unsqueeze(0)
|
||||
img_emb2 = self.model.image_proj_model(cond_images2)
|
||||
img_embeds = img_emb * image_embed_ratio + img_emb2 * (1.0 - image_embed_ratio)
|
||||
else:
|
||||
img_embeds = img_emb
|
||||
|
||||
imtext_cond = torch.cat([text_emb, img_embeds], dim=1)
|
||||
del cond_images, img_emb, img_emb2, text_emb
|
||||
del cond_images, img_emb, text_emb
|
||||
|
||||
fs = torch.tensor([fs], dtype=torch.long, device=self.model.device)
|
||||
cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat]}
|
||||
if comfy.model_management.is_device_mps(device):
|
||||
fs = torch.tensor([fs], dtype=torch.float32, device=self.model.device)
|
||||
else:
|
||||
fs = torch.tensor([fs], dtype=torch.float64, device=self.model.device)
|
||||
|
||||
if controlnet is not None:
|
||||
cn_videos = controlnet["cn_videos"]
|
||||
cn_videos = cn_videos.to(dtype).to(device)
|
||||
else:
|
||||
cn_videos = None
|
||||
|
||||
cond = {"c_crossattn": [imtext_cond], "fs": fs, "c_concat": [img_tensor_repeat], "control_cond": cn_videos}
|
||||
|
||||
if noise_shape[-1] == 32:
|
||||
timestep_spacing = "uniform"
|
||||
@@ -618,7 +906,7 @@ class ToonCrafterInterpolation:
|
||||
uc_emb = negative[0][0].to(device)
|
||||
## process image embedding token
|
||||
if hasattr(self.model, 'embedder'):
|
||||
uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device)
|
||||
uc_img = torch.rand(noise_shape[0],3,224,224).to(self.model.device)
|
||||
## img: b c h w >> b l c
|
||||
uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to(self.model.device)
|
||||
uc_img = self.model.image_proj_model(uc_img)
|
||||
@@ -634,6 +922,16 @@ class ToonCrafterInterpolation:
|
||||
self.model.image_proj_model.to(offload_device)
|
||||
|
||||
#inference
|
||||
if optional_latents is not None:
|
||||
samples_in = optional_latents['samples'].clone().to(device)
|
||||
samples_in = samples_in * 0.18215
|
||||
samples_in = samples_in.unsqueeze(0).permute(0, 2, 1, 3, 4)
|
||||
noise = torch.randn(noise_shape, device=device)
|
||||
samples_in[:, :, 0, :, :] = noise[:, :, 0, :, :]
|
||||
samples_in[:, :, -1, :, :] = noise[:, :, -1, :, :]
|
||||
samples_in = samples_in.to(dtype).to(device)
|
||||
else:
|
||||
samples_in = None
|
||||
|
||||
self.model.model.diffusion_model.to(device)
|
||||
ddim_sampler = DDIMSampler(self.model)
|
||||
@@ -647,7 +945,7 @@ class ToonCrafterInterpolation:
|
||||
eta=eta,
|
||||
temporal_length=noise_shape[2],
|
||||
conditional_guidance_scale_temporal=None,
|
||||
x_T=None,
|
||||
x_T=samples_in,
|
||||
fs=fs,
|
||||
timestep_spacing=timestep_spacing,
|
||||
guidance_rescale=guidance_rescale,
|
||||
@@ -656,11 +954,13 @@ class ToonCrafterInterpolation:
|
||||
x0=None,
|
||||
frame_window_size = 16,
|
||||
frame_window_stride = 4,
|
||||
ddpm_from=ddpm_from
|
||||
)
|
||||
|
||||
print(f"Sampled {i+1} out of {(len(images) - 1)}")
|
||||
assert not torch.isnan(samples).any().item(), "Resulting tensor containts NaNs. I'm unsure why this happens, changing step count and/or image dimensions might help."
|
||||
samples = samples.squeeze(0).permute(1, 0, 2, 3)
|
||||
samples = samples.squeeze(0).permute(1, 0, 2, 3).cpu().to(self.model.first_stage_model.dtype)
|
||||
out.append(samples)
|
||||
pbar.update(1)
|
||||
|
||||
self.model.to(offload_device)
|
||||
mm.soft_empty_cache()
|
||||
@@ -672,7 +972,43 @@ class ToonCrafterInterpolation:
|
||||
"samples": samples,
|
||||
"hidden_states": hidden_states,
|
||||
}
|
||||
|
||||
return (latent,)
|
||||
|
||||
class DynamiCrafterControlnetApply:
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
return {"required": {
|
||||
"controlnet": ("DC_CN_MODEL",),
|
||||
"images": ("IMAGE",),
|
||||
"control_scale": ("FLOAT", {"default": 0.6, "min": 0.0, "max": 1.0, "step": 0.01}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("DC_CONTROL",)
|
||||
RETURN_NAMES = ("controlnet",)
|
||||
FUNCTION = "process"
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def process(self, controlnet, images, control_scale):
|
||||
|
||||
controlnet['model'].control_scale = control_scale
|
||||
|
||||
#images = images * 2.0 - 1.0
|
||||
|
||||
cn_tensor = images.permute(3, 0, 1, 2).unsqueeze(0)
|
||||
print("control frame: ", cn_tensor.shape) # b c t h w
|
||||
print(controlnet["config"])
|
||||
if controlnet["config"]["hint_channels"] == 1:
|
||||
cn_tensor = cn_tensor[:, :1, :, :, :]
|
||||
print("control frame: ", cn_tensor.shape) # b c t h w
|
||||
|
||||
controlnet = {
|
||||
"model": controlnet['model'],
|
||||
"cn_videos": cn_tensor,
|
||||
}
|
||||
|
||||
return (controlnet,)
|
||||
|
||||
class ToonCrafterDecode:
|
||||
@classmethod
|
||||
@@ -702,52 +1038,63 @@ class ToonCrafterDecode:
|
||||
|
||||
def process(self, model, latent, vae_dtype, prune_last_frame=False):
|
||||
device = mm.get_torch_device()
|
||||
offload_device = mm.unet_offload_device()
|
||||
mm.unload_all_models()
|
||||
mm.soft_empty_cache()
|
||||
|
||||
self.model = model['model']
|
||||
samples = latent["samples"]
|
||||
num_samples = samples.shape[0]
|
||||
samples = samples * 0.18215
|
||||
self.model.first_stage_model.to(device)
|
||||
#samples = samples.to(model.first_stage_model.device)
|
||||
|
||||
hs = latent["hidden_states"]
|
||||
|
||||
model.en_and_decode_n_samples_a_time = 16
|
||||
self.model.en_and_decode_n_samples_a_time = 16
|
||||
|
||||
if vae_dtype == "auto":
|
||||
try:
|
||||
if mm.should_use_bf16():
|
||||
model.first_stage_model.to(convert_dtype('bf16'))
|
||||
self.model.first_stage_model.to(convert_dtype('bf16'))
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype('fp32'))
|
||||
self.model.first_stage_model.to(convert_dtype('fp32'))
|
||||
except:
|
||||
raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.")
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {model.first_stage_model.dtype}")
|
||||
self.model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {self.model.first_stage_model.dtype}")
|
||||
out = []
|
||||
iteration_counter = 0
|
||||
for i in range(0, samples.shape[0], 16):
|
||||
|
||||
pbar = comfy.utils.ProgressBar(num_samples // 16)
|
||||
autocast_condition = (self.model.first_stage_model.dtype != torch.float32) and not comfy.model_management.is_device_mps(device)
|
||||
for i in range(0, num_samples, 16):
|
||||
batch_start = i
|
||||
batch_end = min(i + 16, samples.shape[0]) # Ensure we don't go beyond the tensor's size
|
||||
batch_samples = samples[batch_start:batch_end]
|
||||
model.first_stage_model.to(device)
|
||||
if mm.XFORMERS_IS_AVAILABLE:
|
||||
print("Using xformers")
|
||||
additional_decode_kwargs = {'ref_context': hs[iteration_counter]}
|
||||
decoded_images = model.decode_first_stage(batch_samples, **additional_decode_kwargs) #b c t h w
|
||||
else:
|
||||
raise Exception("XFormers not available, it is required for ToonCrafter decoder. Alternatively you can use a standard VAE Decode -node instead, but this has a negative effect on the image quality though.")
|
||||
#print("xformers not available, ToonCrafter does not work well without it.")
|
||||
#decoded_images = model.decode_first_stage(batch_samples) #b c t h w
|
||||
|
||||
video = decoded_images.detach().cpu()
|
||||
video = torch.clamp(video.float(), -1., 1.)
|
||||
video = (video + 1.0) / 2.0
|
||||
video = video.squeeze(0).permute(0, 2, 3, 1)
|
||||
iteration_counter += 1
|
||||
out.append(video)
|
||||
del decoded_images
|
||||
mm.soft_empty_cache()
|
||||
batch_end = min(i + 16, num_samples) # Ensure we don't go beyond the tensor's size
|
||||
batch_samples = samples[batch_start:batch_end].to(self.model.first_stage_model.device)
|
||||
with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=self.model.first_stage_model.dtype) if autocast_condition else nullcontext():
|
||||
#if mm.XFORMERS_IS_AVAILABLE:
|
||||
print(f"Decoding frames {iteration_counter * 16} - {16 + iteration_counter * 16} out of {num_samples} using xformers")
|
||||
if hs is not None:
|
||||
hs_ = hs[iteration_counter]
|
||||
hs_ = [t.to(self.model.first_stage_model.device) for t in hs_]
|
||||
additional_decode_kwargs = {'ref_context': hs_}
|
||||
decoded_images = self.model.decode_first_stage(batch_samples, **additional_decode_kwargs) #b c t h w
|
||||
else:
|
||||
decoded_images = self.model.decode_first_stage(batch_samples) #b c t h w
|
||||
#else:
|
||||
# raise Exception("XFormers not available, it is required for ToonCrafter decoder. Alternatively you can use a standard VAE Decode -node instead, but this has a negative effect on the image quality though.")
|
||||
|
||||
video = decoded_images.detach().cpu()
|
||||
video = torch.clamp(video.float(), -1., 1.)
|
||||
video = (video + 1.0) / 2.0
|
||||
video = video.squeeze(0).permute(0, 2, 3, 1)
|
||||
iteration_counter += 1
|
||||
pbar.update(1)
|
||||
out.append(video)
|
||||
del decoded_images
|
||||
mm.soft_empty_cache()
|
||||
self.model.first_stage_model.to(offload_device)
|
||||
video_out = torch.cat(out, dim=0)
|
||||
model.first_stage_model.to('cpu')
|
||||
if prune_last_frame:
|
||||
video_out = video_out[torch.arange(video_out.shape[0]) % 16!= 0]
|
||||
|
||||
@@ -758,12 +1105,14 @@ class DynamiCrafterBatchInterpolation:
|
||||
def INPUT_TYPES(s):
|
||||
return {"required": {
|
||||
"model": ("DCMODEL",),
|
||||
"clip_vision": ("CLIP_VISION",),
|
||||
"positive": ("CONDITIONING",),
|
||||
"negative": ("CONDITIONING",),
|
||||
"images": ("IMAGE",),
|
||||
"steps": ("INT", {"default": 50, "min": 1, "max": 200, "step": 1}),
|
||||
"cfg": ("FLOAT", {"default": 7.0, "min": 0.0, "max": 20.0, "step": 0.01}),
|
||||
"eta": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 20.0, "step": 0.01}),
|
||||
"frames": ("INT", {"default": 16, "min": 1, "max": 100, "step": 1}),
|
||||
"prompt": ("STRING", {"multiline": True, "default": "",}),
|
||||
"seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff}),
|
||||
"fs": ("INT", {"default": 10, "min": 2, "max": 100, "step": 1}),
|
||||
"keep_model_loaded": ("BOOLEAN", {"default": True}),
|
||||
@@ -785,30 +1134,31 @@ class DynamiCrafterBatchInterpolation:
|
||||
FUNCTION = "process"
|
||||
CATEGORY = "DynamiCrafterWrapper"
|
||||
|
||||
def process(self, model, images, prompt, cfg, steps, eta, seed, fs, keep_model_loaded, frames, vae_dtype, cut_near_keyframes):
|
||||
def process(self, model, images, clip_vision, positive, negative, cfg, steps, eta, seed, fs, keep_model_loaded,
|
||||
frames, vae_dtype, cut_near_keyframes):
|
||||
assert images.shape[0] > 1, "DynamiCrafterBatchInterpolation needs at least 2 images"
|
||||
device = mm.get_torch_device()
|
||||
mm.unload_all_models()
|
||||
mm.soft_empty_cache()
|
||||
|
||||
torch.manual_seed(seed)
|
||||
dtype = model.dtype
|
||||
|
||||
self.model = model['model']
|
||||
dtype = self.model.dtype
|
||||
|
||||
if vae_dtype == "auto":
|
||||
try:
|
||||
if mm.should_use_bf16():
|
||||
model.first_stage_model.to(convert_dtype('bf16'))
|
||||
self.model.first_stage_model.to(convert_dtype('bf16'))
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype('fp32'))
|
||||
self.model.first_stage_model.to(convert_dtype('fp32'))
|
||||
except:
|
||||
raise AttributeError("ComfyUI version too old, can't autodetect properly. Set your dtype manually.")
|
||||
else:
|
||||
model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {model.first_stage_model.dtype}")
|
||||
|
||||
self.model = model
|
||||
self.model.first_stage_model.to(convert_dtype(vae_dtype))
|
||||
print(f"VAE using dtype: {self.model.first_stage_model.dtype}")
|
||||
|
||||
self.model.to(device)
|
||||
images = images * 2 - 1
|
||||
images = images.permute(0, 3, 1, 2).to(dtype).to(device)
|
||||
B, C, H, W = images.shape
|
||||
orig_H, orig_W = H, W
|
||||
@@ -819,7 +1169,6 @@ class DynamiCrafterBatchInterpolation:
|
||||
if orig_H % 64 != 0 or orig_W % 64 != 0:
|
||||
images = F.interpolate(images, size=(H, W), mode="bicubic")
|
||||
|
||||
split_prompt = split_and_trim(prompt)
|
||||
out = []
|
||||
autocast_condition = (dtype != torch.float32) and not comfy.model_management.is_device_mps(device)
|
||||
with torch.autocast(comfy.model_management.get_autocast_device(device), dtype=dtype) if autocast_condition else nullcontext():
|
||||
@@ -832,8 +1181,11 @@ class DynamiCrafterBatchInterpolation:
|
||||
|
||||
self.model.first_stage_model.to(device)
|
||||
|
||||
z = get_latent_z(self.model, image.unsqueeze(2)) #bc,1,hw
|
||||
z2 = get_latent_z(self.model, image2.unsqueeze(2)) #bc,1,hw
|
||||
encode_pixels1 = image * 2 - 1
|
||||
encode_pixels2 = image2 * 2 - 1
|
||||
|
||||
z = get_latent_z(self.model, encode_pixels1.unsqueeze(2)) #bc,1,hw
|
||||
z2 = get_latent_z(self.model, encode_pixels2.unsqueeze(2)) #bc,1,hw
|
||||
img_tensor_repeat = repeat(z, 'b c t h w -> b c (repeat t) h w', repeat=frames)
|
||||
img_tensor_repeat = torch.zeros_like(img_tensor_repeat)
|
||||
img_tensor_repeat[:,:,:1,:,:] = z
|
||||
@@ -841,23 +1193,18 @@ class DynamiCrafterBatchInterpolation:
|
||||
|
||||
self.model.first_stage_model.to('cpu')
|
||||
|
||||
self.model.cond_stage_model.to(device)
|
||||
self.model.embedder.to(device)
|
||||
self.model.image_proj_model.to(device)
|
||||
|
||||
try:
|
||||
text_emb = self.model.get_learned_conditioning([split_prompt[i]])
|
||||
print("Prompt: ", split_prompt[i])
|
||||
except:
|
||||
text_emb = self.model.get_learned_conditioning([split_prompt[0]])
|
||||
print("Prompt: ", split_prompt[0])
|
||||
text_emb = positive[0][0].to(device)
|
||||
cond_images = clip_vision.encode_image(image.permute(0, 2, 3, 1))['last_hidden_state'].to(device)
|
||||
|
||||
cond_images = self.model.embedder(image)
|
||||
img_emb = self.model.image_proj_model(cond_images)
|
||||
imtext_cond = torch.cat([text_emb, img_emb], dim=1)
|
||||
|
||||
fs = torch.tensor([fs], dtype=torch.long, device=self.model.device)
|
||||
cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat]}
|
||||
cond = {"c_crossattn": [imtext_cond], "c_concat": [img_tensor_repeat], "control_cond": None}
|
||||
|
||||
self.model.control_model = None
|
||||
|
||||
if noise_shape[-1] == 32:
|
||||
timestep_spacing = "uniform"
|
||||
@@ -867,13 +1214,14 @@ class DynamiCrafterBatchInterpolation:
|
||||
guidance_rescale = 0.7
|
||||
|
||||
## construct unconditional guidance
|
||||
if cfg != 1.0:
|
||||
uc_emb = self.model.get_learned_conditioning([""])
|
||||
if cfg != 1.0:
|
||||
uc_emb = negative[0][0].to(device)
|
||||
## process image embedding token
|
||||
if hasattr(self.model, 'embedder'):
|
||||
uc_img = torch.zeros(noise_shape[0],3,224,224).to(self.model.device)
|
||||
uc_img = torch.rand(noise_shape[0], 3, 224, 224).to(self.model.device)
|
||||
## img: b c h w >> b l c
|
||||
uc_img = self.model.embedder(uc_img)
|
||||
uc_img = clip_vision.encode_image(uc_img.permute(0, 2, 3, 1))['last_hidden_state'].to(
|
||||
self.model.device)
|
||||
uc_img = self.model.image_proj_model(uc_img)
|
||||
uc_emb = torch.cat([uc_emb, uc_img], dim=1)
|
||||
if isinstance(cond, dict):
|
||||
@@ -883,8 +1231,6 @@ class DynamiCrafterBatchInterpolation:
|
||||
uc = uc_emb
|
||||
else:
|
||||
uc = None
|
||||
|
||||
self.model.cond_stage_model.to('cpu')
|
||||
self.model.embedder.to('cpu')
|
||||
self.model.image_proj_model.to('cpu')
|
||||
|
||||
@@ -965,16 +1311,24 @@ NODE_CLASS_MAPPINGS = {
|
||||
"ToonCrafterDecode": ToonCrafterDecode,
|
||||
"DownloadAndLoadDynamiCrafterModel": DownloadAndLoadDynamiCrafterModel,
|
||||
"DownloadAndLoadCLIPModel": DownloadAndLoadCLIPModel,
|
||||
"DownloadAndLoadCLIPVisionModel": DownloadAndLoadCLIPVisionModel
|
||||
"DownloadAndLoadCLIPVisionModel": DownloadAndLoadCLIPVisionModel,
|
||||
"DynamiCrafterLoadInitNoise": DynamiCrafterLoadInitNoise,
|
||||
"DownloadAndLoadDynamiCrafterCNModel": DownloadAndLoadDynamiCrafterCNModel,
|
||||
"DynamiCrafterControlnetApply": DynamiCrafterControlnetApply,
|
||||
"DynamiCrafterCNLoader": DynamiCrafterCNLoader
|
||||
|
||||
}
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"DynamiCrafterI2V": "DynamiCrafterI2V",
|
||||
"DynamiCrafterModelLoader": "DynamiCrafterModelLoader",
|
||||
"DynamiCrafterBatchInterpolation": "DynamiCrafterBatchInterpolation",
|
||||
"ToonCrafterInterpolation": "ToonCrafterInterpolation",
|
||||
"ToonCrafterDecode": "ToonCrafterDecode",
|
||||
"DownloadAndLoadDynamiCrafterModel": "DownloadAndLoadDynamiCrafterModel",
|
||||
"DownloadAndLoadCLIPModel": "DownloadAndLoadCLIPModel",
|
||||
"DownloadAndLoadCLIPVisionModel": "DownloadAndLoadCLIPVisionModel"
|
||||
"DynamiCrafterModelLoader": "DynamiCrafter ModelLoader",
|
||||
"DynamiCrafterBatchInterpolation": "DynamiCrafter BatchInterpolation",
|
||||
"ToonCrafterInterpolation": "ToonCrafter Interpolation",
|
||||
"ToonCrafterDecode": "ToonCrafter Decode",
|
||||
"DownloadAndLoadDynamiCrafterModel": "(Down)Load DynamiCrafterModel",
|
||||
"DownloadAndLoadCLIPModel": "(Down)Load CLIPModel",
|
||||
"DownloadAndLoadCLIPVisionModel": "(Down)Load CLIPVisionModel",
|
||||
"DynamiCrafterLoadInitNoise": "DynamiCrafter LoadInitNoise",
|
||||
"DownloadAndLoadDynamiCrafterCNModel": "(Down)Load DynamiCrafter CNModel",
|
||||
"DynamiCrafterControlnetApply": "DynamiCrafter ControlnetApply",
|
||||
"DynamiCrafterCNLoader": "DynamiCrafter CNLoader"
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user