diff --git a/.idea/.gitignore b/.idea/.gitignore new file mode 100644 index 0000000..26d3352 --- /dev/null +++ b/.idea/.gitignore @@ -0,0 +1,3 @@ +# Default ignored files +/shelf/ +/workspace.xml diff --git a/.idea/Diffusion360_ComfyUI.iml b/.idea/Diffusion360_ComfyUI.iml new file mode 100644 index 0000000..aa93672 --- /dev/null +++ b/.idea/Diffusion360_ComfyUI.iml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/inspectionProfiles/Project_Default.xml b/.idea/inspectionProfiles/Project_Default.xml new file mode 100644 index 0000000..fbdd2b4 --- /dev/null +++ b/.idea/inspectionProfiles/Project_Default.xml @@ -0,0 +1,25 @@ + + + + \ No newline at end of file diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml new file mode 100644 index 0000000..105ce2d --- /dev/null +++ b/.idea/inspectionProfiles/profiles_settings.xml @@ -0,0 +1,6 @@ + + + + \ No newline at end of file diff --git a/.idea/markdown-navigator-enh.xml b/.idea/markdown-navigator-enh.xml new file mode 100644 index 0000000..a8fcc84 --- /dev/null +++ b/.idea/markdown-navigator-enh.xml @@ -0,0 +1,10 @@ + + + + + + + + + + \ No newline at end of file diff --git a/.idea/markdown-navigator.xml b/.idea/markdown-navigator.xml new file mode 100644 index 0000000..40d1864 --- /dev/null +++ b/.idea/markdown-navigator.xml @@ -0,0 +1,55 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + \ No newline at end of file diff --git a/.idea/misc.xml b/.idea/misc.xml new file mode 100644 index 0000000..80131ab --- /dev/null +++ b/.idea/misc.xml @@ -0,0 +1,4 @@ + + + + \ No newline at end of file diff --git a/.idea/modules.xml b/.idea/modules.xml new file mode 100644 index 0000000..d0dfb90 --- /dev/null +++ b/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/vcs.xml b/.idea/vcs.xml new file mode 100644 index 0000000..35eb1dd --- /dev/null +++ b/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/Diffusion360_nodes.py b/Diffusion360_nodes.py new file mode 100644 index 0000000..a147de8 --- /dev/null +++ b/Diffusion360_nodes.py @@ -0,0 +1,69 @@ +import comfy +from comfy.ldm.models.autoencoder import AutoencoderKL +from comfy.samplers import KSAMPLER +from .utils import decode_tiled_blended, decode_tiled_blended_, common_ksampler + + +class VAEDecodeTiledBlended: + @classmethod + def INPUT_TYPES(s): + return {"required": {"samples": ("LATENT", ), "vae": ("VAE", ), + "tile_size": ("INT", {"default": 512, "min": 320, "max": 4096, "step": 64}) + }} + RETURN_TYPES = ("IMAGE",) + FUNCTION = "decode" + + CATEGORY = "Diffusion360" + + def decode(self, vae, samples, tile_size): + vae.decode_tiled = decode_tiled_blended.__get__(vae, AutoencoderKL) + vae.decode_tiled_blended_ = decode_tiled_blended_.__get__(vae, AutoencoderKL) + return (vae.decode_tiled(samples["samples"], tile_x=tile_size // 8, tile_y=tile_size // 8, ), ) + + +class Diffusion360Sampler: + @classmethod + def INPUT_TYPES(s): + return {"required": + {"model": ("MODEL",), + "add_noise": (["enable", "disable"], ), + "noise_seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff}), + "steps": ("INT", {"default": 20, "min": 1, "max": 10000}), + "cfg": ("FLOAT", {"default": 8.0, "min": 0.0, "max": 100.0, "step":0.1, "round": 0.01}), + "sampler_name": (['euler_blend'], ), + "scheduler": (comfy.samplers.KSampler.SCHEDULERS, ), + "positive": ("CONDITIONING", ), + "negative": ("CONDITIONING", ), + "latent_image": ("LATENT", ), + "start_at_step": ("INT", {"default": 0, "min": 0, "max": 10000}), + "end_at_step": ("INT", {"default": 10000, "min": 0, "max": 10000}), + "return_with_leftover_noise": (["disable", "enable"], ), + } + } + + RETURN_TYPES = ("LATENT",) + FUNCTION = "sample" + + CATEGORY = "Diffusion360" + + def sample(self, model, add_noise, noise_seed, steps, cfg, sampler_name, scheduler, positive, negative, latent_image, start_at_step, end_at_step, return_with_leftover_noise, denoise=1.0): + force_full_denoise = True + if return_with_leftover_noise == "enable": + force_full_denoise = False + disable_noise = False + if add_noise == "disable": + disable_noise = True + return common_ksampler(model, noise_seed, steps, cfg, sampler_name, scheduler, positive, negative, latent_image, denoise=denoise, disable_noise=disable_noise, start_step=start_at_step, last_step=end_at_step, force_full_denoise=force_full_denoise) + + + +NODE_CLASS_MAPPINGS = { + "VAEDecodeTiledBlended": VAEDecodeTiledBlended, + "Diffusion360Sampler": Diffusion360Sampler, +} + +# A dictionary that contains the friendly/humanly readable titles for the nodes +NODE_DISPLAY_NAME_MAPPINGS = { + "VAEDecodeTiledBlended": "VAE Decode (Tiled Blended)", + "Diffusion360Sampler": "Diffusion360Sampler", +} diff --git a/README.md b/README.md index f31fc5f..d29d02d 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,19 @@ # Diffusion360_ComfyUI ComfyUI plugin of https://github.com/ArcherFMY/SD-T2I-360PanoImage + +## features +- [x] base t2i-pipeline for generating 512*1024 panorama image from text input +- [ ] SR-pipeline for higher-resolution panorama image generation + +## Installation +1. Install ComfyUI following https://github.com/comfyanonymous/ComfyUI +2. Install this plugin by the following commands. + ``` + cd ComfyUI/custom_nodes + git clone https://github.com/ArcherFMY/Diffusion360_ComfyUI + ``` + +## Usage +add 'Diffusion360Sampler' and 'VAE Decode (Tiled Blended)' from 'Diffusion360' category, and the pipeline looks like +![pipeline](pipeline.png) + diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..99d2a14 --- /dev/null +++ b/__init__.py @@ -0,0 +1,12 @@ +from .Diffusion360_nodes import VAEDecodeTiledBlended, Diffusion360Sampler + +NODE_CLASS_MAPPINGS = { + "VAEDecodeTiledBlended": VAEDecodeTiledBlended, + "Diffusion360Sampler": Diffusion360Sampler, +} + +# A dictionary that contains the friendly/humanly readable titles for the nodes +NODE_DISPLAY_NAME_MAPPINGS = { + "VAEDecodeTiledBlended": "VAE Decode (Tiled Blended)", + "Diffusion360Sampler": "Diffusion360Sampler", +} diff --git a/__pycache__/Diffusion360_nodes.cpython-310.pyc b/__pycache__/Diffusion360_nodes.cpython-310.pyc new file mode 100644 index 0000000..5a513e8 Binary files /dev/null and b/__pycache__/Diffusion360_nodes.cpython-310.pyc differ diff --git a/__pycache__/__init__.cpython-310.pyc b/__pycache__/__init__.cpython-310.pyc new file mode 100644 index 0000000..090c0f9 Binary files /dev/null and b/__pycache__/__init__.cpython-310.pyc differ diff --git a/__pycache__/utils.cpython-310.pyc b/__pycache__/utils.cpython-310.pyc new file mode 100644 index 0000000..2340720 Binary files /dev/null and b/__pycache__/utils.cpython-310.pyc differ diff --git a/pipeline.png b/pipeline.png new file mode 100644 index 0000000..d045489 Binary files /dev/null and b/pipeline.png differ diff --git a/utils.py b/utils.py new file mode 100644 index 0000000..13119c6 --- /dev/null +++ b/utils.py @@ -0,0 +1,194 @@ +import torch +import comfy +from comfy import model_management +from tqdm.auto import trange +import comfy.k_diffusion.utils as utils +import latent_preview +from comfy.samplers import KSAMPLER, ksampler, CFGGuider +from comfy.extra_samplers import uni_pc + + +def sampler_object(name): + if name == "uni_pc": + sampler = KSAMPLER(uni_pc.sample_unipc) + elif name == "uni_pc_bh2": + sampler = KSAMPLER(uni_pc.sample_unipc_bh2) + elif name == "ddim": + sampler = ksampler("euler", inpaint_options={"random": True}) + else: + sampler = ksampler(name) + return sampler + + +def sample_(model, noise, positive, negative, cfg, device, sampler, sigmas, model_options={}, latent_image=None, denoise_mask=None, callback=None, disable_pbar=False, seed=None): + cfg_guider = CFGGuider(model) + cfg_guider.set_conds(positive, negative) + cfg_guider.set_cfg(cfg) + return cfg_guider.sample(noise, latent_image, sampler, sigmas, denoise_mask, callback, disable_pbar, seed) + + +def sample(self, noise, positive, negative, cfg, latent_image=None, start_step=None, last_step=None, + force_full_denoise=False, denoise_mask=None, sigmas=None, callback=None, disable_pbar=False, seed=None): + if sigmas is None: + sigmas = self.sigmas + + if last_step is not None and last_step < (len(sigmas) - 1): + sigmas = sigmas[:last_step + 1] + if force_full_denoise: + sigmas[-1] = 0 + + if start_step is not None: + if start_step < (len(sigmas) - 1): + sigmas = sigmas[start_step:] + else: + if latent_image is not None: + return latent_image + else: + return torch.zeros_like(noise) + + sampler = sampler_object(self.sampler) + sampler.sampler_function = sample_euler_blend + + return sample_(self.model, noise, positive, negative, cfg, self.device, sampler, sigmas, self.model_options, + latent_image=latent_image, denoise_mask=denoise_mask, callback=callback, disable_pbar=disable_pbar, + seed=seed) + + +def common_ksampler(model, seed, steps, cfg, sampler_name, scheduler, positive, negative, latent, denoise=1.0, disable_noise=False, start_step=None, last_step=None, force_full_denoise=False): + latent_image = latent["samples"] + if disable_noise: + noise = torch.zeros(latent_image.size(), dtype=latent_image.dtype, layout=latent_image.layout, device="cpu") + else: + batch_inds = latent["batch_index"] if "batch_index" in latent else None + noise = comfy.sample.prepare_noise(latent_image, seed, batch_inds) + + noise_mask = None + if "noise_mask" in latent: + noise_mask = latent["noise_mask"] + + callback = latent_preview.prepare_callback(model, steps) + disable_pbar = not comfy.utils.PROGRESS_BAR_ENABLED + + sampler = comfy.samplers.KSampler(model, steps=steps, device=model.load_device, sampler=sampler_name, scheduler=scheduler, denoise=denoise, model_options=model.model_options) + sampler.sample = sample.__get__(sampler, KSAMPLER) + samples = sampler.sample(noise, positive, negative, cfg=cfg, latent_image=latent_image, start_step=start_step, last_step=last_step, force_full_denoise=force_full_denoise, denoise_mask=noise_mask, sigmas=None, callback=callback, disable_pbar=disable_pbar, seed=seed) + samples = samples.to(comfy.model_management.intermediate_device()) + + out = latent.copy() + out["samples"] = samples + return (out, ) + + +def to_d(x, sigma, denoised): + """Converts a denoiser output to a Karras ODE derivative.""" + return (x - denoised) / utils.append_dims(sigma, x.ndim) + + +@torch.no_grad() +def blend_h(a, b, blend_extent): + blend_extent = min(a.shape[3], b.shape[3], blend_extent) + for x in range(blend_extent): + b[:, :, :, x] = a[:, :, :, -blend_extent + + x] * (1 - x / blend_extent) + b[:, :, :, x] * ( + x / blend_extent) + return b + + +@torch.no_grad() +def sample_euler_blend(model, x, sigmas, extra_args=None, callback=None, disable=None, s_churn=0., s_tmin=0., s_tmax=float('inf'), s_noise=1.): + """Implements Algorithm 2 (Euler steps) from Karras et al. (2022).""" + extra_args = {} if extra_args is None else extra_args + s_in = x.new_ones([x.shape[0]]) + w = x.shape[-1] + for i in trange(len(sigmas) - 1, disable=disable): + gamma = min(s_churn / (len(sigmas) - 1), 2 ** 0.5 - 1) if s_tmin <= sigmas[i] <= s_tmax else 0. + sigma_hat = sigmas[i] * (gamma + 1) + if gamma > 0: + eps = torch.randn_like(x) * s_noise + x = x + eps * (sigma_hat ** 2 - sigmas[i] ** 2) ** 0.5 + denoised = model(x, sigma_hat * s_in, **extra_args) + d = to_d(x, sigma_hat, denoised) + if callback is not None: + callback({'x': x, 'i': i, 'sigma': sigmas[i], 'sigma_hat': sigma_hat, 'denoised': denoised}) + dt = sigmas[i + 1] - sigma_hat + # Euler method + x = x + d * dt + x = blend_h(x, x, 4) + x = blend_h(x, x, 4) + x = x[:, :, :, :w] + return x + + +def decode_tiled_blended_(self, samples, tile_x=64, tile_y=64, overlap=16): + steps = samples.shape[0] * comfy.utils.get_tiled_scale_steps(samples.shape[3], samples.shape[2], tile_x, tile_y, + overlap) + steps += samples.shape[0] * comfy.utils.get_tiled_scale_steps(samples.shape[3], samples.shape[2], tile_x // 2, + tile_y * 2, overlap) + steps += samples.shape[0] * comfy.utils.get_tiled_scale_steps(samples.shape[3], samples.shape[2], tile_x * 2, + tile_y // 2, overlap) + pbar = comfy.utils.ProgressBar(steps) + + decode_fn = lambda a: self.first_stage_model.decode(a.to(self.vae_dtype).to(self.device)).float() + output = self.process_output( + (tiled_scale_blended(samples, decode_fn, tile_x // 2, tile_y * 2, overlap, + upscale_amount=self.upscale_ratio, output_device=self.output_device, + pbar=pbar) + + tiled_scale_blended(samples, decode_fn, tile_x * 2, tile_y // 2, overlap, + upscale_amount=self.upscale_ratio, output_device=self.output_device, + pbar=pbar) + + tiled_scale_blended(samples, decode_fn, tile_x, tile_y, overlap, upscale_amount=self.upscale_ratio, + output_device=self.output_device, pbar=pbar)) + / 3.0) + return output + + +def decode_tiled_blended(self, samples, tile_x=64, tile_y=64, overlap = 16): + model_management.load_model_gpu(self.patcher) + output = self.decode_tiled_blended_(samples, tile_x, tile_y, overlap) + return output.movedim(1, -1) + + +def blend_h(a, b, blend_extent): + blend_extent = min(a.shape[3], b.shape[3], blend_extent) + for x in range(blend_extent): + b[:, :, :, x] = a[:, :, :, -blend_extent + + x] * (1 - x / blend_extent) + b[:, :, :, x] * ( + x / blend_extent) + return b + + +def tiled_scale_blended(samples, function, tile_x=64, tile_y=64, overlap = 8, upscale_amount = 4, out_channels = 3, output_device="cpu", pbar = None): + output = torch.empty((samples.shape[0], out_channels, round(samples.shape[2] * upscale_amount), round(samples.shape[3] * upscale_amount)), device=output_device) + for b in range(samples.shape[0]): + s = samples[b:b+1] + out = torch.zeros((s.shape[0], out_channels, round(s.shape[2] * upscale_amount), round(s.shape[3] * upscale_amount)), device=output_device) + out_div = torch.zeros((s.shape[0], out_channels, round(s.shape[2] * upscale_amount), round(s.shape[3] * upscale_amount)), device=output_device) + w = samples.shape[3] + samples = torch.cat([samples, samples[:, :, :, :w // 4]], dim=-1) + s = samples[b:b + 1] + for y in range(0, s.shape[2], tile_y - overlap): + # for x in range(0, s.shape[3], tile_x - overlap): + # x = max(0, min(s.shape[-1] - overlap, x)) + y = max(0, min(s.shape[-2] - overlap, y)) + s_in = s[:,:,y:y+tile_y,:] + + ps = function(s_in).to(output_device) + ps = blend_h( + ps[:,:,:,w*upscale_amount:], + ps[:,:,:,:w*upscale_amount], + w // 4 * upscale_amount + ) + mask = torch.ones_like(ps) + feather = round(overlap * upscale_amount) + for t in range(feather): + mask[:,:,t:1+t,:] *= ((1.0/feather) * (t + 1)) + mask[:,:,mask.shape[2] -1 -t: mask.shape[2]-t,:] *= ((1.0/feather) * (t + 1)) + mask[:,:,:,t:1+t] *= ((1.0/feather) * (t + 1)) + mask[:,:,:,mask.shape[3]- 1 - t: mask.shape[3]- t] *= ((1.0/feather) * (t + 1)) + out[:,:,round(y*upscale_amount):round((y+tile_y)*upscale_amount),:] += ps * mask + out_div[:,:,round(y*upscale_amount):round((y+tile_y)*upscale_amount),:] += mask + if pbar is not None: + pbar.update(1) + + output[b:b+1] = out/out_div + return output \ No newline at end of file