# Taken from: https://github.com/Stability-AI/generative-models/blob/main/scripts/sampling/simple_video_sample.py from omegaconf import OmegaConf from .util import default, instantiate_from_config import torch import folder_paths from typing import Optional def load_model( config: str, device: str, num_frames: int, num_steps: int, checkpoint: Optional[str] = None, ): config = OmegaConf.load(config) if checkpoint: config.model.params.ckpt_path = checkpoint if device == "cuda": config.model.params.conditioner_config.params.emb_models[ 0 ].params.open_clip_embedding_config.params.init_device = device config.model.params.sampler_config.params.num_steps = num_steps config.model.params.sampler_config.params.guider_config.params.num_frames = ( num_frames ) if device == "cuda": with torch.device(device): model = instantiate_from_config(config.model).to(device).eval() else: model = instantiate_from_config(config.model).to(device).eval() # filter = DeepFloydDataFiltering(verbose=False, device=device) return model #, filter def sample( model, image, num_frames: Optional[int] = None, num_steps: Optional[int] = None, version: str = "svd", fps_id: int = 6, motion_bucket_id: int = 127, cond_aug: float = 0.02, seed: int = 23, decoding_t: int = 14, # Number of frames decoded at a time! This eats most VRAM. Reduce if necessary. device: Optional[str] = None, ): """ Simple script to generate a single sample conditioned on an image `input_path` or multiple images, one for each image file in folder `input_path`. If you run out of VRAM, try decreasing `decoding_t`. """ if device is None: device = "cuda" if torch.cuda.is_available() else "cpu" if version == "svd": num_frames = default(num_frames, 14) num_steps = default(num_steps, 25) model_config = "scripts/sampling/configs/svd.yaml" elif version == "svd_xt": num_frames = default(num_frames, 25) num_steps = default(num_steps, 30) model_config = "scripts/sampling/configs/svd_xt.yaml" elif version == "svd_image_decoder": num_frames = default(num_frames, 14) num_steps = default(num_steps, 25) model_config = "scripts/sampling/configs/svd_image_decoder.yaml" elif version == "svd_xt_image_decoder": num_frames = default(num_frames, 25) num_steps = default(num_steps, 30) model_config = "scripts/sampling/configs/svd_xt_image_decoder.yaml" else: raise ValueError(f"Version {version} does not exist.") model = load_model( model_config, device, num_frames, num_steps, ) torch.manual_seed(seed) if image.mode == "RGBA": image = image.convert("RGB") w, h = image.size if h % 64 != 0 or w % 64 != 0: width, height = map(lambda x: x - x % 64, (w, h)) image = image.resize((width, height)) print( f"WARNING: Your image is of size {h}x{w} which is not divisible by 64. We are resizing to {height}x{width}!" ) image = ToTensor()(image) image = image * 2.0 - 1.0 image = image.unsqueeze(0).to(device) H, W = image.shape[2:] assert image.shape[1] == 3 F = 8 C = 4 shape = (num_frames, C, H // F, W // F) if (H, W) != (576, 1024): print( "WARNING: The conditioning frame you provided is not 576x1024. This leads to suboptimal performance as model was only trained on 576x1024. Consider increasing `cond_aug`." ) if motion_bucket_id > 255: print( "WARNING: High motion bucket! This may lead to suboptimal performance." ) if fps_id < 5: print("WARNING: Small fps value! This may lead to suboptimal performance.") if fps_id > 30: print("WARNING: Large fps value! This may lead to suboptimal performance.") value_dict = {} value_dict["motion_bucket_id"] = motion_bucket_id value_dict["fps_id"] = fps_id value_dict["cond_aug"] = cond_aug value_dict["cond_frames_without_noise"] = image value_dict["cond_frames"] = image + cond_aug * torch.randn_like(image) value_dict["cond_aug"] = cond_aug with torch.no_grad(): with torch.autocast(device): batch, batch_uc = get_batch( get_unique_embedder_keys_from_conditioner(model.conditioner), value_dict, [1, num_frames], T=num_frames, device=device, ) c, uc = model.conditioner.get_unconditional_conditioning( batch, batch_uc=batch_uc, force_uc_zero_embeddings=[ "cond_frames", "cond_frames_without_noise", ], ) for k in ["crossattn", "concat"]: uc[k] = repeat(uc[k], "b ... -> b t ...", t=num_frames) uc[k] = rearrange(uc[k], "b t ... -> (b t) ...", t=num_frames) c[k] = repeat(c[k], "b ... -> b t ...", t=num_frames) c[k] = rearrange(c[k], "b t ... -> (b t) ...", t=num_frames) randn = torch.randn(shape, device=device) additional_model_inputs = {} additional_model_inputs["image_only_indicator"] = torch.zeros( 2, num_frames ).to(device) additional_model_inputs["num_video_frames"] = batch["num_video_frames"] def denoiser(input, sigma, c): return model.denoiser( model.model, input, sigma, c, **additional_model_inputs ) samples_z = model.sampler(denoiser, randn, cond=c, uc=uc) model.en_and_decode_n_samples_a_time = decoding_t samples_x = model.decode_first_stage(samples_z) samples = torch.clamp((samples_x + 1.0) / 2.0, min=0.0, max=1.0) # samples = embed_watermark(samples) # samples = filter(samples) vid = ( (rearrange(samples, "t c h w -> t h w c") * 255) .cpu() .numpy() .astype(np.uint8) ) return vid # shape: (num_frames, height, width, channels), type: np.uint8