5 Commits
Author SHA1 Message Date
kijai b185059ba8 Merge branch 'main' into develop 2025-01-28 16:39:38 +02:00
kijai 2871f9b9d5 Update nodes.py 2024-11-28 18:20:09 +02:00
kijai 17a7aed013 CogVideoConcatLatent 2024-11-28 18:03:47 +02:00
kijai f9c747eff5 mid_image 2024-11-27 01:52:58 +02:00
kijai 2b211b9d1b sageattn 2.0.0 options 2024-11-27 01:16:22 +02:00
3 changed files with 5 additions and 20 deletions
+1 -4
View File
@@ -67,9 +67,8 @@ class CogVideoXPatchEmbed(nn.Module):
post_time_compression_frames,
self.spatial_interpolation_scale,
self.temporal_interpolation_scale,
output_type="pt",
)
pos_embedding = pos_embedding.flatten(0, 1)
pos_embedding = torch.from_numpy(pos_embedding).flatten(0, 1)
joint_pos_embedding = torch.zeros(
1, self.max_text_seq_length + num_patches, self.embed_dim, requires_grad=False
)
@@ -174,8 +173,6 @@ def get_3d_rotary_pos_embed(
grid_t = np.arange(temporal_size, dtype=np.float32)
grid_t = np.linspace(0, temporal_size, temporal_size, endpoint=False, dtype=np.float32)
elif grid_type == "slice":
if max_size is None:
raise ValueError("`max_size` must be provided when `grid_type` is 'slice'")
max_h, max_w = max_size
grid_size_h, grid_size_w = grid_size
grid_h = np.arange(max_h, dtype=np.float32)
+3 -15
View File
@@ -658,9 +658,10 @@ class CogVideoXPipeline(DiffusionPipeline, CogVideoXLoraLoaderMixin):
latent_model_input = self.scheduler.scale_model_input(latent_model_input, t)
counter = torch.zeros_like(latent_model_input)
noise_pred = torch.zeros_like(latent_model_input)
if image_cond_latents is not None:
latent_image_input = torch.cat([image_cond_latents] * 2) if do_classifier_free_guidance else image_cond_latents
latent_model_input = torch.cat([latent_model_input, latent_image_input], dim=2)
# broadcast to batch dimension in a way that's compatible with ONNX/Core ML
timestep = t.expand(latent_model_input.shape[0])
@@ -723,14 +724,7 @@ class CogVideoXPipeline(DiffusionPipeline, CogVideoXLoraLoaderMixin):
noise_pred = noise_pred.float()
else:
for c in context_queue:
print("c:", c)
partial_latent_model_input = latent_model_input[:, c, :, :, :]
if image_cond_latents is not None:
partial_latent_image_input = latent_image_input[:, :len(c), :, :, :]
partial_latent_model_input = torch.cat([partial_latent_model_input,partial_latent_image_input], dim=2)
print(partial_latent_model_input.shape)
if (tora is not None and tora["start_percent"] <= current_step_percentage <= tora["end_percent"]):
if do_classifier_free_guidance:
partial_video_flow_features = tora["video_flow_features"][:, c, :, :, :].repeat(1, 2, 1, 1, 1).contiguous()
@@ -774,13 +768,7 @@ class CogVideoXPipeline(DiffusionPipeline, CogVideoXLoraLoaderMixin):
if i == len(timesteps) - 1 or ((i + 1) > num_warmup_steps and (i + 1) % self.scheduler.order == 0):
progress_bar.update()
if callback is not None:
alpha_prod_t = self.scheduler.alphas_cumprod[t]
beta_prod_t = 1 - alpha_prod_t
callback_tensor = (alpha_prod_t**0.5) * latent_model_input[0][:, :16, :, :] - (beta_prod_t**0.5) * noise_pred.detach()[0]
callback(i, callback_tensor * 5, None, num_inference_steps)
else:
comfy_pbar.update(1)
comfy_pbar.update(1)
# region sampling
else:
+1 -1
View File
@@ -1,5 +1,5 @@
huggingface_hub
diffusers>=0.33.1
diffusers>=0.31.0
accelerate>=0.33.0
einops
peft