Version 1.3

This commit is contained in:
Stefano Flore
2024-10-16 14:31:34 +02:00
committed by GitHub
parent 0ed383c4e8
commit 556ee51635
3 changed files with 236 additions and 57 deletions
+106 -3
View File
@@ -7,15 +7,118 @@ Original project: https://github.com/THUDM/CogVideo
![Overview](/assets/screenshot.png)
## Version 1.2
## Basic Parameters
1. prompt (STRING)
Description: The text that guides the video generation.
Type: Multiline string
Impact: Directly influences the content and style of the generated video.
2. image (IMAGE)
Description: The input image from which to start the video generation.
Type: Image
Impact: Serves as the starting point for the video, strongly influencing the first frames.
3. num_frames (INT)
Description: The total number of frames to generate.
Default: 98
Range: 49 - 2^31-1 (practically unlimited)
Step: 49
Impact: Determines the length of the final video.
4. num_inference_steps (INT)
Description: The number of inference steps for each frame.
Default: 10
Range: 1 - 1000
Impact: Influences the quality and detail of each generated frame. More steps generally mean higher quality but longer processing times.
5. guidance_scale (FLOAT)
Description: Controls how closely the model follows the prompt.
Default: 6.0
Range: 0.1 - 30.0
Impact: Higher values produce results more faithful to the prompt but can lead to artifacts.
6. use_dynamic_cfg (BOOLEAN)
Description: Enables or disables Dynamic Guided Configuration.
Default: True
Impact: When enabled, it can improve video consistency and quality.
7. seed (INT)
Description: The seed for random generation.
Default: 0
Range: 0 - 99999999999999
Impact: Controls result reproducibility. The same seed will always produce the same output with the same parameters.
Optical Flow Interpolation Parameters
8. interpolation_factor (INT)
Description: Determines how many intermediate frames to create between each pair of original frames.
Default: 1
Range: 1 - 7 (steps of 2)
Impact: Increases video smoothness by adding interpolated frames. A value of 1 doubles the number of frames, 3 quadruples it, etc.
9. flow_precision (FLOAT)
Description: Controls the level of detail in optical flow calculation.
Default: 0.5
Range: 0.1 - 1.0
Impact: Higher values produce more precise optical flow but require more computation time.
10. motion_threshold (FLOAT)
Description: Determines the minimum amount of movement required to apply interpolation.
Default: 0.1
Range: 0.0 - 1.0
Impact: Lower values interpolate even small movements, higher values only significant movements.
11. smoothness (FLOAT)
Description: Controls how "smooth" the interpolated movement should be.
Default: 0.5
Range: 0.0 - 1.0
Impact: Higher values produce smoother transitions but may reduce movement details.
12. flow_method (COMBO)
Description: The method used to calculate optical flow.
Options: ["Farneback", "TV-L1", "DIS"]
Impact: Each method has different characteristics in terms of accuracy and speed.
13. edge_mode (COMBO)
Description: How to handle image edges during interpolation.
Options: ["Replicate", "Reflect", "Wrap", "Constant"]
Impact: Influences the appearance of edges in interpolated frames.
14. interpolation_strength (FLOAT)
Description: How heavily to apply the calculated interpolation.
Default: 1.0
Range: 0.0 - 1.0
Impact: Lower values produce subtler interpolation, higher values more pronounced interpolation.
## Updates
### Version 1.3
Added opticalflow frame iterpolation
### Version 1.2
Update experimental node for video duration extension
## Version 1.1
### Version 1.1
New experimental node for video duration extension
## Version 1.0
### Version 1.0
First release
+113 -38
View File
@@ -1,6 +1,6 @@
# CogVideoX
# Created by AI Wiz Art (Stefano Flore)
# Version: 1.2
# Version: 1.3
# https://stefanoflore.it
# https://ai-wiz.art
@@ -18,6 +18,7 @@ from diffusers import (
from huggingface_hub import snapshot_download
from PIL import Image
from tqdm import tqdm
from scipy.ndimage import gaussian_filter
def download_model_if_needed(model_name, local_dir="models/CogVideoX"):
model_dir = os.path.join(local_dir, model_name.split("/")[-1])
@@ -166,7 +167,7 @@ class CogVideoXImageToVideoNode:
except Exception as e:
print(f"Error during video generation: {str(e)}")
raise
class CogVideoXImageToVideoNodeExtended:
pipe = None
@@ -181,11 +182,18 @@ class CogVideoXImageToVideoNodeExtended:
"guidance_scale": ("FLOAT", {"default": 6.0, "min": 0.1, "max": 30.0}),
"use_dynamic_cfg": ("BOOLEAN", {"default": True}),
"seed": ("INT", {"default": 0, "min": 0, "max": 99999999999999}),
"interpolation_factor": ("INT", {"default": 1, "min": 1, "max": 7, "step": 2}),
"flow_precision": ("FLOAT", {"default": 0.5, "min": 0.1, "max": 1.0, "step": 0.1}),
"motion_threshold": ("FLOAT", {"default": 0.1, "min": 0.0, "max": 1.0, "step": 0.05}),
"smoothness": ("FLOAT", {"default": 0.5, "min": 0.0, "max": 1.0, "step": 0.1}),
"flow_method": (["Farneback", "TV-L1", "DIS"],),
"edge_mode": (["Replicate", "Reflect", "Wrap", "Constant"],),
"interpolation_strength": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.1}),
}
}
RETURN_TYPES = ("VIDEO",)
RETURN_NAMES = ("video",)
RETURN_TYPES = ("VIDEO","VIDEO",)
RETURN_NAMES = ("normal video","interpolated video",)
FUNCTION = "generate_extended_video"
CATEGORY = "AI WizArt/CogVideoX"
@@ -215,7 +223,8 @@ class CogVideoXImageToVideoNodeExtended:
print(f"Model {model_name} downloaded successfully.")
return model_dir
def generate_extended_video(self, prompt, image, num_frames, num_inference_steps, guidance_scale, use_dynamic_cfg, seed):
def generate_extended_video(self, prompt, image, num_frames, num_inference_steps, guidance_scale, use_dynamic_cfg, seed,
interpolation_factor, flow_precision, motion_threshold, smoothness, flow_method, edge_mode, interpolation_strength):
num_frames = max(49, (num_frames // 49) * 49)
try:
@@ -254,7 +263,6 @@ class CogVideoXImageToVideoNodeExtended:
if new_frames:
all_frames.extend(new_frames)
# Aggiorna last_frame con l'ultimo frame generato
last_frame = Image.fromarray(new_frames[-1])
else:
print("Warning: No new frames generated in this iteration")
@@ -262,13 +270,105 @@ class CogVideoXImageToVideoNodeExtended:
progress_bar.update(len(new_frames))
all_frames = all_frames[:num_frames]
print(f"Final video length: {len(all_frames)} frames")
print(f"Final video length before interpolation: {len(all_frames)} frames")
return (all_frames,)
interpolated_frames = self.apply_optical_flow_interpolation(all_frames, interpolation_factor, flow_precision,
motion_threshold, smoothness, flow_method, edge_mode,
interpolation_strength)
print(f"Final video length after interpolation: {len(interpolated_frames)} frames")
return (all_frames,interpolated_frames,)
except Exception as e:
print(f"Error during extended video generation: {str(e)}")
raise
def apply_optical_flow_interpolation(self, frames, factor, precision, threshold, smoothness, method, edge_mode, strength):
interpolated = []
flow_params = {
'pyr_scale': 0.5,
'levels': 3,
'winsize': 15,
'iterations': 3,
'poly_n': 5,
'poly_sigma': 1.2,
'flags': 0
}
with tqdm(total=len(frames) - 1, desc="Applying optical flow interpolation") as pbar:
for i in range(len(frames) - 1):
frame1 = frames[i]
frame2 = frames[i + 1]
gray1 = cv2.cvtColor(frame1, cv2.COLOR_RGB2GRAY)
gray2 = cv2.cvtColor(frame2, cv2.COLOR_RGB2GRAY)
if method == "Farneback":
flow = cv2.calcOpticalFlowFarneback(gray1, gray2, None, **flow_params)
elif method == "TV-L1":
try:
optical_flow = cv2.optflow.DualTVL1OpticalFlow_create()
except AttributeError:
try:
optical_flow = cv2.createOptFlow_DualTVL1()
except AttributeError:
print("TV-L1 optical flow not available. Using Farneback method instead.")
flow = cv2.calcOpticalFlowFarneback(gray1, gray2, None, **flow_params)
else:
flow = optical_flow.calc(gray1, gray2, None)
else:
flow = optical_flow.calc(gray1, gray2, None)
elif method == "DIS":
try:
flow = cv2.DISOpticalFlow_create(cv2.DISOPTICAL_FLOW_PRESET_MEDIUM).calc(gray1, gray2, None)
except AttributeError:
print("DIS optical flow not available. Using Farneback method instead.")
flow = cv2.calcOpticalFlowFarneback(gray1, gray2, None, **flow_params)
flow = self.apply_smoothness(flow, smoothness)
flow_magnitude = np.sqrt(flow[..., 0]**2 + flow[..., 1]**2)
flow[flow_magnitude < threshold] = 0
interpolated.append(frame1)
for j in range(1, factor + 1):
t = j / (factor + 1)
warped = self.warp_flow(frame1, flow * t * strength, edge_mode)
blended = cv2.addWeighted(frame1, 1 - t, warped, t, 0)
interpolated.append(blended)
pbar.update(1)
interpolated.append(frames[-1])
return interpolated
def apply_smoothness(self, flow, smoothness):
if smoothness > 0:
sigma = smoothness * 5
flow[:,:,0] = gaussian_filter(flow[:,:,0], sigma=sigma)
flow[:,:,1] = gaussian_filter(flow[:,:,1], sigma=sigma)
return flow
def warp_flow(self, img, flow, edge_mode):
h, w = flow.shape[:2]
flow = -flow
flow[:,:,0] += np.arange(w)
flow[:,:,1] += np.arange(h)[:,np.newaxis]
if edge_mode == "Replicate":
border_mode = cv2.BORDER_REPLICATE
elif edge_mode == "Reflect":
border_mode = cv2.BORDER_REFLECT
elif edge_mode == "Wrap":
border_mode = cv2.BORDER_WRAP
elif edge_mode == "Constant":
border_mode = cv2.BORDER_CONSTANT
else:
border_mode = cv2.BORDER_REPLICATE
return cv2.remap(img, flow, None, cv2.INTER_LINEAR, borderMode=border_mode)
def process_output_frames(self, frames):
processed_frames = []
for frame in frames:
@@ -292,48 +392,23 @@ class CogVideoXImageToVideoNodeExtended:
def preprocess_image(self, image):
if isinstance(image, torch.Tensor):
image = image.cpu().numpy()
if image.ndim == 4 and image.shape[0] == 1:
image = image[0]
if image.ndim == 3:
if image.shape[0] == 3:
image = np.transpose(image, (1, 2, 0))
elif image.shape[2] != 3:
raise ValueError(f"The image must have 3 color channels, found: {image.shape[2]}")
if image.dtype != np.uint8:
image = (image * 255).astype(np.uint8)
pil_image = Image.fromarray(image)
# Ridimensiona l'immagine mantenendo l'aspect ratio
target_size = (720, 480)
aspect_ratio = pil_image.width / pil_image.height
target_aspect_ratio = target_size[0] / target_size[1]
if aspect_ratio > target_aspect_ratio:
# L'immagine è più larga, ridimensioniamo basandoci sull'altezza
new_height = target_size[1]
new_width = int(new_height * aspect_ratio)
else:
# L'immagine è più alta, ridimensioniamo basandoci sulla larghezza
new_width = target_size[0]
new_height = int(new_width / aspect_ratio)
resized_image = pil_image.resize((new_width, new_height), Image.LANCZOS)
# Calcola le coordinate per il ritaglio centrale
left = (resized_image.width - target_size[0]) // 2
top = (resized_image.height - target_size[1]) // 2
right = left + target_size[0]
bottom = top + target_size[1]
# Ritaglia l'immagine
cropped_image = resized_image.crop((left, top, right, bottom))
print(f"Preprocessed image size: {cropped_image.size}")
return cropped_image
return resize_and_crop(pil_image, target_size)
class SaveVideoNode:
@classmethod
+17 -16
View File
@@ -1,16 +1,17 @@
diffusers>=0.30.3
accelerate>=0.34.2
transformers>=4.44.2
numpy==1.26.0
torch>=2.4.0
torchvision>=0.19.0
sentencepiece>=0.2.0
gradio>=4.44.0
imageio>=2.35.1
imageio-ffmpeg>=0.5.1
ffmpeg-python>=0.2.0
openai>=1.45.0
moviepy>=1.0.3
pillow==9.5.0
huggingface_hub>=0.14.0
tqdm>=4.65.0
diffusers
accelerate
transformers
numpy
torch
torchvision
sentencepiece
gradio
imageio
imageio-ffmpeg
ffmpeg-python
openai
moviepy
pillow
huggingface_hub
tqdm
scipy