diff --git a/README.md b/README.md index 1b8c489..2f95f4e 100644 --- a/README.md +++ b/README.md @@ -95,7 +95,6 @@ We need about 60GB available on disk (for saving weights), please check! #### b. Weights We'd better place the [weights](#model-zoo) along the specified path: -EasyAnimateV4: ``` 📦 models/ ├── 📂 Diffusion_Transformer/ @@ -111,7 +110,7 @@ EasyAnimateV4: #### a. Using Python Code - Step 1: Download the corresponding [weights](#model-zoo) and place them in the models folder. - Step 2: Modify prompt, neg_prompt, guidance_scale, and seed in the predict_t2v.py file. -- Step 3: Run the predict_t2v.py file, wait for the generated results, and save the results in the samples/easyanimate-videos-t2v folder. +- Step 3: Run the predict_t2v.py file, wait for the generated results, and save the results in the samples/cogvideox-fun-videos-t2v folder. - Step 4: If you want to combine other backbones you have trained with Lora, modify the predict_t2v.py and Lora_path in predict_t2v.py depending on the situation. #### b. Using webui diff --git a/README_zh-CN.md b/README_zh-CN.md index deefe42..70a6d93 100644 --- a/README_zh-CN.md +++ b/README_zh-CN.md @@ -109,7 +109,7 @@ Linux 的详细信息: ##### i、运行python文件 - 步骤1:下载对应[权重](#model-zoo)放入models文件夹。 - 步骤2:在predict_t2v.py文件中修改prompt、neg_prompt、guidance_scale和seed。 -- 步骤3:运行predict_t2v.py文件,等待生成结果,结果保存在samples/cogvideox-videos-t2v文件夹中。 +- 步骤3:运行predict_t2v.py文件,等待生成结果,结果保存在samples/cogvideox-fun-videos-t2v文件夹中。 - 步骤4:如果想结合自己训练的其他backbone与Lora,则看情况修改predict_t2v.py中的predict_t2v.py和lora_path。 ##### ii、通过ui界面 diff --git a/cogvideox/pipeline/pipeline_cogvideox.py b/cogvideox/pipeline/pipeline_cogvideox.py index 6edefa2..81da03c 100644 --- a/cogvideox/pipeline/pipeline_cogvideox.py +++ b/cogvideox/pipeline/pipeline_cogvideox.py @@ -38,11 +38,11 @@ EXAMPLE_DOC_STRING = """ Examples: ```python >>> import torch - >>> from diffusers import CogVideoX_FUN_Pipeline + >>> from diffusers import CogVideoX_Fun_Pipeline >>> from diffusers.utils import export_to_video >>> # Models: "THUDM/CogVideoX-2b" or "THUDM/CogVideoX-5b" - >>> pipe = CogVideoX_FUN_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda") + >>> pipe = CogVideoX_Fun_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda") >>> prompt = ( ... "A panda, dressed in a small, red jacket and a tiny hat, sits on a wooden stool in a serene bamboo forest. " ... "The panda's fluffy paws strum a miniature acoustic guitar, producing soft, melodic tunes. Nearby, a few other " @@ -137,7 +137,7 @@ def retrieve_timesteps( @dataclass -class CogVideoX_FUN_PipelineOutput(BaseOutput): +class CogVideoX_Fun_PipelineOutput(BaseOutput): r""" Output class for CogVideo pipelines. @@ -151,9 +151,9 @@ class CogVideoX_FUN_PipelineOutput(BaseOutput): videos: torch.Tensor -class CogVideoX_FUN_Pipeline(DiffusionPipeline): +class CogVideoX_Fun_Pipeline(DiffusionPipeline): r""" - Pipeline for text-to-video generation using CogVideoX_FUN. + Pipeline for text-to-video generation using CogVideoX_Fun. This model inherits from [`DiffusionPipeline`]. Check the superclass documentation for the generic methods the library implements for all the pipelines (such as downloading or saving, running on a particular device, etc.) @@ -511,7 +511,7 @@ class CogVideoX_FUN_Pipeline(DiffusionPipeline): ] = None, callback_on_step_end_tensor_inputs: List[str] = ["latents"], max_sequence_length: int = 226, - ) -> Union[CogVideoX_FUN_PipelineOutput, Tuple]: + ) -> Union[CogVideoX_Fun_PipelineOutput, Tuple]: """ Function invoked when calling the pipeline for generation. @@ -583,8 +583,8 @@ class CogVideoX_FUN_Pipeline(DiffusionPipeline): Examples: Returns: - [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] or `tuple`: - [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] if `return_dict` is True, otherwise a + [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] or `tuple`: + [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] if `return_dict` is True, otherwise a `tuple`. When returning a tuple, the first element is a list with the generated images. """ @@ -748,4 +748,4 @@ class CogVideoX_FUN_Pipeline(DiffusionPipeline): if not return_dict: video = torch.from_numpy(video) - return CogVideoX_FUN_PipelineOutput(videos=video) + return CogVideoX_Fun_PipelineOutput(videos=video) diff --git a/cogvideox/pipeline/pipeline_cogvideox_inpaint.py b/cogvideox/pipeline/pipeline_cogvideox_inpaint.py index f074690..d86160d 100644 --- a/cogvideox/pipeline/pipeline_cogvideox_inpaint.py +++ b/cogvideox/pipeline/pipeline_cogvideox_inpaint.py @@ -42,11 +42,11 @@ EXAMPLE_DOC_STRING = """ Examples: ```python >>> import torch - >>> from diffusers import CogVideoX_FUN_Pipeline + >>> from diffusers import CogVideoX_Fun_Pipeline >>> from diffusers.utils import export_to_video >>> # Models: "THUDM/CogVideoX-2b" or "THUDM/CogVideoX-5b" - >>> pipe = CogVideoX_FUN_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda") + >>> pipe = CogVideoX_Fun_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda") >>> prompt = ( ... "A panda, dressed in a small, red jacket and a tiny hat, sits on a wooden stool in a serene bamboo forest. " ... "The panda's fluffy paws strum a miniature acoustic guitar, producing soft, melodic tunes. Nearby, a few other " @@ -178,7 +178,7 @@ def resize_mask(mask, latent, process_first_frame_only=True): @dataclass -class CogVideoX_FUN_PipelineOutput(BaseOutput): +class CogVideoX_Fun_PipelineOutput(BaseOutput): r""" Output class for CogVideo pipelines. @@ -192,7 +192,7 @@ class CogVideoX_FUN_PipelineOutput(BaseOutput): videos: torch.Tensor -class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline): +class CogVideoX_Fun_Pipeline_Inpaint(DiffusionPipeline): r""" Pipeline for text-to-video generation using CogVideoX. @@ -203,7 +203,7 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline): vae ([`AutoencoderKL`]): Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations. text_encoder ([`T5EncoderModel`]): - Frozen text-encoder. CogVideoX_FUN uses + Frozen text-encoder. CogVideoX_Fun uses [T5](https://huggingface.co/docs/transformers/model_doc/t5#transformers.T5EncoderModel); specifically the [t5-v1_1-xxl](https://huggingface.co/PixArt-alpha/PixArt-alpha/tree/main/t5-v1_1-xxl) variant. tokenizer (`T5Tokenizer`): @@ -651,7 +651,7 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline): max_sequence_length: int = 226, strength: float = 1, comfyui_progressbar: bool = False, - ) -> Union[CogVideoX_FUN_PipelineOutput, Tuple]: + ) -> Union[CogVideoX_Fun_PipelineOutput, Tuple]: """ Function invoked when calling the pipeline for generation. @@ -669,7 +669,7 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline): The width in pixels of the generated image. This is set to 1024 by default for the best results. num_frames (`int`, defaults to `48`): Number of frames to generate. Must be divisible by self.vae_scale_factor_temporal. Generated video will - contain 1 extra frame because CogVideoX_FUN is conditioned with (num_seconds * fps + 1) frames where + contain 1 extra frame because CogVideoX_Fun is conditioned with (num_seconds * fps + 1) frames where num_seconds is 6 and fps is 4. However, since videos can be saved at any fps, the only condition that needs to be satisfied is that of divisibility mentioned above. num_inference_steps (`int`, *optional*, defaults to 50): @@ -723,8 +723,8 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline): Examples: Returns: - [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] or `tuple`: - [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] if `return_dict` is True, otherwise a + [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] or `tuple`: + [`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] if `return_dict` is True, otherwise a `tuple`. When returning a tuple, the first element is a list with the generated images. """ @@ -1000,4 +1000,4 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline): if not return_dict: video = torch.from_numpy(video) - return CogVideoX_FUN_PipelineOutput(videos=video) + return CogVideoX_Fun_PipelineOutput(videos=video) diff --git a/cogvideox/ui/ui.py b/cogvideox/ui/ui.py index f86cdc7..7f4824c 100644 --- a/cogvideox/ui/ui.py +++ b/cogvideox/ui/ui.py @@ -29,9 +29,9 @@ from transformers import (CLIPImageProcessor, CLIPVisionModelWithProjection, from cogvideox.data.bucket_sampler import ASPECT_RATIO_512, get_closest_ratio from ..models.autoencoder_magvit import AutoencoderKLCogVideoX from cogvideox.models.transformer3d import CogVideoXTransformer3DModel -from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline +from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline from cogvideox.pipeline.pipeline_cogvideox_inpaint import \ - CogVideoX_FUN_Pipeline_Inpaint + CogVideoX_Fun_Pipeline_Inpaint from cogvideox.utils.lora_utils import merge_lora, unmerge_lora from cogvideox.utils.utils import ( get_image_to_video_latent, get_video_to_video_latent, @@ -119,7 +119,7 @@ class CogVideoX_I2VController: # Get pipeline if self.transformer.config.in_channels != self.vae.config.latent_channels: - self.pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + self.pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( diffusion_transformer_dropdown, vae=self.vae, transformer=self.transformer, @@ -127,7 +127,7 @@ class CogVideoX_I2VController: torch_dtype=self.weight_dtype ) else: - self.pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + self.pipeline = CogVideoX_Fun_Pipeline.from_pretrained( diffusion_transformer_dropdown, vae=self.vae, transformer=self.transformer, @@ -673,7 +673,7 @@ class CogVideoX_I2VController_Modelscope: # Get pipeline if self.transformer.config.in_channels != self.vae.config.latent_channels: - self.pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + self.pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( model_name, vae=self.vae, transformer=self.transformer, @@ -681,7 +681,7 @@ class CogVideoX_I2VController_Modelscope: torch_dtype=self.weight_dtype ) else: - self.pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + self.pipeline = CogVideoX_Fun_Pipeline.from_pretrained( model_name, vae=self.vae, transformer=self.transformer, diff --git a/comfyui/README.md b/comfyui/README.md index d091fac..69f9198 100644 --- a/comfyui/README.md +++ b/comfyui/README.md @@ -1,16 +1,9 @@ -# ComfyUI EasyAnimate -Easily use EasyAnimate inside ComfyUI! - -[![Arxiv Page](https://img.shields.io/badge/Arxiv-Page-red)](https://arxiv.org/abs/2405.18991) -[![Project Page](https://img.shields.io/badge/Project-Website-green)](https://easyanimate.github.io/) -[![Modelscope Studio](https://img.shields.io/badge/Modelscope-Studio-blue)](https://modelscope.cn/studios/PAI/EasyAnimate/summary) -[![Hugging Face Spaces](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-yellow)](https://huggingface.co/spaces/alibaba-pai/EasyAnimate) +# ComfyUI CogVideoX-Fun +Easily use CogVideoX-Fun inside ComfyUI! - [Installation](#1-installation) - [Node types](#node-types) - [Example workflows](#example-workflows) - - [Image to video](#image-to-video) - - [Image to video generation (high FPS w/ frame interpolation)](#image-to-video-generation-high-fps-w-frame-interpolation) ## 1. Installation @@ -18,72 +11,55 @@ Easily use EasyAnimate inside ComfyUI! TBD ### Option 2: Install manually -The EasyAnimate repository needs to be placed at `ComfyUI/custom_nodes/EasyAnimate/`. +The CogVideoX-Fun repository needs to be placed at `ComfyUI/custom_nodes/CogVideoX-Fun/`. ``` cd ComfyUI/custom_nodes/ -# Git clone the easyanimate itself -git clone https://github.com/aigc-apps/EasyAnimate.git +# Git clone the cogvideox_fun itself +git clone https://github.com/aigc-apps/CogVideoX-Fun.git # Git clone the video outout node git clone https://github.com/Kosinkadink/ComfyUI-VideoHelperSuite.git -cd EasyAnimate/ +cd CogVideoX-Fun/ python install.py ``` -### 2. Download models into `ComfyUI/models/EasyAnimate/` +### 2. Download models into `ComfyUI/models/CogVideoX-Fun/` -| Name | Type | Storage Space | Url | Hugging Face | Description | -| EasyAnimateV4-XL-2-InP.tar | EasyAnimateV4 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV4-XL-2-InP.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV4-XL-2-InP)| Our official graph-generated video model is capable of predicting videos at multiple resolutions (512, 768, 1024, 1280) and has been trained on 144 frames at a rate of 24 frames per second. | - -
- (Obsolete) EasyAnimateV3: - -| Name | Type | Storage Space | Url | Hugging Face | Description | -|--|--|--|--|--|--| -| EasyAnimateV3-XL-2-InP-512x512.tar | EasyAnimateV3 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV3-XL-2-InP-512x512.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-512x512) | EasyAnimateV3 official weights for 512x512 text and image to video resolution. Training with 144 frames and fps 24 | -| EasyAnimateV3-XL-2-InP-768x768.tar | EasyAnimateV3 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV3-XL-2-InP-768x768.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-768x768) | EasyAnimateV3 official weights for 768x768 text and image to video resolution. Training with 144 frames and fps 24 | -| EasyAnimateV3-XL-2-InP-960x960.tar | EasyAnimateV3 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV3-XL-2-InP-960x960.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-960x960) | EasyAnimateV3 official weights for 960x960 text and image to video resolution. Training with 144 frames and fps 24 | -
+| Name | Storage Space | Url | Hugging Face | Description | +|--|--|--|--|--| +| CogVideoX-Fun-2b-InP.tar.gz | Before extraction:9.69 GB \/ After extraction: 13.0 GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/Diffusion_Transformer/CogVideoX-Fun-2b-InP.tar.gz) | [🤗Link](https://huggingface.co/alibaba-pai/CogVideoX-Fun-2b-InP)| Our official graph-generated video model is capable of predicting videos at multiple resolutions (512, 768, 1024, 1280) and has been trained on 144 frames at a rate of 24 frames per second. | ## Node types -- **LoadEasyAnimateModel** - - Loads the EasyAnimate model +- **LoadCogVideoX_Fun_Model** + - Loads the CogVideoX-Fun model - **TextBox** - - Write the prompt for EasyAnimate model -- **EasyAnimateI2VSampler** - - EasyAnimate Sampler for Image to Video -- **EasyAnimateT2VSampler** - - EasyAnimate Sampler for Text to Video -- **EasyAnimateV2VSampler** - - EasyAnimate Sampler for Video to Video + - Write the prompt for CogVideoX-Fun model +- **CogVideoX_Fun_I2VSampler** + - CogVideoX-Fun Sampler for Image to Video +- **CogVideoX_Fun_T2VSampler** + - CogVideoX-Fun Sampler for Text to Video +- **CogVideoX_Fun_V2VSampler** + - CogVideoX-Fun Sampler for Video to Video ## Example workflows ### Video to video generation -Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_v2v.json) of the json: -![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_v2v.jpg) +Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/cogvideoxfunv1_workflow_v2v.json) of the json: +![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/comfyui_v2v.jpg) You can run the demo using following video: -[demo video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/play_guitar.mp4) +[demo video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/play_guitar.mp4) ### Image to video generation -Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_i2v.json) of the json: -![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_i2v.jpg) +Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/cogvideoxfunv1_workflow_i2v.json) of the json: +![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/comfyui_i2v.jpg) You can run the demo using following photo: -![demo image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/firework.png) +![demo image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png) ### Text to video generation -Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_t2v.json) of the json: -![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_t2v.jpg) - -### Text to video generation With Lora -We have provided a v4 version of the portrait Lora for testing, and the specific download link is [Lora download Link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimatev4_minimalism_lora.safetensors). - -You can put this Lora at ```ComfyUI/models/loras/easyanimate```. - -Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_lora.json) of the json: -![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_lora.jpg) \ No newline at end of file +Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/cogvideoxfunv1_workflow_t2v.json) of the json: +![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/comfyui_t2v.jpg) \ No newline at end of file diff --git a/comfyui/comfyui_nodes.py b/comfyui/comfyui_nodes.py index faddf55..c572352 100644 --- a/comfyui/comfyui_nodes.py +++ b/comfyui/comfyui_nodes.py @@ -22,9 +22,9 @@ from transformers import T5EncoderModel, T5Tokenizer from ..cogvideox.data.bucket_sampler import ASPECT_RATIO_512, get_closest_ratio from ..cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX from ..cogvideox.models.transformer3d import CogVideoXTransformer3DModel -from ..cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline +from ..cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline from ..cogvideox.pipeline.pipeline_cogvideox_inpaint import ( - CogVideoX_FUN_Pipeline_Inpaint) + CogVideoX_Fun_Pipeline_Inpaint) from ..cogvideox.utils.lora_utils import merge_lora, unmerge_lora from ..cogvideox.utils.utils import (get_image_to_video_latent, get_video_to_video_latent, @@ -50,7 +50,7 @@ def to_pil(image): return numpy2pil(image) raise ValueError(f"Cannot convert {type(image)} to PIL.Image") -class LoadCogVideoX_FUN_Model: +class LoadCogVideoX_Fun_Model: @classmethod def INPUT_TYPES(s): return { @@ -94,11 +94,11 @@ class LoadCogVideoX_FUN_Model: pbar = ProgressBar(3) # Detect model is existing or not - model_path = os.path.join(folder_paths.models_dir, "CogVideoX_FUN", model) + model_path = os.path.join(folder_paths.models_dir, "CogVideoX_Fun", model) if not os.path.exists(model_path): if os.path.exists(eas_cache_dir): - model_path = os.path.join(eas_cache_dir, 'CogVideoX_FUN', model) + model_path = os.path.join(eas_cache_dir, 'CogVideoX_Fun', model) else: print(f"Please download cogvideoxfun model to: {model_path}") @@ -125,7 +125,7 @@ class LoadCogVideoX_FUN_Model: # Get pipeline if transformer.config.in_channels != vae.config.latent_channels: - pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( model_path, vae=vae, transformer=transformer, @@ -133,7 +133,7 @@ class LoadCogVideoX_FUN_Model: torch_dtype=weight_dtype ) else: - pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline.from_pretrained( model_path, vae=vae, transformer=transformer, @@ -154,7 +154,7 @@ class LoadCogVideoX_FUN_Model: } return (cogvideoxfun_model,) -class LoadCogVideoX_FUN_Lora: +class LoadCogVideoX_Fun_Lora: @classmethod def INPUT_TYPES(s): return { @@ -200,7 +200,7 @@ class TextBox: def process(self, prompt): return (prompt, ) -class CogVideoX_FUN_I2VSampler: +class CogVideoX_Fun_I2VSampler: @classmethod def INPUT_TYPES(s): return { @@ -320,7 +320,7 @@ class CogVideoX_FUN_I2VSampler: return (videos,) -class CogVideoX_FUN_T2VSampler: +class CogVideoX_Fun_T2VSampler: @classmethod def INPUT_TYPES(s): return { @@ -435,7 +435,7 @@ class CogVideoX_FUN_T2VSampler: pipeline = unmerge_lora(pipeline, _lora_path, _lora_weight) return (videos,) -class CogVideoX_FUN_V2VSampler: +class CogVideoX_Fun_V2VSampler: @classmethod def INPUT_TYPES(s): return { @@ -559,19 +559,19 @@ class CogVideoX_FUN_V2VSampler: NODE_CLASS_MAPPINGS = { "TextBox": TextBox, - "LoadCogVideoX_FUN_Model": LoadCogVideoX_FUN_Model, - "LoadCogVideoX_FUN_Lora": LoadCogVideoX_FUN_Lora, - "CogVideoX_FUN_I2VSampler": CogVideoX_FUN_I2VSampler, - "CogVideoX_FUN_T2VSampler": CogVideoX_FUN_T2VSampler, - "CogVideoX_FUN_V2VSampler": CogVideoX_FUN_V2VSampler, + "LoadCogVideoX_Fun_Model": LoadCogVideoX_Fun_Model, + "LoadCogVideoX_Fun_Lora": LoadCogVideoX_Fun_Lora, + "CogVideoX_Fun_I2VSampler": CogVideoX_Fun_I2VSampler, + "CogVideoX_Fun_T2VSampler": CogVideoX_Fun_T2VSampler, + "CogVideoX_Fun_V2VSampler": CogVideoX_Fun_V2VSampler, } NODE_DISPLAY_NAME_MAPPINGS = { "TextBox": "TextBox", - "LoadCogVideoX_FUN_Model": "Load CogVideoX-Fun Model", - "LoadCogVideoX_FUN_Lora": "Load CogVideoX-Fun Lora", - "CogVideoX_FUN_I2VSampler": "CogVideoX-Fun Sampler for Image to Video", - "CogVideoX_FUN_T2VSampler": "CogVideoX-Fun Sampler for Text to Video", - "CogVideoX_FUN_V2VSampler": "CogVideoX-Fun Sampler for Video to Video", + "LoadCogVideoX_Fun_Model": "Load CogVideoX-Fun Model", + "LoadCogVideoX_Fun_Lora": "Load CogVideoX-Fun Lora", + "CogVideoX_Fun_I2VSampler": "CogVideoX-Fun Sampler for Image to Video", + "CogVideoX_Fun_T2VSampler": "CogVideoX-Fun Sampler for Text to Video", + "CogVideoX_Fun_V2VSampler": "CogVideoX-Fun Sampler for Video to Video", } \ No newline at end of file diff --git a/comfyui/v1/cogvideoxfunv1_workflow_i2v.json b/comfyui/v1/cogvideoxfunv1_workflow_i2v.json index f59ba13..0fca261 100644 --- a/comfyui/v1/cogvideoxfunv1_workflow_i2v.json +++ b/comfyui/v1/cogvideoxfunv1_workflow_i2v.json @@ -115,7 +115,7 @@ }, { "id": 83, - "type": "LoadCogVideoX_FUN_Model", + "type": "LoadCogVideoX_Fun_Model", "pos": [ 300, -294 @@ -139,7 +139,7 @@ } ], "properties": { - "Node name for S&R": "LoadCogVideoX_FUN_Model" + "Node name for S&R": "LoadCogVideoX_Fun_Model" }, "widgets_values": [ "CogVideoX-Fun-2b-InP", @@ -215,7 +215,7 @@ }, { "id": 82, - "type": "CogVideoX_FUN_I2VSampler", + "type": "CogVideoX_Fun_I2VSampler", "pos": [ 758, 93 @@ -267,7 +267,7 @@ } ], "properties": { - "Node name for S&R": "CogVideoX_FUN_I2VSampler" + "Node name for S&R": "CogVideoX_Fun_I2VSampler" }, "widgets_values": [ 49, diff --git a/comfyui/v1/cogvideoxfunv1_workflow_t2v.json b/comfyui/v1/cogvideoxfunv1_workflow_t2v.json index 4b88e0e..fd90ce7 100644 --- a/comfyui/v1/cogvideoxfunv1_workflow_t2v.json +++ b/comfyui/v1/cogvideoxfunv1_workflow_t2v.json @@ -83,7 +83,7 @@ }, { "id": 87, - "type": "LoadCogVideoX_FUN_Model", + "type": "LoadCogVideoX_Fun_Model", "pos": [ 302, -285 @@ -107,7 +107,7 @@ } ], "properties": { - "Node name for S&R": "LoadCogVideoX_FUN_Model" + "Node name for S&R": "LoadCogVideoX_Fun_Model" }, "widgets_values": [ "CogVideoX-Fun-2b-InP", @@ -150,7 +150,7 @@ }, { "id": 88, - "type": "CogVideoX_FUN_T2VSampler", + "type": "CogVideoX_Fun_T2VSampler", "pos": [ 728, -68 @@ -192,7 +192,7 @@ } ], "properties": { - "Node name for S&R": "CogVideoX_FUN_T2VSampler" + "Node name for S&R": "CogVideoX_Fun_T2VSampler" }, "widgets_values": [ 49, diff --git a/comfyui/v1/cogvideoxfunv1_workflow_v2v.json b/comfyui/v1/cogvideoxfunv1_workflow_v2v.json index a34c479..c179757 100644 --- a/comfyui/v1/cogvideoxfunv1_workflow_v2v.json +++ b/comfyui/v1/cogvideoxfunv1_workflow_v2v.json @@ -189,7 +189,7 @@ }, { "id": 88, - "type": "LoadCogVideoX_FUN_Model", + "type": "LoadCogVideoX_Fun_Model", "pos": [ 309, -286 @@ -213,7 +213,7 @@ } ], "properties": { - "Node name for S&R": "LoadCogVideoX_FUN_Model" + "Node name for S&R": "LoadCogVideoX_Fun_Model" }, "widgets_values": [ "CogVideoX-Fun-2b-InP", @@ -256,7 +256,7 @@ }, { "id": 87, - "type": "CogVideoX_FUN_V2VSampler", + "type": "CogVideoX_Fun_V2VSampler", "pos": [ 778, 93 @@ -304,7 +304,7 @@ } ], "properties": { - "Node name for S&R": "CogVideoX_FUN_V2VSampler" + "Node name for S&R": "CogVideoX_Fun_V2VSampler" }, "widgets_values": [ 49, diff --git a/predict_i2v.py b/predict_i2v.py index e8b6319..515dd85 100644 --- a/predict_i2v.py +++ b/predict_i2v.py @@ -15,8 +15,8 @@ from PIL import Image from cogvideox.models.transformer3d import CogVideoXTransformer3DModel from cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX -from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline -from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint +from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline +from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint from cogvideox.utils.lora_utils import merge_lora, unmerge_lora from cogvideox.utils.utils import get_image_to_video_latent, save_videos_grid @@ -112,7 +112,7 @@ scheduler = Choosen_Scheduler.from_pretrained( ) if transformer.config.in_channels != vae.config.latent_channels: - pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( model_name, vae=vae, text_encoder=text_encoder, @@ -121,7 +121,7 @@ if transformer.config.in_channels != vae.config.latent_channels: torch_dtype=weight_dtype ) else: - pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline.from_pretrained( model_name, vae=vae, text_encoder=text_encoder, diff --git a/predict_t2v.py b/predict_t2v.py index f0e250b..ce35b60 100644 --- a/predict_t2v.py +++ b/predict_t2v.py @@ -15,8 +15,8 @@ from PIL import Image from cogvideox.models.transformer3d import CogVideoXTransformer3DModel from cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX -from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline -from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint +from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline +from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint from cogvideox.utils.lora_utils import merge_lora, unmerge_lora from cogvideox.utils.utils import get_image_to_video_latent, save_videos_grid @@ -103,7 +103,7 @@ scheduler = Choosen_Scheduler.from_pretrained( ) if transformer.config.in_channels != vae.config.latent_channels: - pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( model_name, vae=vae, text_encoder=text_encoder, @@ -112,7 +112,7 @@ if transformer.config.in_channels != vae.config.latent_channels: torch_dtype=weight_dtype ) else: - pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline.from_pretrained( model_name, vae=vae, text_encoder=text_encoder, diff --git a/predict_v2v.py b/predict_v2v.py index fc1c657..038dc52 100644 --- a/predict_v2v.py +++ b/predict_v2v.py @@ -15,9 +15,9 @@ from transformers import (CLIPImageProcessor, CLIPVisionModelWithProjection, from cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX from cogvideox.models.transformer3d import CogVideoXTransformer3DModel -from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline +from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline from cogvideox.pipeline.pipeline_cogvideox_inpaint import \ - CogVideoX_FUN_Pipeline_Inpaint + CogVideoX_Fun_Pipeline_Inpaint from cogvideox.utils.lora_utils import merge_lora, unmerge_lora from cogvideox.utils.utils import get_video_to_video_latent, save_videos_grid @@ -108,7 +108,7 @@ scheduler = Choosen_Scheduler.from_pretrained( ) if transformer.config.in_channels != vae.config.latent_channels: - pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( model_name, vae=vae, text_encoder=text_encoder, @@ -117,7 +117,7 @@ if transformer.config.in_channels != vae.config.latent_channels: torch_dtype=weight_dtype ) else: - pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline.from_pretrained( model_name, vae=vae, text_encoder=text_encoder, diff --git a/scripts/train.py b/scripts/train.py index 06e5acb..233dcb6 100644 --- a/scripts/train.py +++ b/scripts/train.py @@ -72,8 +72,8 @@ from cogvideox.data.dataset_image_video import (ImageVideoDataset, ImageVideoSampler, get_random_mask) from cogvideox.models.transformer3d import CogVideoXTransformer3DModel -from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline -from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint +from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline +from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint from cogvideox.utils.utils import get_image_to_video_latent, save_videos_grid if is_wandb_available(): @@ -163,7 +163,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, args, accelerato transformer3d_val.load_state_dict(accelerator.unwrap_model(transformer3d).state_dict()) if args.train_mode != "normal": - pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( args.pretrained_model_name_or_path, vae=accelerator.unwrap_model(vae).to(weight_dtype), text_encoder=accelerator.unwrap_model(text_encoder), @@ -172,7 +172,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, args, accelerato torch_dtype=weight_dtype ) else: - pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline.from_pretrained( args.pretrained_model_name_or_path, vae=accelerator.unwrap_model(vae).to(weight_dtype), text_encoder=accelerator.unwrap_model(text_encoder), diff --git a/scripts/train_lora.py b/scripts/train_lora.py index fb6b4fb..7f7f7b8 100644 --- a/scripts/train_lora.py +++ b/scripts/train_lora.py @@ -70,8 +70,8 @@ from cogvideox.data.bucket_sampler import (ASPECT_RATIO_512, AspectRatioBatchImageVideoSampler, AspectRatioBatchSampler, RandomSampler, get_closest_ratio) -from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline -from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint +from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline +from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint from cogvideox.data.dataset_image import CC15M from cogvideox.data.dataset_image_video import (ImageVideoDataset, ImageVideoSampler, @@ -168,7 +168,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, network, args, a transformer3d_val.load_state_dict(accelerator.unwrap_model(transformer3d).state_dict()) if args.train_mode != "normal": - pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained( args.pretrained_model_name_or_path, vae=accelerator.unwrap_model(vae).to(weight_dtype), text_encoder=accelerator.unwrap_model(text_encoder), @@ -177,7 +177,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, network, args, a torch_dtype=weight_dtype, ) else: - pipeline = CogVideoX_FUN_Pipeline.from_pretrained( + pipeline = CogVideoX_Fun_Pipeline.from_pretrained( args.pretrained_model_name_or_path, vae=accelerator.unwrap_model(vae).to(weight_dtype), text_encoder=accelerator.unwrap_model(text_encoder),