Rename && Update Readme

This commit is contained in:
bubbliiiing
2024-09-10 11:22:35 +08:00
parent d29d193177
commit ef55bd3bbe
15 changed files with 107 additions and 132 deletions
+1 -2
View File
@@ -95,7 +95,6 @@ We need about 60GB available on disk (for saving weights), please check!
#### b. Weights
We'd better place the [weights](#model-zoo) along the specified path:
EasyAnimateV4:
```
📦 models/
├── 📂 Diffusion_Transformer/
@@ -111,7 +110,7 @@ EasyAnimateV4:
#### a. Using Python Code
- Step 1: Download the corresponding [weights](#model-zoo) and place them in the models folder.
- Step 2: Modify prompt, neg_prompt, guidance_scale, and seed in the predict_t2v.py file.
- Step 3: Run the predict_t2v.py file, wait for the generated results, and save the results in the samples/easyanimate-videos-t2v folder.
- Step 3: Run the predict_t2v.py file, wait for the generated results, and save the results in the samples/cogvideox-fun-videos-t2v folder.
- Step 4: If you want to combine other backbones you have trained with Lora, modify the predict_t2v.py and Lora_path in predict_t2v.py depending on the situation.
#### b. Using webui
+1 -1
View File
@@ -109,7 +109,7 @@ Linux 的详细信息:
##### i、运行python文件
- 步骤1:下载对应[权重](#model-zoo)放入models文件夹。
- 步骤2:在predict_t2v.py文件中修改prompt、neg_prompt、guidance_scale和seed。
- 步骤3:运行predict_t2v.py文件,等待生成结果,结果保存在samples/cogvideox-videos-t2v文件夹中。
- 步骤3:运行predict_t2v.py文件,等待生成结果,结果保存在samples/cogvideox-fun-videos-t2v文件夹中。
- 步骤4:如果想结合自己训练的其他backbone与Lora,则看情况修改predict_t2v.py中的predict_t2v.py和lora_path。
##### ii、通过ui界面
+9 -9
View File
@@ -38,11 +38,11 @@ EXAMPLE_DOC_STRING = """
Examples:
```python
>>> import torch
>>> from diffusers import CogVideoX_FUN_Pipeline
>>> from diffusers import CogVideoX_Fun_Pipeline
>>> from diffusers.utils import export_to_video
>>> # Models: "THUDM/CogVideoX-2b" or "THUDM/CogVideoX-5b"
>>> pipe = CogVideoX_FUN_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda")
>>> pipe = CogVideoX_Fun_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda")
>>> prompt = (
... "A panda, dressed in a small, red jacket and a tiny hat, sits on a wooden stool in a serene bamboo forest. "
... "The panda's fluffy paws strum a miniature acoustic guitar, producing soft, melodic tunes. Nearby, a few other "
@@ -137,7 +137,7 @@ def retrieve_timesteps(
@dataclass
class CogVideoX_FUN_PipelineOutput(BaseOutput):
class CogVideoX_Fun_PipelineOutput(BaseOutput):
r"""
Output class for CogVideo pipelines.
@@ -151,9 +151,9 @@ class CogVideoX_FUN_PipelineOutput(BaseOutput):
videos: torch.Tensor
class CogVideoX_FUN_Pipeline(DiffusionPipeline):
class CogVideoX_Fun_Pipeline(DiffusionPipeline):
r"""
Pipeline for text-to-video generation using CogVideoX_FUN.
Pipeline for text-to-video generation using CogVideoX_Fun.
This model inherits from [`DiffusionPipeline`]. Check the superclass documentation for the generic methods the
library implements for all the pipelines (such as downloading or saving, running on a particular device, etc.)
@@ -511,7 +511,7 @@ class CogVideoX_FUN_Pipeline(DiffusionPipeline):
] = None,
callback_on_step_end_tensor_inputs: List[str] = ["latents"],
max_sequence_length: int = 226,
) -> Union[CogVideoX_FUN_PipelineOutput, Tuple]:
) -> Union[CogVideoX_Fun_PipelineOutput, Tuple]:
"""
Function invoked when calling the pipeline for generation.
@@ -583,8 +583,8 @@ class CogVideoX_FUN_Pipeline(DiffusionPipeline):
Examples:
Returns:
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] or `tuple`:
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] if `return_dict` is True, otherwise a
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] or `tuple`:
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] if `return_dict` is True, otherwise a
`tuple`. When returning a tuple, the first element is a list with the generated images.
"""
@@ -748,4 +748,4 @@ class CogVideoX_FUN_Pipeline(DiffusionPipeline):
if not return_dict:
video = torch.from_numpy(video)
return CogVideoX_FUN_PipelineOutput(videos=video)
return CogVideoX_Fun_PipelineOutput(videos=video)
@@ -42,11 +42,11 @@ EXAMPLE_DOC_STRING = """
Examples:
```python
>>> import torch
>>> from diffusers import CogVideoX_FUN_Pipeline
>>> from diffusers import CogVideoX_Fun_Pipeline
>>> from diffusers.utils import export_to_video
>>> # Models: "THUDM/CogVideoX-2b" or "THUDM/CogVideoX-5b"
>>> pipe = CogVideoX_FUN_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda")
>>> pipe = CogVideoX_Fun_Pipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16).to("cuda")
>>> prompt = (
... "A panda, dressed in a small, red jacket and a tiny hat, sits on a wooden stool in a serene bamboo forest. "
... "The panda's fluffy paws strum a miniature acoustic guitar, producing soft, melodic tunes. Nearby, a few other "
@@ -178,7 +178,7 @@ def resize_mask(mask, latent, process_first_frame_only=True):
@dataclass
class CogVideoX_FUN_PipelineOutput(BaseOutput):
class CogVideoX_Fun_PipelineOutput(BaseOutput):
r"""
Output class for CogVideo pipelines.
@@ -192,7 +192,7 @@ class CogVideoX_FUN_PipelineOutput(BaseOutput):
videos: torch.Tensor
class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline):
class CogVideoX_Fun_Pipeline_Inpaint(DiffusionPipeline):
r"""
Pipeline for text-to-video generation using CogVideoX.
@@ -203,7 +203,7 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline):
vae ([`AutoencoderKL`]):
Variational Auto-Encoder (VAE) Model to encode and decode videos to and from latent representations.
text_encoder ([`T5EncoderModel`]):
Frozen text-encoder. CogVideoX_FUN uses
Frozen text-encoder. CogVideoX_Fun uses
[T5](https://huggingface.co/docs/transformers/model_doc/t5#transformers.T5EncoderModel); specifically the
[t5-v1_1-xxl](https://huggingface.co/PixArt-alpha/PixArt-alpha/tree/main/t5-v1_1-xxl) variant.
tokenizer (`T5Tokenizer`):
@@ -651,7 +651,7 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline):
max_sequence_length: int = 226,
strength: float = 1,
comfyui_progressbar: bool = False,
) -> Union[CogVideoX_FUN_PipelineOutput, Tuple]:
) -> Union[CogVideoX_Fun_PipelineOutput, Tuple]:
"""
Function invoked when calling the pipeline for generation.
@@ -669,7 +669,7 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline):
The width in pixels of the generated image. This is set to 1024 by default for the best results.
num_frames (`int`, defaults to `48`):
Number of frames to generate. Must be divisible by self.vae_scale_factor_temporal. Generated video will
contain 1 extra frame because CogVideoX_FUN is conditioned with (num_seconds * fps + 1) frames where
contain 1 extra frame because CogVideoX_Fun is conditioned with (num_seconds * fps + 1) frames where
num_seconds is 6 and fps is 4. However, since videos can be saved at any fps, the only condition that
needs to be satisfied is that of divisibility mentioned above.
num_inference_steps (`int`, *optional*, defaults to 50):
@@ -723,8 +723,8 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline):
Examples:
Returns:
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] or `tuple`:
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_FUN_PipelineOutput`] if `return_dict` is True, otherwise a
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] or `tuple`:
[`~pipelines.cogvideo.pipeline_cogvideox.CogVideoX_Fun_PipelineOutput`] if `return_dict` is True, otherwise a
`tuple`. When returning a tuple, the first element is a list with the generated images.
"""
@@ -1000,4 +1000,4 @@ class CogVideoX_FUN_Pipeline_Inpaint(DiffusionPipeline):
if not return_dict:
video = torch.from_numpy(video)
return CogVideoX_FUN_PipelineOutput(videos=video)
return CogVideoX_Fun_PipelineOutput(videos=video)
+6 -6
View File
@@ -29,9 +29,9 @@ from transformers import (CLIPImageProcessor, CLIPVisionModelWithProjection,
from cogvideox.data.bucket_sampler import ASPECT_RATIO_512, get_closest_ratio
from ..models.autoencoder_magvit import AutoencoderKLCogVideoX
from cogvideox.models.transformer3d import CogVideoXTransformer3DModel
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import \
CogVideoX_FUN_Pipeline_Inpaint
CogVideoX_Fun_Pipeline_Inpaint
from cogvideox.utils.lora_utils import merge_lora, unmerge_lora
from cogvideox.utils.utils import (
get_image_to_video_latent, get_video_to_video_latent,
@@ -119,7 +119,7 @@ class CogVideoX_I2VController:
# Get pipeline
if self.transformer.config.in_channels != self.vae.config.latent_channels:
self.pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
self.pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
diffusion_transformer_dropdown,
vae=self.vae,
transformer=self.transformer,
@@ -127,7 +127,7 @@ class CogVideoX_I2VController:
torch_dtype=self.weight_dtype
)
else:
self.pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
self.pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
diffusion_transformer_dropdown,
vae=self.vae,
transformer=self.transformer,
@@ -673,7 +673,7 @@ class CogVideoX_I2VController_Modelscope:
# Get pipeline
if self.transformer.config.in_channels != self.vae.config.latent_channels:
self.pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
self.pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
model_name,
vae=self.vae,
transformer=self.transformer,
@@ -681,7 +681,7 @@ class CogVideoX_I2VController_Modelscope:
torch_dtype=self.weight_dtype
)
else:
self.pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
self.pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
model_name,
vae=self.vae,
transformer=self.transformer,
+27 -51
View File
@@ -1,16 +1,9 @@
# ComfyUI EasyAnimate
Easily use EasyAnimate inside ComfyUI!
[![Arxiv Page](https://img.shields.io/badge/Arxiv-Page-red)](https://arxiv.org/abs/2405.18991)
[![Project Page](https://img.shields.io/badge/Project-Website-green)](https://easyanimate.github.io/)
[![Modelscope Studio](https://img.shields.io/badge/Modelscope-Studio-blue)](https://modelscope.cn/studios/PAI/EasyAnimate/summary)
[![Hugging Face Spaces](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-yellow)](https://huggingface.co/spaces/alibaba-pai/EasyAnimate)
# ComfyUI CogVideoX-Fun
Easily use CogVideoX-Fun inside ComfyUI!
- [Installation](#1-installation)
- [Node types](#node-types)
- [Example workflows](#example-workflows)
- [Image to video](#image-to-video)
- [Image to video generation (high FPS w/ frame interpolation)](#image-to-video-generation-high-fps-w-frame-interpolation)
## 1. Installation
@@ -18,72 +11,55 @@ Easily use EasyAnimate inside ComfyUI!
TBD
### Option 2: Install manually
The EasyAnimate repository needs to be placed at `ComfyUI/custom_nodes/EasyAnimate/`.
The CogVideoX-Fun repository needs to be placed at `ComfyUI/custom_nodes/CogVideoX-Fun/`.
```
cd ComfyUI/custom_nodes/
# Git clone the easyanimate itself
git clone https://github.com/aigc-apps/EasyAnimate.git
# Git clone the cogvideox_fun itself
git clone https://github.com/aigc-apps/CogVideoX-Fun.git
# Git clone the video outout node
git clone https://github.com/Kosinkadink/ComfyUI-VideoHelperSuite.git
cd EasyAnimate/
cd CogVideoX-Fun/
python install.py
```
### 2. Download models into `ComfyUI/models/EasyAnimate/`
### 2. Download models into `ComfyUI/models/CogVideoX-Fun/`
| Name | Type | Storage Space | Url | Hugging Face | Description |
| EasyAnimateV4-XL-2-InP.tar | EasyAnimateV4 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV4-XL-2-InP.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV4-XL-2-InP)| Our official graph-generated video model is capable of predicting videos at multiple resolutions (512, 768, 1024, 1280) and has been trained on 144 frames at a rate of 24 frames per second. |
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
| Name | Type | Storage Space | Url | Hugging Face | Description |
|--|--|--|--|--|--|
| EasyAnimateV3-XL-2-InP-512x512.tar | EasyAnimateV3 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV3-XL-2-InP-512x512.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-512x512) | EasyAnimateV3 official weights for 512x512 text and image to video resolution. Training with 144 frames and fps 24 |
| EasyAnimateV3-XL-2-InP-768x768.tar | EasyAnimateV3 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV3-XL-2-InP-768x768.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-768x768) | EasyAnimateV3 official weights for 768x768 text and image to video resolution. Training with 144 frames and fps 24 |
| EasyAnimateV3-XL-2-InP-960x960.tar | EasyAnimateV3 | 18.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV3-XL-2-InP-960x960.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-960x960) | EasyAnimateV3 official weights for 960x960 text and image to video resolution. Training with 144 frames and fps 24 |
</details>
| Name | Storage Space | Url | Hugging Face | Description |
|--|--|--|--|--|
| CogVideoX-Fun-2b-InP.tar.gz | Before extraction:9.69 GB \/ After extraction: 13.0 GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/Diffusion_Transformer/CogVideoX-Fun-2b-InP.tar.gz) | [🤗Link](https://huggingface.co/alibaba-pai/CogVideoX-Fun-2b-InP)| Our official graph-generated video model is capable of predicting videos at multiple resolutions (512, 768, 1024, 1280) and has been trained on 144 frames at a rate of 24 frames per second. |
## Node types
- **LoadEasyAnimateModel**
- Loads the EasyAnimate model
- **LoadCogVideoX_Fun_Model**
- Loads the CogVideoX-Fun model
- **TextBox**
- Write the prompt for EasyAnimate model
- **EasyAnimateI2VSampler**
- EasyAnimate Sampler for Image to Video
- **EasyAnimateT2VSampler**
- EasyAnimate Sampler for Text to Video
- **EasyAnimateV2VSampler**
- EasyAnimate Sampler for Video to Video
- Write the prompt for CogVideoX-Fun model
- **CogVideoX_Fun_I2VSampler**
- CogVideoX-Fun Sampler for Image to Video
- **CogVideoX_Fun_T2VSampler**
- CogVideoX-Fun Sampler for Text to Video
- **CogVideoX_Fun_V2VSampler**
- CogVideoX-Fun Sampler for Video to Video
## Example workflows
### Video to video generation
Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_v2v.json) of the json:
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_v2v.jpg)
Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/cogvideoxfunv1_workflow_v2v.json) of the json:
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/comfyui_v2v.jpg)
You can run the demo using following video:
[demo video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/play_guitar.mp4)
[demo video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/play_guitar.mp4)
### Image to video generation
Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_i2v.json) of the json:
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_i2v.jpg)
Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/cogvideoxfunv1_workflow_i2v.json) of the json:
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/comfyui_i2v.jpg)
You can run the demo using following photo:
![demo image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/firework.png)
![demo image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png)
### Text to video generation
Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_t2v.json) of the json:
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_t2v.jpg)
### Text to video generation With Lora
We have provided a v4 version of the portrait Lora for testing, and the specific download link is [Lora download Link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimatev4_minimalism_lora.safetensors).
You can put this Lora at ```ComfyUI/models/loras/easyanimate```.
Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/easyanimatev4_workflow_lora.json) of the json:
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v4/comfyui_lora.jpg)
Our ui is shown as follow, this is the [download link](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/cogvideoxfunv1_workflow_t2v.json) of the json:
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/comfyui_t2v.jpg)
+21 -21
View File
@@ -22,9 +22,9 @@ from transformers import T5EncoderModel, T5Tokenizer
from ..cogvideox.data.bucket_sampler import ASPECT_RATIO_512, get_closest_ratio
from ..cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX
from ..cogvideox.models.transformer3d import CogVideoXTransformer3DModel
from ..cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline
from ..cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline
from ..cogvideox.pipeline.pipeline_cogvideox_inpaint import (
CogVideoX_FUN_Pipeline_Inpaint)
CogVideoX_Fun_Pipeline_Inpaint)
from ..cogvideox.utils.lora_utils import merge_lora, unmerge_lora
from ..cogvideox.utils.utils import (get_image_to_video_latent,
get_video_to_video_latent,
@@ -50,7 +50,7 @@ def to_pil(image):
return numpy2pil(image)
raise ValueError(f"Cannot convert {type(image)} to PIL.Image")
class LoadCogVideoX_FUN_Model:
class LoadCogVideoX_Fun_Model:
@classmethod
def INPUT_TYPES(s):
return {
@@ -94,11 +94,11 @@ class LoadCogVideoX_FUN_Model:
pbar = ProgressBar(3)
# Detect model is existing or not
model_path = os.path.join(folder_paths.models_dir, "CogVideoX_FUN", model)
model_path = os.path.join(folder_paths.models_dir, "CogVideoX_Fun", model)
if not os.path.exists(model_path):
if os.path.exists(eas_cache_dir):
model_path = os.path.join(eas_cache_dir, 'CogVideoX_FUN', model)
model_path = os.path.join(eas_cache_dir, 'CogVideoX_Fun', model)
else:
print(f"Please download cogvideoxfun model to: {model_path}")
@@ -125,7 +125,7 @@ class LoadCogVideoX_FUN_Model:
# Get pipeline
if transformer.config.in_channels != vae.config.latent_channels:
pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
model_path,
vae=vae,
transformer=transformer,
@@ -133,7 +133,7 @@ class LoadCogVideoX_FUN_Model:
torch_dtype=weight_dtype
)
else:
pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
model_path,
vae=vae,
transformer=transformer,
@@ -154,7 +154,7 @@ class LoadCogVideoX_FUN_Model:
}
return (cogvideoxfun_model,)
class LoadCogVideoX_FUN_Lora:
class LoadCogVideoX_Fun_Lora:
@classmethod
def INPUT_TYPES(s):
return {
@@ -200,7 +200,7 @@ class TextBox:
def process(self, prompt):
return (prompt, )
class CogVideoX_FUN_I2VSampler:
class CogVideoX_Fun_I2VSampler:
@classmethod
def INPUT_TYPES(s):
return {
@@ -320,7 +320,7 @@ class CogVideoX_FUN_I2VSampler:
return (videos,)
class CogVideoX_FUN_T2VSampler:
class CogVideoX_Fun_T2VSampler:
@classmethod
def INPUT_TYPES(s):
return {
@@ -435,7 +435,7 @@ class CogVideoX_FUN_T2VSampler:
pipeline = unmerge_lora(pipeline, _lora_path, _lora_weight)
return (videos,)
class CogVideoX_FUN_V2VSampler:
class CogVideoX_Fun_V2VSampler:
@classmethod
def INPUT_TYPES(s):
return {
@@ -559,19 +559,19 @@ class CogVideoX_FUN_V2VSampler:
NODE_CLASS_MAPPINGS = {
"TextBox": TextBox,
"LoadCogVideoX_FUN_Model": LoadCogVideoX_FUN_Model,
"LoadCogVideoX_FUN_Lora": LoadCogVideoX_FUN_Lora,
"CogVideoX_FUN_I2VSampler": CogVideoX_FUN_I2VSampler,
"CogVideoX_FUN_T2VSampler": CogVideoX_FUN_T2VSampler,
"CogVideoX_FUN_V2VSampler": CogVideoX_FUN_V2VSampler,
"LoadCogVideoX_Fun_Model": LoadCogVideoX_Fun_Model,
"LoadCogVideoX_Fun_Lora": LoadCogVideoX_Fun_Lora,
"CogVideoX_Fun_I2VSampler": CogVideoX_Fun_I2VSampler,
"CogVideoX_Fun_T2VSampler": CogVideoX_Fun_T2VSampler,
"CogVideoX_Fun_V2VSampler": CogVideoX_Fun_V2VSampler,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"TextBox": "TextBox",
"LoadCogVideoX_FUN_Model": "Load CogVideoX-Fun Model",
"LoadCogVideoX_FUN_Lora": "Load CogVideoX-Fun Lora",
"CogVideoX_FUN_I2VSampler": "CogVideoX-Fun Sampler for Image to Video",
"CogVideoX_FUN_T2VSampler": "CogVideoX-Fun Sampler for Text to Video",
"CogVideoX_FUN_V2VSampler": "CogVideoX-Fun Sampler for Video to Video",
"LoadCogVideoX_Fun_Model": "Load CogVideoX-Fun Model",
"LoadCogVideoX_Fun_Lora": "Load CogVideoX-Fun Lora",
"CogVideoX_Fun_I2VSampler": "CogVideoX-Fun Sampler for Image to Video",
"CogVideoX_Fun_T2VSampler": "CogVideoX-Fun Sampler for Text to Video",
"CogVideoX_Fun_V2VSampler": "CogVideoX-Fun Sampler for Video to Video",
}
+4 -4
View File
@@ -115,7 +115,7 @@
},
{
"id": 83,
"type": "LoadCogVideoX_FUN_Model",
"type": "LoadCogVideoX_Fun_Model",
"pos": [
300,
-294
@@ -139,7 +139,7 @@
}
],
"properties": {
"Node name for S&R": "LoadCogVideoX_FUN_Model"
"Node name for S&R": "LoadCogVideoX_Fun_Model"
},
"widgets_values": [
"CogVideoX-Fun-2b-InP",
@@ -215,7 +215,7 @@
},
{
"id": 82,
"type": "CogVideoX_FUN_I2VSampler",
"type": "CogVideoX_Fun_I2VSampler",
"pos": [
758,
93
@@ -267,7 +267,7 @@
}
],
"properties": {
"Node name for S&R": "CogVideoX_FUN_I2VSampler"
"Node name for S&R": "CogVideoX_Fun_I2VSampler"
},
"widgets_values": [
49,
+4 -4
View File
@@ -83,7 +83,7 @@
},
{
"id": 87,
"type": "LoadCogVideoX_FUN_Model",
"type": "LoadCogVideoX_Fun_Model",
"pos": [
302,
-285
@@ -107,7 +107,7 @@
}
],
"properties": {
"Node name for S&R": "LoadCogVideoX_FUN_Model"
"Node name for S&R": "LoadCogVideoX_Fun_Model"
},
"widgets_values": [
"CogVideoX-Fun-2b-InP",
@@ -150,7 +150,7 @@
},
{
"id": 88,
"type": "CogVideoX_FUN_T2VSampler",
"type": "CogVideoX_Fun_T2VSampler",
"pos": [
728,
-68
@@ -192,7 +192,7 @@
}
],
"properties": {
"Node name for S&R": "CogVideoX_FUN_T2VSampler"
"Node name for S&R": "CogVideoX_Fun_T2VSampler"
},
"widgets_values": [
49,
+4 -4
View File
@@ -189,7 +189,7 @@
},
{
"id": 88,
"type": "LoadCogVideoX_FUN_Model",
"type": "LoadCogVideoX_Fun_Model",
"pos": [
309,
-286
@@ -213,7 +213,7 @@
}
],
"properties": {
"Node name for S&R": "LoadCogVideoX_FUN_Model"
"Node name for S&R": "LoadCogVideoX_Fun_Model"
},
"widgets_values": [
"CogVideoX-Fun-2b-InP",
@@ -256,7 +256,7 @@
},
{
"id": 87,
"type": "CogVideoX_FUN_V2VSampler",
"type": "CogVideoX_Fun_V2VSampler",
"pos": [
778,
93
@@ -304,7 +304,7 @@
}
],
"properties": {
"Node name for S&R": "CogVideoX_FUN_V2VSampler"
"Node name for S&R": "CogVideoX_Fun_V2VSampler"
},
"widgets_values": [
49,
+4 -4
View File
@@ -15,8 +15,8 @@ from PIL import Image
from cogvideox.models.transformer3d import CogVideoXTransformer3DModel
from cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint
from cogvideox.utils.lora_utils import merge_lora, unmerge_lora
from cogvideox.utils.utils import get_image_to_video_latent, save_videos_grid
@@ -112,7 +112,7 @@ scheduler = Choosen_Scheduler.from_pretrained(
)
if transformer.config.in_channels != vae.config.latent_channels:
pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
model_name,
vae=vae,
text_encoder=text_encoder,
@@ -121,7 +121,7 @@ if transformer.config.in_channels != vae.config.latent_channels:
torch_dtype=weight_dtype
)
else:
pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
model_name,
vae=vae,
text_encoder=text_encoder,
+4 -4
View File
@@ -15,8 +15,8 @@ from PIL import Image
from cogvideox.models.transformer3d import CogVideoXTransformer3DModel
from cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint
from cogvideox.utils.lora_utils import merge_lora, unmerge_lora
from cogvideox.utils.utils import get_image_to_video_latent, save_videos_grid
@@ -103,7 +103,7 @@ scheduler = Choosen_Scheduler.from_pretrained(
)
if transformer.config.in_channels != vae.config.latent_channels:
pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
model_name,
vae=vae,
text_encoder=text_encoder,
@@ -112,7 +112,7 @@ if transformer.config.in_channels != vae.config.latent_channels:
torch_dtype=weight_dtype
)
else:
pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
model_name,
vae=vae,
text_encoder=text_encoder,
+4 -4
View File
@@ -15,9 +15,9 @@ from transformers import (CLIPImageProcessor, CLIPVisionModelWithProjection,
from cogvideox.models.autoencoder_magvit import AutoencoderKLCogVideoX
from cogvideox.models.transformer3d import CogVideoXTransformer3DModel
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import \
CogVideoX_FUN_Pipeline_Inpaint
CogVideoX_Fun_Pipeline_Inpaint
from cogvideox.utils.lora_utils import merge_lora, unmerge_lora
from cogvideox.utils.utils import get_video_to_video_latent, save_videos_grid
@@ -108,7 +108,7 @@ scheduler = Choosen_Scheduler.from_pretrained(
)
if transformer.config.in_channels != vae.config.latent_channels:
pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
model_name,
vae=vae,
text_encoder=text_encoder,
@@ -117,7 +117,7 @@ if transformer.config.in_channels != vae.config.latent_channels:
torch_dtype=weight_dtype
)
else:
pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
model_name,
vae=vae,
text_encoder=text_encoder,
+4 -4
View File
@@ -72,8 +72,8 @@ from cogvideox.data.dataset_image_video import (ImageVideoDataset,
ImageVideoSampler,
get_random_mask)
from cogvideox.models.transformer3d import CogVideoXTransformer3DModel
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint
from cogvideox.utils.utils import get_image_to_video_latent, save_videos_grid
if is_wandb_available():
@@ -163,7 +163,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, args, accelerato
transformer3d_val.load_state_dict(accelerator.unwrap_model(transformer3d).state_dict())
if args.train_mode != "normal":
pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
args.pretrained_model_name_or_path,
vae=accelerator.unwrap_model(vae).to(weight_dtype),
text_encoder=accelerator.unwrap_model(text_encoder),
@@ -172,7 +172,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, args, accelerato
torch_dtype=weight_dtype
)
else:
pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
args.pretrained_model_name_or_path,
vae=accelerator.unwrap_model(vae).to(weight_dtype),
text_encoder=accelerator.unwrap_model(text_encoder),
+4 -4
View File
@@ -70,8 +70,8 @@ from cogvideox.data.bucket_sampler import (ASPECT_RATIO_512,
AspectRatioBatchImageVideoSampler,
AspectRatioBatchSampler,
RandomSampler, get_closest_ratio)
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_FUN_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_FUN_Pipeline_Inpaint
from cogvideox.pipeline.pipeline_cogvideox import CogVideoX_Fun_Pipeline
from cogvideox.pipeline.pipeline_cogvideox_inpaint import CogVideoX_Fun_Pipeline_Inpaint
from cogvideox.data.dataset_image import CC15M
from cogvideox.data.dataset_image_video import (ImageVideoDataset,
ImageVideoSampler,
@@ -168,7 +168,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, network, args, a
transformer3d_val.load_state_dict(accelerator.unwrap_model(transformer3d).state_dict())
if args.train_mode != "normal":
pipeline = CogVideoX_FUN_Pipeline_Inpaint.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline_Inpaint.from_pretrained(
args.pretrained_model_name_or_path,
vae=accelerator.unwrap_model(vae).to(weight_dtype),
text_encoder=accelerator.unwrap_model(text_encoder),
@@ -177,7 +177,7 @@ def log_validation(vae, text_encoder, tokenizer, transformer3d, network, args, a
torch_dtype=weight_dtype,
)
else:
pipeline = CogVideoX_FUN_Pipeline.from_pretrained(
pipeline = CogVideoX_Fun_Pipeline.from_pretrained(
args.pretrained_model_name_or_path,
vae=accelerator.unwrap_model(vae).to(weight_dtype),
text_encoder=accelerator.unwrap_model(text_encoder),