From 9dbb4f88d417bece937be4107aba36a32740c796 Mon Sep 17 00:00:00 2001 From: hkz Date: Wed, 12 Feb 2025 10:06:07 +0800 Subject: [PATCH] Update 7b && Support low vram inference (#196) --------- Co-authored-by: bubbliiiing <3323290568@qq.com> --- README.md | 17 ++-- README_ja-JP.md | 8 +- README_zh-CN.md | 16 ++-- app.py | 5 +- comfyui/README.md | 8 ++ comfyui/README_zh-CN.md | 9 ++ comfyui/comfyui_nodes.py | 26 +++++- ...syanimatev5.1_workflow_control_camera.json | 2 +- ...imatev5.1_workflow_control_trajectory.json | 4 +- .../v5.1/easyanimatev5.1_workflow_i2v.json | 2 +- .../v5.1/easyanimatev5.1_workflow_t2v.json | 4 +- .../v5.1/easyanimatev5.1_workflow_v2v.json | 4 +- .../easyanimatev5.1_workflow_v2v_control.json | 4 +- easyanimate/models/processor.py | 11 +-- easyanimate/models/transformer3d.py | 32 +++++-- easyanimate/pipeline/pipeline_easyanimate.py | 72 +++++++++++++++- .../pipeline/pipeline_easyanimate_control.py | 72 +++++++++++++++- .../pipeline/pipeline_easyanimate_inpaint.py | 86 +++++++++++++++++-- easyanimate/ui/ui.py | 69 ++++++++++----- easyanimate/utils/fp8_optimization.py | 9 +- .../vae/ldm/data/dataset_image_video.py | 2 +- .../compute_semantic_consistency.py | 1 + .../video_caption/compute_video_quality.py | 5 +- .../video_caption/filter_meta_train.py | 8 +- predict_i2v.py | 33 ++++--- predict_t2v.py | 44 ++++++---- predict_v2v.py | 36 +++++--- predict_v2v_control.py | 48 +++++++---- scripts/README_TRAIN_REWARD.md | 3 + scripts/train.py | 2 - scripts/train_control.py | 2 - scripts/train_lora.py | 2 - 32 files changed, 500 insertions(+), 146 deletions(-) mode change 100644 => 100755 app.py mode change 100644 => 100755 comfyui/README.md mode change 100644 => 100755 comfyui/README_zh-CN.md mode change 100644 => 100755 comfyui/comfyui_nodes.py mode change 100644 => 100755 easyanimate/models/transformer3d.py mode change 100644 => 100755 easyanimate/pipeline/pipeline_easyanimate.py mode change 100644 => 100755 easyanimate/pipeline/pipeline_easyanimate_control.py mode change 100644 => 100755 easyanimate/pipeline/pipeline_easyanimate_inpaint.py mode change 100644 => 100755 easyanimate/utils/fp8_optimization.py mode change 100644 => 100755 predict_i2v.py mode change 100644 => 100755 predict_t2v.py mode change 100644 => 100755 predict_v2v.py mode change 100644 => 100755 predict_v2v_control.py mode change 100644 => 100755 scripts/README_TRAIN_REWARD.md mode change 100644 => 100755 scripts/train.py mode change 100644 => 100755 scripts/train_control.py mode change 100644 => 100755 scripts/train_lora.py diff --git a/README.md b/README.md index 2f275bb..9b40ba3 100755 --- a/README.md +++ b/README.md @@ -117,22 +117,19 @@ We need about 60GB available on disk (for saving weights), please check! The video size for EasyAnimateV5.1-12B can be generated by different GPU Memory, including: | GPU memory | 384x672x25 | 384x672x49 | 576x1008x25 | 576x1008x49 | 768x1344x25 | 768x1344x49 | |------------|------------|------------|------------|------------|------------|------------| -| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ | +| 16GB | 🧡 | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ | | 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ | | 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | -Due to the float16 weights of qwen2-vl-7b, it cannot run on a 16GB GPU. If your GPU memory is 16GB, please visit [Huggingface](https://huggingface.co/Qwen/Qwen2-VL-7B-Instruct-GPTQ-Int8) or [Modelscope](https://modelscope.cn/models/Qwen/Qwen2-VL-7B-Instruct-GPTQ-Int8) to download the quantized version of qwen2-vl-7b to replace the original text encoder, and install the corresponding dependency libraries (auto-gptq, optimum). - -The video size for EasyAnimateV5-7B can be generated by different GPU Memory, including: +The video size for EasyAnimateV5.1-7B can be generated by different GPU Memory, including: | GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49| |----------|----------|----------|----------|----------|----------|----------| -| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ | +| 16GB | 🧡 | 🧡 | ⭕️ | ⭕️ | ❌ | ❌ | | 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ | | 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | - ✅ indicates it can run under "model_cpu_offload", 🧡 represents it can run under "model_cpu_offload_and_qfloat8", ⭕️ indicates it can run under "sequential_cpu_offload", ❌ means it can't run. Please note that running with sequential_cpu_offload will be slower. Some GPUs that do not support torch.bfloat16, such as 2080ti and V100, require changing the weight_dtype in app.py and predict files to torch.float16 in order to run. @@ -501,6 +498,14 @@ For details on setting some parameters, please refer to [Readme Train](scripts/R EasyAnimateV5.1: +7B: +| Name | Type | Storage Space | Hugging Face | Model Scope | Description | +|--|--|--|--|--|--| +| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | +| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, and trajectory control. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | +| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera) | Official video camera control weights, supporting direction generation control by inputting camera motion trajectories. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | +| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | + 12B: | Name | Type | Storage Space | Hugging Face | Model Scope | Description | |--|--|--|--|--|--| diff --git a/README_ja-JP.md b/README_ja-JP.md index 155906a..0173d91 100755 --- a/README_ja-JP.md +++ b/README_ja-JP.md @@ -118,17 +118,15 @@ Linuxの詳細: EasyAnimateV5.1-12Bのビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください: | GPUメモリ |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49| |----------|----------|----------|----------|----------|----------|----------| -| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ | +| 16GB | 🧡 | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ | | 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ | | 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | -qwen2-vl-7bのfloat16の重みのため、16GBのVRAMでは実行できません。もしお使いのVRAMが16GBである場合は、[Huggingface](https://huggingface.co/Qwen/Qwen2-VL-7B-Instruct-GPTQ- - -EasyAnimateV5-7Bのビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください: +EasyAnimateV5.1-7Bのビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください: | GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49| |----------|----------|----------|----------|----------|----------|----------| -| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ | +| 16GB | 🧡 | 🧡 | ⭕️ | ⭕️ | ❌ | ❌ | | 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ | | 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | diff --git a/README_zh-CN.md b/README_zh-CN.md index 880fc13..2f0a179 100755 --- a/README_zh-CN.md +++ b/README_zh-CN.md @@ -115,17 +115,15 @@ Linux 的详细信息: EasyAnimateV5.1-12B的视频大小可以由不同的GPU Memory生成,包括: | GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49| |----------|----------|----------|----------|----------|----------|----------| -| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ | +| 16GB | 🧡 | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ | | 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ | | 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | -由于qwen2-vl-7b的float16的权重,无法在16GB显存下运行,如果您的显存是16GB,请前往[Huggingface](https://huggingface.co/Qwen/Qwen2-VL-7B-Instruct-GPTQ-Int8)或者[Modelscope](https://modelscope.cn/models/Qwen/Qwen2-VL-7B-Instruct-GPTQ-Int8)下载量化后的qwen2-vl-7b对原有的text encoder进行替换,并安装对应的依赖库(auto-gptq, optimum)。 - -EasyAnimateV5-7B的视频大小可以由不同的GPU Memory生成,包括: +EasyAnimateV5.1-7B的视频大小可以由不同的GPU Memory生成,包括: | GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49| |----------|----------|----------|----------|----------|----------|----------| -| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ | +| 16GB | 🧡 | 🧡 | ⭕️ | ⭕️ | ❌ | ❌ | | 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ | | 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | @@ -495,6 +493,14 @@ sh scripts/train.sh # 模型地址 EasyAnimateV5.1: +7B: +| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 | +|--|--|--|--|--|--| +| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | +| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control)| 官方的视频控制权重,支持不同的控制条件,如Canny、Depth、Pose、MLSD等,同时支持使用轨迹控制。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | +| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera)| 官方的视频相机控制权重,支持通过输入相机运动轨迹控制生成方向。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | +| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh)| 官方的文生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | + 12B: | 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 | |--|--|--|--|--|--| diff --git a/app.py b/app.py old mode 100644 new mode 100755 index eec6879..77ac7d5 --- a/app.py +++ b/app.py @@ -21,14 +21,13 @@ if __name__ == "__main__": # resulting in slower speeds but saving a large amount of GPU memory. # # EasyAnimateV1, V2 and V3 support "model_cpu_offload" "sequential_cpu_offload" - # EasyAnimateV4, V5 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" - # EasyAnimateV5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" + # EasyAnimateV4, V5 and V5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" GPU_memory_mode = "model_cpu_offload_and_qfloat8" # EasyAnimateV5.1 support TeaCache. enable_teacache = True # Recommended to be set between 0.05 and 0.1. A larger threshold can cache more steps, speeding up the inference process, # but it may cause slight differences between the generated content and the original content. - teacache_threshold = 0.1 + teacache_threshold = 0.08 # Use torch.float16 if GPU does not support torch.bfloat16 # ome graphics cards, such as v100, 2080ti, do not support torch.bfloat16 weight_dtype = torch.bfloat16 diff --git a/comfyui/README.md b/comfyui/README.md old mode 100644 new mode 100755 index c92fe83..2875390 --- a/comfyui/README.md +++ b/comfyui/README.md @@ -38,6 +38,14 @@ pip install -r comfyui/requirements.txt EasyAnimateV5.1: +7B: +| Name | Type | Storage Space | Hugging Face | Model Scope | Description | +|--|--|--|--|--|--| +| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | +| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, and trajectory control. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | +| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera) | Official video camera control weights, supporting direction generation control by inputting camera motion trajectories. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | +| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. | + 12B: | Name | Type | Storage Space | Hugging Face | Model Scope | Description | |--|--|--|--|--|--| diff --git a/comfyui/README_zh-CN.md b/comfyui/README_zh-CN.md old mode 100644 new mode 100755 index bb53015..14d6bbc --- a/comfyui/README_zh-CN.md +++ b/comfyui/README_zh-CN.md @@ -36,6 +36,15 @@ pip install -r comfyui/requirements.txt ## 将模型下载到`ComfyUI/models/EasyAnimate/` EasyAnimateV5.1: + +7B: +| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 | +|--|--|--|--|--|--| +| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | +| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control)| 官方的视频控制权重,支持不同的控制条件,如Canny、Depth、Pose、MLSD等,同时支持使用轨迹控制。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | +| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera)| 官方的视频相机控制权重,支持通过输入相机运动轨迹控制生成方向。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | +| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh)| 官方的文生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 | + 12B: |名称|类型|存储空间|拥抱面|型号范围|描述| |--|--|--|--|--|--| diff --git a/comfyui/comfyui_nodes.py b/comfyui/comfyui_nodes.py old mode 100644 new mode 100755 index c9cf642..3977aeb --- a/comfyui/comfyui_nodes.py +++ b/comfyui/comfyui_nodes.py @@ -39,7 +39,8 @@ from ..easyanimate.pipeline.pipeline_easyanimate_control import \ from ..easyanimate.utils.lora_utils import merge_lora, unmerge_lora from ..easyanimate.utils.utils import (get_image_to_video_latent, get_image_latent, get_video_to_video_latent) -from ..easyanimate.utils.fp8_optimization import convert_weight_dtype_wrapper +from ..easyanimate.utils.fp8_optimization import (convert_model_weight_to_float8, + convert_weight_dtype_wrapper) from ..easyanimate.ui.ui import ddpm_scheduler_dict, flow_scheduler_dict, all_cheduler_dict # Compatible with Alibaba EAS for quick launch @@ -98,6 +99,11 @@ class LoadEasyAnimateModel: 'EasyAnimateV5-12b-zh-InP', 'EasyAnimateV5-12b-zh-Control', 'EasyAnimateV5-12b-zh', + 'EasyAnimateV5.1-7b-zh', + 'EasyAnimateV5.1-7b-zh-InP', + 'EasyAnimateV5.1-7b-zh-Control', + 'EasyAnimateV5.1-7b-zh-Control-Camera', + 'EasyAnimateV5.1-12b-zh', 'EasyAnimateV5.1-12b-zh-InP', 'EasyAnimateV5.1-12b-zh-Control', 'EasyAnimateV5.1-12b-zh-Control-Camera', @@ -174,7 +180,7 @@ class LoadEasyAnimateModel: model_name, subfolder="vae" ).to(weight_dtype) - if config['vae_kwargs'].get('vae_type', 'AutoencoderKL') == 'AutoencoderKLMagvit' and weight_dtype == torch.float16: + if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): vae.upcast_vae = True # Update pbar pbar.update(1) @@ -185,7 +191,7 @@ class LoadEasyAnimateModel: ] transformer_additional_kwargs = OmegaConf.to_container(config['transformer_additional_kwargs']) - if weight_dtype == torch.float16: + if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): transformer_additional_kwargs["upcast_attention"] = True transformer = Choosen_Transformer3DModel.from_pretrained_2d( @@ -299,11 +305,23 @@ class LoadEasyAnimateModel: transformer=transformer, scheduler=scheduler, ) + if GPU_memory_mode == "sequential_cpu_offload": + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [] + for name, _text_encoder in zip(["text_encoder", "text_encoder_2"], [pipeline.text_encoder, pipeline.text_encoder_2]): + if isinstance(_text_encoder, Qwen2VLForConditionalGeneration): + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_model_weight_to_float8(_text_encoder) + convert_weight_dtype_wrapper(_text_encoder, weight_dtype) + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [name] pipeline.enable_sequential_cpu_offload() elif GPU_memory_mode == "model_cpu_offload_and_qfloat8": - pipeline.enable_model_cpu_offload() + for _text_encoder in [pipeline.text_encoder, pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual convert_weight_dtype_wrapper(transformer, weight_dtype) + pipeline.enable_model_cpu_offload() else: pipeline.enable_model_cpu_offload() easyanimate_model = { diff --git a/comfyui/v5.1/easyanimatev5.1_workflow_control_camera.json b/comfyui/v5.1/easyanimatev5.1_workflow_control_camera.json index 55a7555..b0dd3f4 100644 --- a/comfyui/v5.1/easyanimatev5.1_workflow_control_camera.json +++ b/comfyui/v5.1/easyanimatev5.1_workflow_control_camera.json @@ -547,7 +547,7 @@ 6, 1, "Flow", - 0.10, + 0.08, true, "" ] diff --git a/comfyui/v5.1/easyanimatev5.1_workflow_control_trajectory.json b/comfyui/v5.1/easyanimatev5.1_workflow_control_trajectory.json index 7d88aa5..dd2bb81 100644 --- a/comfyui/v5.1/easyanimatev5.1_workflow_control_trajectory.json +++ b/comfyui/v5.1/easyanimatev5.1_workflow_control_trajectory.json @@ -140,7 +140,7 @@ "Node name for S&R": "EasyAnimate_TextBox" }, "widgets_values": [ - "一只棕褐色的狗在摇晃脑袋,坐在一个舒适的房间里的浅色沙发上。在狗的后面,架子上有一幅镶框的画,周围是粉红色的花朵。房间里的灯光柔和温暖,营造出舒适的氛围。" + "一只棕褐色的狗正摇晃着脑袋,坐在一个舒适的房间里的浅色沙发上。沙发看起来柔软而宽敞,为这只活泼的狗狗提供了一个完美的休息地点。在狗的后面,靠墙摆放着一个架子,架子上挂着一幅精美的镶框画,画中描绘着一些美丽的风景或场景。画框周围装饰着粉红色的花朵,这些花朵不仅增添了房间的色彩,还带来了一丝自然和生机。房间里的灯光柔和而温暖,从天花板上的吊灯和角落里的台灯散发出来,营造出一种温馨舒适的氛围。整个空间给人一种宁静和谐的感觉,仿佛时间在这里变得缓慢而美好。" ] }, { @@ -776,7 +776,7 @@ 6, 1, "Flow", - 0.10, + 0.08, true, "" ] diff --git a/comfyui/v5.1/easyanimatev5.1_workflow_i2v.json b/comfyui/v5.1/easyanimatev5.1_workflow_i2v.json index 6d96d3c..a7df65e 100644 --- a/comfyui/v5.1/easyanimatev5.1_workflow_i2v.json +++ b/comfyui/v5.1/easyanimatev5.1_workflow_i2v.json @@ -290,7 +290,7 @@ 50, 6, "Flow", - 0.10, + 0.08, true ] }, diff --git a/comfyui/v5.1/easyanimatev5.1_workflow_t2v.json b/comfyui/v5.1/easyanimatev5.1_workflow_t2v.json index b8496c4..b737ed2 100644 --- a/comfyui/v5.1/easyanimatev5.1_workflow_t2v.json +++ b/comfyui/v5.1/easyanimatev5.1_workflow_t2v.json @@ -85,7 +85,7 @@ 50, 6, "Flow", - 0.10, + 0.08, true ] }, @@ -321,7 +321,7 @@ "Node name for S&R": "EasyAnimate_TextBox" }, "widgets_values": [ - "一只棕褐色的狗在摇晃脑袋,坐在一个舒适的房间里的浅色沙发上。在狗的后面,架子上有一幅镶框的画,周围是粉红色的花朵。房间里的灯光柔和温暖,营造出舒适的氛围。" + "一只棕褐色的狗正摇晃着脑袋,坐在一个舒适的房间里的浅色沙发上。沙发看起来柔软而宽敞,为这只活泼的狗狗提供了一个完美的休息地点。在狗的后面,靠墙摆放着一个架子,架子上挂着一幅精美的镶框画,画中描绘着一些美丽的风景或场景。画框周围装饰着粉红色的花朵,这些花朵不仅增添了房间的色彩,还带来了一丝自然和生机。房间里的灯光柔和而温暖,从天花板上的吊灯和角落里的台灯散发出来,营造出一种温馨舒适的氛围。整个空间给人一种宁静和谐的感觉,仿佛时间在这里变得缓慢而美好。" ] } ], diff --git a/comfyui/v5.1/easyanimatev5.1_workflow_v2v.json b/comfyui/v5.1/easyanimatev5.1_workflow_v2v.json index 419526d..b3028c8 100644 --- a/comfyui/v5.1/easyanimatev5.1_workflow_v2v.json +++ b/comfyui/v5.1/easyanimatev5.1_workflow_v2v.json @@ -117,7 +117,7 @@ "Node name for S&R": "EasyAnimate_TextBox" }, "widgets_values": [ - "一只穿着小外套的猫咪正在花园秋千上安静地弹吉他。晚霞的余光洒在它柔软的毛皮上,和煦的微风轻轻拂过,周围斑驳的光影随着音乐的旋律轻轻摇曳。" + "一只穿着小外套的猫咪正安静地坐在花园的秋千上弹吉他。它的小外套精致而合身,增添了几分俏皮与可爱。晚霞的余光洒在它柔软的毛皮上,给它的毛发镀上了一层温暖的金色光辉。和煦的微风轻轻拂过,带来阵阵花香和草木的气息,令人心旷神怡。周围斑驳的光影随着音乐的旋律轻轻摇曳,仿佛整个花园都在为这只小猫咪的演奏伴舞。阳光透过树叶间的缝隙,投下一片片光影交错的图案,与悠扬的吉他声交织在一起,营造出一种梦幻而宁静的氛围。猫咪专注而投入地弹奏着,每一个音符都似乎充满了魔力,让这个傍晚变得更加美好。" ] }, { @@ -305,7 +305,7 @@ 6, 0.7000000000000001, "Flow", - 0.10, + 0.08, true, "" ] diff --git a/comfyui/v5.1/easyanimatev5.1_workflow_v2v_control.json b/comfyui/v5.1/easyanimatev5.1_workflow_v2v_control.json index 3ef4133..935194f 100644 --- a/comfyui/v5.1/easyanimatev5.1_workflow_v2v_control.json +++ b/comfyui/v5.1/easyanimatev5.1_workflow_v2v_control.json @@ -117,7 +117,7 @@ "Node name for S&R": "EasyAnimate_TextBox" }, "widgets_values": [ - "一个穿着及膝白色无袖连衣裙和白色高跟凉鞋的美女在一个光线充足、木地板的房间里跳舞。房间的背景是一扇紧闭的门、一个展示透明玻璃瓶酒精饮料的架子和一个部分可见的深色沙发。" + "在这个阳光明媚的户外花园里,美女身穿一袭及膝的白色无袖连衣裙,裙摆在她轻盈的舞姿中轻柔地摆动,宛如一只翩翩起舞的蝴蝶。阳光透过树叶间洒下斑驳的光影,映衬出她柔和的脸庞和清澈的眼眸,显得格外优雅。仿佛每一个动作都在诉说着青春与活力,她在草地上旋转,裙摆随之飞扬,仿佛整个花园都因她的舞动而欢愉。周围五彩缤纷的花朵在微风中摇曳,玫瑰、菊花、百合,各自释放出阵阵香气,营造出一种轻松而愉快的氛围。" ] }, { @@ -309,7 +309,7 @@ 6, 1, "Flow", - 0.10, + 0.08, true, "" ] diff --git a/easyanimate/models/processor.py b/easyanimate/models/processor.py index 0cea72e..77439e2 100644 --- a/easyanimate/models/processor.py +++ b/easyanimate/models/processor.py @@ -318,8 +318,8 @@ except: print("Flash Attention is not installed. Please install with `pip install flash-attn`, if you want to use SWA.") class EasyAnimateSWAttnProcessor2_0: - def __init__(self, window_size=1024): - self.window_size = window_size + def __init__(self, cross_attention_size=1024): + self.cross_attention_size = cross_attention_size def __call__( self, @@ -334,6 +334,7 @@ class EasyAnimateSWAttnProcessor2_0: attn2: Attention = None, ) -> torch.Tensor: text_seq_length = encoder_hidden_states.size(1) + windows_size = height * width batch_size, sequence_length, _ = ( hidden_states.shape if encoder_hidden_states is None else encoder_hidden_states.shape @@ -387,7 +388,7 @@ class EasyAnimateSWAttnProcessor2_0: query = query.transpose(1, 2).to(value) key = key.transpose(1, 2).to(value) - interval = max((query.size(1) - text_seq_length) // (self.window_size - text_seq_length), 1) + interval = max((query.size(1) - text_seq_length) // (self.cross_attention_size - text_seq_length), 1) cross_key = torch.cat([key[:, :text_seq_length], key[:, text_seq_length::interval]], dim=1) cross_val = torch.cat([value[:, :text_seq_length], value[:, text_seq_length::interval]], dim=1) @@ -418,8 +419,8 @@ class EasyAnimateSWAttnProcessor2_0: value = torch.cat(new_values, dim=2) # apply attention - hidden_states = flash_attn_func(query, key, value, dropout_p=0.0, causal=False, window_size=(self.window_size, self.window_size)) - + hidden_states = flash_attn_func(query, key, value, dropout_p=0.0, causal=False, window_size=(windows_size, windows_size)) + hidden_states = torch.tensor_split(hidden_states, 6, 2) new_hidden_states = [hidden_states[0]] for index, mode in enumerate( diff --git a/easyanimate/models/transformer3d.py b/easyanimate/models/transformer3d.py old mode 100644 new mode 100755 index 53b7ead..c1a86a6 --- a/easyanimate/models/transformer3d.py +++ b/easyanimate/models/transformer3d.py @@ -121,6 +121,22 @@ class TeaCache(): self.previous_residual = None +def get_teacache_coefficients(model_name): + # The coefficients for EasyAnimateV5-7b-zh-InP should be: + # [-3.64204720e+03, 1.43764725e+03, -1.93045263e+02, 1.09596499e+01, -1.70663507e-01] + if "v5.1-7b" in model_name.lower(): + # The coefficient was obtained by sampling videos from T2V CompBench using EasyAnimateV5.1-7b-zh-InP. + # This coefficient can be applied to both the EasyAnimateV5.1-7b-zh and EasyAnimateV5.1-7b-Control. + return [1.07862322, -4.19362456, 3.06725828, 0.33161686, 0.02374758] + elif "v5.1-12b" in model_name.lower(): + # The coefficient was obtained by sampling videos from T2V CompBench using EasyAnimateV5.1-12b-zh-InP. + # This coefficient can be applied to both the EasyAnimateV5.1-12b-zh and EasyAnimateV5.1-12b-Control. + return [-10.47857366, 8.33844143, -0.78477557, 0.68798618, 0.0136149] + else: + print(f"The model {model_name} is not supported by TeaCache.") + return None + + class Transformer3DModel(ModelMixin, ConfigMixin): """ A 3D Transformer model for image-like data. @@ -1472,10 +1488,6 @@ class EasyAnimateTransformer3DModel(ModelMixin, ConfigMixin): rel_l1_thresh: float, coefficients: list[float] = [-10.47857366, 8.33844143, -0.78477557, 0.68798618, 0.0136149] ): - # The coefficient was obtained by sampling videos from T2V CompBench using EasyAnimateV5.1-12b-zh-InP. - # This coefficient can be applied to both the EasyAnimateV5.1-12b-zh and EasyAnimateV5.1-12b-Control. - # The coefficients for EasyAnimateV5.1-7b-zh-InP should be: - # [-3.64204720e+03, 1.43764725e+03, -1.93045263e+02, 1.09596499e+01, -1.70663507e-01] self.teacache = TeaCache(coefficients, num_steps, rel_l1_thresh=rel_l1_thresh) def _set_gradient_checkpointing(self, module, value=False): @@ -1558,25 +1570,26 @@ class EasyAnimateTransformer3DModel(ModelMixin, ConfigMixin): should_calc = True self.teacache.accumulated_rel_l1_distance = 0 else: - rel_l1_distance = self.teacache.compute_rel_l1_distance(self.teacache.previous_modulated_input, modulated_inp) + rel_l1_distance = self.teacache.compute_rel_l1_distance(self.teacache.previous_modulated_input.to(modulated_inp.device), modulated_inp) self.teacache.accumulated_rel_l1_distance += self.teacache.rescale_func(rel_l1_distance) if self.teacache.accumulated_rel_l1_distance < self.teacache.rel_l1_thresh: should_calc = False else: should_calc = True self.teacache.accumulated_rel_l1_distance = 0 - self.teacache.previous_modulated_input = modulated_inp + self.teacache.previous_modulated_input = modulated_inp.cpu() self.teacache.cnt += 1 if self.teacache.cnt == self.teacache.num_steps: # self.cnt = 0 self.teacache.reset() + del inp, temb_, encoder_hidden_states_ # TeaCache if self.teacache is not None: if not should_calc: - hidden_states += self.teacache.previous_residual + hidden_states += self.teacache.previous_residual.to(modulated_inp.device) else: - ori_hidden_states = hidden_states.clone() + ori_hidden_states = hidden_states.clone().cpu() # 4. Transformer blocks for i, block in enumerate(self.transformer_blocks): @@ -1619,7 +1632,8 @@ class EasyAnimateTransformer3DModel(ModelMixin, ConfigMixin): # 5. Final block hidden_states = self.norm_out(hidden_states, temb=temb) - self.teacache.previous_residual = hidden_states - ori_hidden_states + self.teacache.previous_residual = hidden_states.cpu() - ori_hidden_states + del ori_hidden_states else: # 4. Transformer blocks for i, block in enumerate(self.transformer_blocks): diff --git a/easyanimate/pipeline/pipeline_easyanimate.py b/easyanimate/pipeline/pipeline_easyanimate.py old mode 100644 new mode 100755 index 79b84f6..9258abd --- a/easyanimate/pipeline/pipeline_easyanimate.py +++ b/easyanimate/pipeline/pipeline_easyanimate.py @@ -240,14 +240,69 @@ class EasyAnimatePipeline(DiffusionPipeline): ) self.vae_scale_factor = 2 ** (len(self.vae.config.block_out_channels) - 1) + self.manual_cpu_offload_flag = False + + def enable_sequential_cpu_offload(self, gpu_id: Optional[int] = None, device: Union[torch.device, str] = "cuda"): + from diffusers.pipelines.pipeline_utils import is_accelerate_available, is_accelerate_version + + if is_accelerate_available() and is_accelerate_version(">=", "0.14.0"): + from accelerate import cpu_offload + from accelerate import cpu_offload_with_hook + else: + raise ImportError("`enable_sequential_cpu_offload` requires `accelerate v0.14.0` or higher") + self.remove_all_hooks() + + is_pipeline_device_mapped = self.hf_device_map is not None and len(self.hf_device_map) > 1 + if is_pipeline_device_mapped: + raise ValueError( + "It seems like you have activated a device mapping strategy on the pipeline so calling `enable_sequential_cpu_offload() isn't allowed. You can call `reset_device_map()` first and then call `enable_sequential_cpu_offload()`." + ) + + torch_device = torch.device(device) + device_index = torch_device.index + + if gpu_id is not None and device_index is not None: + raise ValueError( + f"You have passed both `gpu_id`={gpu_id} and an index as part of the passed device `device`={device}" + f"Cannot pass both. Please make sure to either not define `gpu_id` or not pass the index as part of the device: `device`={torch_device.type}" + ) + + # _offload_gpu_id should be set to passed gpu_id (or id in passed `device`) or default to previously set id or default to 0 + self._offload_gpu_id = gpu_id or torch_device.index or getattr(self, "_offload_gpu_id", 0) + + device_type = torch_device.type + device = torch.device(f"{device_type}:{self._offload_gpu_id}") + self._offload_device = device + + if self.device.type != "cpu": + self.to("cpu", silence_dtype_warnings=True) + device_mod = getattr(torch, self.device.type, None) + if hasattr(device_mod, "empty_cache") and device_mod.is_available(): + device_mod.empty_cache() # otherwise we don't see the memory savings (but they probably exist) + + for name, model in self.components.items(): + if not isinstance(model, torch.nn.Module): + continue + + if name in self._manual_cpu_offload_in_sequential_cpu_offload: + pass + else: + # make sure to offload buffers if not all high level weights + # are of type nn.Module + offload_buffers = len(model._parameters) > 0 + cpu_offload(model, device, offload_buffers=offload_buffers) - def enable_sequential_cpu_offload(self, *args, **kwargs): - super().enable_sequential_cpu_offload(*args, **kwargs) if hasattr(self.transformer, "clip_projection") and self.transformer.clip_projection is not None: import accelerate accelerate.hooks.remove_hook_from_module(self.transformer.clip_projection, recurse=True) self.transformer.clip_projection = self.transformer.clip_projection.to("cuda") + self.manual_cpu_offload_flag = True + + def enable_model_cpu_offload(self, *args, **kwargs): + super().enable_model_cpu_offload(*args, **kwargs) + self.manual_cpu_offload_flag = True + def encode_prompt( self, prompt: str, @@ -855,6 +910,12 @@ class EasyAnimatePipeline(DiffusionPipeline): else: dtype = self.transformer.dtype + if self.manual_cpu_offload_flag: + if isinstance(self.text_encoder, Qwen2VLForConditionalGeneration): + self.text_encoder.to(device) + if isinstance(self.text_encoder_2, Qwen2VLForConditionalGeneration) and self.text_encoder_2 is not None: + self.text_encoder_2.to(device) + # 3. Encode input prompt ( prompt_embeds, @@ -899,6 +960,13 @@ class EasyAnimatePipeline(DiffusionPipeline): prompt_attention_mask_2 = None negative_prompt_attention_mask_2 = None + if self.manual_cpu_offload_flag: + if isinstance(self.text_encoder, Qwen2VLForConditionalGeneration): + self.text_encoder.to("cpu") + if isinstance(self.text_encoder_2, Qwen2VLForConditionalGeneration) and self.text_encoder_2 is not None: + self.text_encoder_2.to("cpu") + torch.cuda.empty_cache() + # 4. Prepare timesteps if isinstance(self.scheduler, FlowMatchEulerDiscreteScheduler): timesteps, num_inference_steps = retrieve_timesteps(self.scheduler, num_inference_steps, device, timesteps, mu=1) diff --git a/easyanimate/pipeline/pipeline_easyanimate_control.py b/easyanimate/pipeline/pipeline_easyanimate_control.py old mode 100644 new mode 100755 index f5ab704..7d2db71 --- a/easyanimate/pipeline/pipeline_easyanimate_control.py +++ b/easyanimate/pipeline/pipeline_easyanimate_control.py @@ -269,14 +269,69 @@ class EasyAnimateControlPipeline(DiffusionPipeline): self.mask_processor = VaeImageProcessor( vae_scale_factor=self.vae_scale_factor, do_normalize=False, do_binarize=True, do_convert_grayscale=True ) + self.manual_cpu_offload_flag = False + + def enable_sequential_cpu_offload(self, gpu_id: Optional[int] = None, device: Union[torch.device, str] = "cuda"): + from diffusers.pipelines.pipeline_utils import is_accelerate_available, is_accelerate_version + + if is_accelerate_available() and is_accelerate_version(">=", "0.14.0"): + from accelerate import cpu_offload + from accelerate import cpu_offload_with_hook + else: + raise ImportError("`enable_sequential_cpu_offload` requires `accelerate v0.14.0` or higher") + self.remove_all_hooks() + + is_pipeline_device_mapped = self.hf_device_map is not None and len(self.hf_device_map) > 1 + if is_pipeline_device_mapped: + raise ValueError( + "It seems like you have activated a device mapping strategy on the pipeline so calling `enable_sequential_cpu_offload() isn't allowed. You can call `reset_device_map()` first and then call `enable_sequential_cpu_offload()`." + ) + + torch_device = torch.device(device) + device_index = torch_device.index + + if gpu_id is not None and device_index is not None: + raise ValueError( + f"You have passed both `gpu_id`={gpu_id} and an index as part of the passed device `device`={device}" + f"Cannot pass both. Please make sure to either not define `gpu_id` or not pass the index as part of the device: `device`={torch_device.type}" + ) + + # _offload_gpu_id should be set to passed gpu_id (or id in passed `device`) or default to previously set id or default to 0 + self._offload_gpu_id = gpu_id or torch_device.index or getattr(self, "_offload_gpu_id", 0) + + device_type = torch_device.type + device = torch.device(f"{device_type}:{self._offload_gpu_id}") + self._offload_device = device + + if self.device.type != "cpu": + self.to("cpu", silence_dtype_warnings=True) + device_mod = getattr(torch, self.device.type, None) + if hasattr(device_mod, "empty_cache") and device_mod.is_available(): + device_mod.empty_cache() # otherwise we don't see the memory savings (but they probably exist) + + for name, model in self.components.items(): + if not isinstance(model, torch.nn.Module): + continue + + if name in self._manual_cpu_offload_in_sequential_cpu_offload: + pass + else: + # make sure to offload buffers if not all high level weights + # are of type nn.Module + offload_buffers = len(model._parameters) > 0 + cpu_offload(model, device, offload_buffers=offload_buffers) - def enable_sequential_cpu_offload(self, *args, **kwargs): - super().enable_sequential_cpu_offload(*args, **kwargs) if hasattr(self.transformer, "clip_projection") and self.transformer.clip_projection is not None: import accelerate accelerate.hooks.remove_hook_from_module(self.transformer.clip_projection, recurse=True) self.transformer.clip_projection = self.transformer.clip_projection.to("cuda") + self.manual_cpu_offload_flag = True + + def enable_model_cpu_offload(self, *args, **kwargs): + super().enable_model_cpu_offload(*args, **kwargs) + self.manual_cpu_offload_flag = True + def encode_prompt( self, prompt: str, @@ -922,6 +977,12 @@ class EasyAnimateControlPipeline(DiffusionPipeline): else: dtype = self.transformer.dtype + if self.manual_cpu_offload_flag: + if isinstance(self.text_encoder, Qwen2VLForConditionalGeneration): + self.text_encoder.to(device) + if isinstance(self.text_encoder_2, Qwen2VLForConditionalGeneration) and self.text_encoder_2 is not None: + self.text_encoder_2.to(device) + # 3. Encode input prompt ( prompt_embeds, @@ -966,6 +1027,13 @@ class EasyAnimateControlPipeline(DiffusionPipeline): prompt_attention_mask_2 = None negative_prompt_attention_mask_2 = None + if self.manual_cpu_offload_flag: + if isinstance(self.text_encoder, Qwen2VLForConditionalGeneration): + self.text_encoder.to("cpu") + if isinstance(self.text_encoder_2, Qwen2VLForConditionalGeneration) and self.text_encoder_2 is not None: + self.text_encoder_2.to("cpu") + torch.cuda.empty_cache() + # 4. Prepare timesteps if isinstance(self.scheduler, FlowMatchEulerDiscreteScheduler): timesteps, num_inference_steps = retrieve_timesteps(self.scheduler, num_inference_steps, device, timesteps, mu=1) diff --git a/easyanimate/pipeline/pipeline_easyanimate_inpaint.py b/easyanimate/pipeline/pipeline_easyanimate_inpaint.py old mode 100644 new mode 100755 index ffc5c45..2dff6cb --- a/easyanimate/pipeline/pipeline_easyanimate_inpaint.py +++ b/easyanimate/pipeline/pipeline_easyanimate_inpaint.py @@ -150,14 +150,18 @@ def resize_mask(mask, latent, process_first_frame_only=True): ## Add noise to reference video -def add_noise_to_reference_video(image, ratio=None): +def add_noise_to_reference_video(image, ratio=None, generator=None): if ratio is None: sigma = torch.normal(mean=-3.0, std=0.5, size=(image.shape[0],)).to(image.device) sigma = torch.exp(sigma).to(image.dtype) else: sigma = torch.ones((image.shape[0],)).to(image.device, image.dtype) * ratio - image_noise = torch.randn_like(image) * sigma[:, None, None, None, None] + if generator is not None: + image_noise = torch.randn(image.size(), generator=generator, dtype=image.dtype, device=image.device) * \ + sigma[:, None, None, None, None] + else: + image_noise = torch.randn_like(image) * sigma[:, None, None, None, None] image_noise = torch.where(image==-1, torch.zeros_like(image), image_noise) image = image + image_noise return image @@ -319,14 +323,69 @@ class EasyAnimateInpaintPipeline(DiffusionPipeline): self.mask_processor = VaeImageProcessor( vae_scale_factor=self.vae_scale_factor, do_normalize=False, do_binarize=True, do_convert_grayscale=True ) + self.manual_cpu_offload_flag = False + + def enable_sequential_cpu_offload(self, gpu_id: Optional[int] = None, device: Union[torch.device, str] = "cuda"): + from diffusers.pipelines.pipeline_utils import is_accelerate_available, is_accelerate_version + + if is_accelerate_available() and is_accelerate_version(">=", "0.14.0"): + from accelerate import cpu_offload + from accelerate import cpu_offload_with_hook + else: + raise ImportError("`enable_sequential_cpu_offload` requires `accelerate v0.14.0` or higher") + self.remove_all_hooks() + + is_pipeline_device_mapped = self.hf_device_map is not None and len(self.hf_device_map) > 1 + if is_pipeline_device_mapped: + raise ValueError( + "It seems like you have activated a device mapping strategy on the pipeline so calling `enable_sequential_cpu_offload() isn't allowed. You can call `reset_device_map()` first and then call `enable_sequential_cpu_offload()`." + ) + + torch_device = torch.device(device) + device_index = torch_device.index + + if gpu_id is not None and device_index is not None: + raise ValueError( + f"You have passed both `gpu_id`={gpu_id} and an index as part of the passed device `device`={device}" + f"Cannot pass both. Please make sure to either not define `gpu_id` or not pass the index as part of the device: `device`={torch_device.type}" + ) + + # _offload_gpu_id should be set to passed gpu_id (or id in passed `device`) or default to previously set id or default to 0 + self._offload_gpu_id = gpu_id or torch_device.index or getattr(self, "_offload_gpu_id", 0) + + device_type = torch_device.type + device = torch.device(f"{device_type}:{self._offload_gpu_id}") + self._offload_device = device + + if self.device.type != "cpu": + self.to("cpu", silence_dtype_warnings=True) + device_mod = getattr(torch, self.device.type, None) + if hasattr(device_mod, "empty_cache") and device_mod.is_available(): + device_mod.empty_cache() # otherwise we don't see the memory savings (but they probably exist) + + for name, model in self.components.items(): + if not isinstance(model, torch.nn.Module): + continue + + if name in self._manual_cpu_offload_in_sequential_cpu_offload: + pass + else: + # make sure to offload buffers if not all high level weights + # are of type nn.Module + offload_buffers = len(model._parameters) > 0 + cpu_offload(model, device, offload_buffers=offload_buffers) - def enable_sequential_cpu_offload(self, *args, **kwargs): - super().enable_sequential_cpu_offload(*args, **kwargs) if hasattr(self.transformer, "clip_projection") and self.transformer.clip_projection is not None: import accelerate accelerate.hooks.remove_hook_from_module(self.transformer.clip_projection, recurse=True) self.transformer.clip_projection = self.transformer.clip_projection.to("cuda") + self.manual_cpu_offload_flag = True + + def enable_model_cpu_offload(self, *args, **kwargs): + super().enable_model_cpu_offload(*args, **kwargs) + self.manual_cpu_offload_flag = True + def encode_prompt( self, prompt: str, @@ -738,7 +797,7 @@ class EasyAnimateInpaintPipeline(DiffusionPipeline): if masked_image is not None: masked_image = masked_image.to(device=device, dtype=dtype) if self.transformer.config.add_noise_in_inpaint_model: - masked_image = add_noise_to_reference_video(masked_image, ratio=noise_aug_strength) + masked_image = add_noise_to_reference_video(masked_image, ratio=noise_aug_strength, generator=generator) if self.vae.quant_conv is None or self.vae.quant_conv.weight.ndim==5: bs = 1 new_mask_pixel_values = [] @@ -809,7 +868,7 @@ class EasyAnimateInpaintPipeline(DiffusionPipeline): for i in range(0, video.shape[0], bs): video_bs = video[i : i + bs] video_bs = self.vae.encode(video_bs)[0] - video_bs = video_bs.sample() + video_bs = video_bs.mode() new_video.append(video_bs) video = torch.cat(new_video, dim = 0) video = video * self.vae.config.scaling_factor @@ -1098,7 +1157,13 @@ class EasyAnimateInpaintPipeline(DiffusionPipeline): dtype = self.text_encoder_2.dtype else: dtype = self.transformer.dtype - + + if self.manual_cpu_offload_flag: + if isinstance(self.text_encoder, Qwen2VLForConditionalGeneration): + self.text_encoder.to(device) + if isinstance(self.text_encoder_2, Qwen2VLForConditionalGeneration) and self.text_encoder_2 is not None: + self.text_encoder_2.to(device) + # 3. Encode input prompt ( prompt_embeds, @@ -1143,6 +1208,13 @@ class EasyAnimateInpaintPipeline(DiffusionPipeline): prompt_attention_mask_2 = None negative_prompt_attention_mask_2 = None + if self.manual_cpu_offload_flag: + if isinstance(self.text_encoder, Qwen2VLForConditionalGeneration): + self.text_encoder.to("cpu") + if isinstance(self.text_encoder_2, Qwen2VLForConditionalGeneration) and self.text_encoder_2 is not None: + self.text_encoder_2.to("cpu") + torch.cuda.empty_cache() + # 4. set timesteps if isinstance(self.scheduler, FlowMatchEulerDiscreteScheduler): timesteps, num_inference_steps = retrieve_timesteps(self.scheduler, num_inference_steps, device, timesteps, mu=1) diff --git a/easyanimate/ui/ui.py b/easyanimate/ui/ui.py index d1e2f0c..6b7a27c 100755 --- a/easyanimate/ui/ui.py +++ b/easyanimate/ui/ui.py @@ -28,19 +28,18 @@ from transformers import (BertModel, BertTokenizer, CLIPImageProcessor, T5Tokenizer) from ..data.bucket_sampler import ASPECT_RATIO_512, get_closest_ratio -from ..models import (name_to_autoencoder_magvit, - name_to_transformer3d) -from ..pipeline.pipeline_easyanimate import \ - EasyAnimatePipeline -from ..pipeline.pipeline_easyanimate_control import \ - EasyAnimateControlPipeline -from ..pipeline.pipeline_easyanimate_inpaint import \ - EasyAnimateInpaintPipeline -from ..utils.fp8_optimization import convert_weight_dtype_wrapper +from ..models import name_to_autoencoder_magvit, name_to_transformer3d +from ..models.transformer3d import get_teacache_coefficients +from ..pipeline.pipeline_easyanimate import EasyAnimatePipeline +from ..pipeline.pipeline_easyanimate_control import EasyAnimateControlPipeline +from ..pipeline.pipeline_easyanimate_inpaint import EasyAnimateInpaintPipeline +from ..utils.fp8_optimization import (convert_model_weight_to_float8, + convert_weight_dtype_wrapper) from ..utils.lora_utils import merge_lora, unmerge_lora -from ..utils.utils import ( - get_image_to_video_latent, get_video_to_video_latent, - get_width_and_height_from_image_and_base_resolution, save_videos_grid) +from ..utils.utils import (get_image_to_video_latent, + get_video_to_video_latent, + get_width_and_height_from_image_and_base_resolution, + save_videos_grid) ddpm_scheduler_dict = { "Euler": EulerDiscreteScheduler, @@ -169,11 +168,11 @@ class EasyAnimateController: diffusion_transformer_dropdown, subfolder="vae", ).to(self.weight_dtype) - if self.inference_config['vae_kwargs'].get('vae_type', 'AutoencoderKL') == 'AutoencoderKLMagvit' and self.weight_dtype == torch.float16: + if self.weight_dtype == torch.float16 and "v5.1" not in diffusion_transformer_dropdown.lower(): self.vae.upcast_vae = True transformer_additional_kwargs = OmegaConf.to_container(self.inference_config['transformer_additional_kwargs']) - if self.weight_dtype == torch.float16: + if self.weight_dtype == torch.float16 and "v5.1" not in diffusion_transformer_dropdown.lower(): transformer_additional_kwargs["upcast_attention"] = True # Get Transformer @@ -294,8 +293,19 @@ class EasyAnimateController: ) if self.GPU_memory_mode == "sequential_cpu_offload": + self.pipeline._manual_cpu_offload_in_sequential_cpu_offload = [] + for name, _text_encoder in zip(["text_encoder", "text_encoder_2"], [self.pipeline.text_encoder, self.pipeline.text_encoder_2]): + if isinstance(_text_encoder, Qwen2VLForConditionalGeneration): + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_model_weight_to_float8(_text_encoder) + convert_weight_dtype_wrapper(_text_encoder, self.weight_dtype) + self.pipeline._manual_cpu_offload_in_sequential_cpu_offload = [name] self.pipeline.enable_sequential_cpu_offload() elif self.GPU_memory_mode == "model_cpu_offload_and_qfloat8": + for _text_encoder in [self.pipeline.text_encoder, self.pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual self.pipeline.enable_model_cpu_offload() convert_weight_dtype_wrapper(self.pipeline.transformer, self.weight_dtype) else: @@ -464,8 +474,10 @@ class EasyAnimateController: # lora part self.pipeline = merge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider) - if self.edition == "v5.1" and self.enable_teacache: - self.pipeline.transformer.enable_teacache(sample_step_slider, self.teacache_threshold) + coefficients = get_teacache_coefficients(self.base_model_path) + if coefficients is not None and self.enable_teacache: + print(f"Enable TeaCache with threshold: {self.teacache_threshold}.") + self.pipeline.transformer.enable_teacache(sample_step_slider, self.teacache_threshold, coefficients=coefficients) try: if self.model_type == "Inpaint": @@ -1017,6 +1029,7 @@ class EasyAnimateController_Modelscope: # Config and model path self.model_type = model_type self.edition = edition + self.model_name = model_name self.enable_teacache = enable_teacache self.teacache_threshold = teacache_threshold self.weight_dtype = weight_dtype @@ -1028,11 +1041,11 @@ class EasyAnimateController_Modelscope: model_name, subfolder="vae", ).to(self.weight_dtype) - if self.inference_config['vae_kwargs'].get('vae_type', 'AutoencoderKL') == 'AutoencoderKLMagvit' and weight_dtype == torch.float16: + if self.weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): self.vae.upcast_vae = True transformer_additional_kwargs = OmegaConf.to_container(self.inference_config['transformer_additional_kwargs']) - if self.weight_dtype == torch.float16: + if self.weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): transformer_additional_kwargs["upcast_attention"] = True # Get Transformer @@ -1151,12 +1164,23 @@ class EasyAnimateController_Modelscope: ) if GPU_memory_mode == "sequential_cpu_offload": + self.pipeline._manual_cpu_offload_in_sequential_cpu_offload = [] + for name, _text_encoder in zip(["text_encoder", "text_encoder_2"], [self.pipeline.text_encoder, self.pipeline.text_encoder_2]): + if isinstance(_text_encoder, Qwen2VLForConditionalGeneration): + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_model_weight_to_float8(_text_encoder) + convert_weight_dtype_wrapper(_text_encoder, weight_dtype) + self.pipeline._manual_cpu_offload_in_sequential_cpu_offload = [name] self.pipeline.enable_sequential_cpu_offload() elif GPU_memory_mode == "model_cpu_offload_and_qfloat8": - self.pipeline.enable_model_cpu_offload() + for _text_encoder in [self.pipeline.text_encoder, self.pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual convert_weight_dtype_wrapper(self.pipeline.transformer, weight_dtype) + self.pipeline.enable_model_cpu_offload() else: - GPU_memory_mode.enable_model_cpu_offload() + self.pipeline.enable_model_cpu_offload() print("Update diffusion transformer done") def refresh_personalized_model(self): @@ -1266,9 +1290,10 @@ class EasyAnimateController_Modelscope: # lora part self.pipeline = merge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider) - if self.edition == "v5.1" and self.enable_teacache: + coefficients = get_teacache_coefficients(self.model_name) + if coefficients is not None and self.enable_teacache: print(f"Enable TeaCache with threshold: {self.teacache_threshold}.") - self.pipeline.transformer.enable_teacache(sample_step_slider, self.teacache_threshold) + self.pipeline.transformer.enable_teacache(sample_step_slider, self.teacache_threshold, coefficients=coefficients) try: if self.model_type == "Inpaint": diff --git a/easyanimate/utils/fp8_optimization.py b/easyanimate/utils/fp8_optimization.py old mode 100644 new mode 100755 index d270605..e8a590e --- a/easyanimate/utils/fp8_optimization.py +++ b/easyanimate/utils/fp8_optimization.py @@ -14,9 +14,16 @@ def autocast_model_forward(cls, origin_dtype, *inputs, **kwargs): cls.to(weight_dtype) return out +def convert_model_weight_to_float8(model, exclude_module_name='embed_tokens'): + for name, module in model.named_modules(): + if exclude_module_name not in name: + for param_name, param in module.named_parameters(): + if exclude_module_name not in param_name: + param.data = param.data.to(torch.float8_e4m3fn) + def convert_weight_dtype_wrapper(module, origin_dtype): for name, module in module.named_modules(): - if name == "": + if name == "" or "embed_tokens" in name: continue original_forward = module.forward if hasattr(module, "weight"): diff --git a/easyanimate/vae/ldm/data/dataset_image_video.py b/easyanimate/vae/ldm/data/dataset_image_video.py index 3eb85fb..4d99e45 100644 --- a/easyanimate/vae/ldm/data/dataset_image_video.py +++ b/easyanimate/vae/ldm/data/dataset_image_video.py @@ -172,7 +172,7 @@ class ImageVideoDataset(Dataset): video_reader = VideoReader(example['file_path']) video_length = len(video_reader) if self.slice_interval == "rand": - slice_interval = np.random.choice([1, 2, 3]) + slice_interval = np.random.choice([1, 2, 3, 4, 5, 6, 7, 8]) else: slice_interval = int(self.slice_interval) clip_length = min(video_length, (self.video_len - 1) * slice_interval + 1) diff --git a/easyanimate/video_caption/compute_semantic_consistency.py b/easyanimate/video_caption/compute_semantic_consistency.py index 4599cfe..993bdb1 100644 --- a/easyanimate/video_caption/compute_semantic_consistency.py +++ b/easyanimate/video_caption/compute_semantic_consistency.py @@ -162,6 +162,7 @@ def main(): video_dataset = VideoDataset( dataset_inputs={args.video_path_column: splitted_video_path_list}, video_folder=args.video_folder, + video_path_column=args.video_path_column, sample_method=args.frame_sample_method, num_sampled_frames=args.num_sampled_frames, sample_stride=args.sample_stride, diff --git a/easyanimate/video_caption/compute_video_quality.py b/easyanimate/video_caption/compute_video_quality.py index c071722..313c84c 100644 --- a/easyanimate/video_caption/compute_video_quality.py +++ b/easyanimate/video_caption/compute_video_quality.py @@ -89,10 +89,10 @@ def main(): saved_metadata_df = pd.read_json(args.saved_path, lines=True) # Filter out the unprocessed video-caption pairs by setting the indicator=True. - merged_df = video_metadata_df.merge(saved_metadata_df, on="video_path", how="outer", indicator=True) + merged_df = video_metadata_df.merge(saved_metadata_df, on=args.video_path_column, how="outer", indicator=True) video_metadata_df = merged_df[merged_df["_merge"] == "left_only"] # Sorting to guarantee the same result for each process. - video_metadata_df = video_metadata_df.iloc[index_natsorted(video_metadata_df["video_path"])].reset_index(drop=True) + video_metadata_df = video_metadata_df.iloc[index_natsorted(video_metadata_df[args.video_path_column])].reset_index(drop=True) if args.caption_column is None: video_metadata_df = video_metadata_df[[args.video_path_column]] else: @@ -160,6 +160,7 @@ def main(): video_dataset = VideoDataset( dataset_inputs=splitted_video_metadata, video_folder=args.video_folder, + video_path_column=args.video_path_column, text_column=args.caption_column, sample_method=args.frame_sample_method, num_sampled_frames=args.num_sampled_frames diff --git a/easyanimate/video_caption/filter_meta_train.py b/easyanimate/video_caption/filter_meta_train.py index 73175b3..5d22bf9 100644 --- a/easyanimate/video_caption/filter_meta_train.py +++ b/easyanimate/video_caption/filter_meta_train.py @@ -18,6 +18,12 @@ def parse_args(): default="video_path", help="The column contains the video path (an absolute path or a relative path w.r.t the video_folder).", ) + parser.add_argument( + "--caption_column", + type=str, + default="caption", + help="The column contains the caption.", + ) parser.add_argument("--video_folder", type=str, default="", help="The video folder.") parser.add_argument( "--basic_metadata_path", type=str, default=None, help="The path to the basic metadata (csv/jsonl)." @@ -76,7 +82,7 @@ def main(): ) filtered_video_path_list = natsorted(filtered_video_path_list) filtered_caption_df = raw_caption_df[raw_caption_df[args.video_path_column].isin(filtered_video_path_list)] - train_df = filtered_caption_df.rename(columns={"video_path": "file_path", "caption": "text"}) + train_df = filtered_caption_df.rename(columns={args.video_path_column: "file_path", args.caption_column: "text"}) train_df["file_path"] = train_df["file_path"].map(lambda x: os.path.join(args.video_folder, x)) train_df["type"] = "video" train_df.to_json(args.saved_path, orient="records", force_ascii=False, indent=2) diff --git a/predict_i2v.py b/predict_i2v.py old mode 100644 new mode 100755 index 654467f..8a40b01 --- a/predict_i2v.py +++ b/predict_i2v.py @@ -14,9 +14,11 @@ from transformers import (BertModel, BertTokenizer, CLIPImageProcessor, from easyanimate.models import (name_to_autoencoder_magvit, name_to_transformer3d) +from easyanimate.models.transformer3d import get_teacache_coefficients from easyanimate.pipeline.pipeline_easyanimate_inpaint import \ EasyAnimateInpaintPipeline -from easyanimate.utils.fp8_optimization import convert_weight_dtype_wrapper +from easyanimate.utils.fp8_optimization import (convert_model_weight_to_float8, + convert_weight_dtype_wrapper) from easyanimate.utils.lora_utils import merge_lora, unmerge_lora from easyanimate.utils.utils import get_image_to_video_latent, save_videos_grid @@ -30,14 +32,13 @@ from easyanimate.utils.utils import get_image_to_video_latent, save_videos_grid # resulting in slower speeds but saving a large amount of GPU memory. # # EasyAnimateV1, V2 and V3 support "model_cpu_offload" "sequential_cpu_offload" -# EasyAnimateV4, V5 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" -# EasyAnimateV5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" +# EasyAnimateV4, V5 and V5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" GPU_memory_mode = "model_cpu_offload_and_qfloat8" # EasyAnimateV5.1 support TeaCache. enable_teacache = True # Recommended to be set between 0.05 and 0.1. A larger threshold can cache more steps, speeding up the inference process, # but it may cause slight differences between the generated content and the original content. -teacache_threshold = 0.1 +teacache_threshold = 0.08 # Config and model path config_path = "config/easyanimate_video_v5.1_magvit_qwen.yaml" @@ -80,7 +81,7 @@ validation_image_end = None # EasyAnimateV4, V5 and V5.1 support English and Chinese. # 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性 # 在neg prompt中添加"安静,固定"等词语可以增加动态性。 -prompt = "一只棕褐色的狗在摇晃脑袋,坐在一个舒适的房间里的浅色沙发上。在狗的后面,架子上有一幅镶框的画,周围是粉红色的花朵。房间里的灯光柔和温暖,营造出舒适的氛围。" +prompt = "一只棕褐色的狗正摇晃着脑袋,坐在一个舒适的房间里的浅色沙发上。沙发看起来柔软而宽敞,为这只活泼的狗狗提供了一个完美的休息地点。在狗的后面,靠墙摆放着一个架子,架子上挂着一幅精美的镶框画,画中描绘着一些美丽的风景或场景。画框周围装饰着粉红色的花朵,这些花朵不仅增添了房间的色彩,还带来了一丝自然和生机。房间里的灯光柔和而温暖,从天花板上的吊灯和角落里的台灯散发出来,营造出一种温馨舒适的氛围。整个空间给人一种宁静和谐的感觉,仿佛时间在这里变得缓慢而美好。" negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。" # # Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability @@ -101,7 +102,7 @@ Choosen_Transformer3DModel = name_to_transformer3d[ ] transformer_additional_kwargs = OmegaConf.to_container(config['transformer_additional_kwargs']) -if weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): transformer_additional_kwargs["upcast_attention"] = True transformer = Choosen_Transformer3DModel.from_pretrained_2d( @@ -145,7 +146,7 @@ vae = Choosen_AutoencoderKL.from_pretrained( subfolder="vae", vae_additional_kwargs=OmegaConf.to_container(config['vae_kwargs']) ).to(weight_dtype) -if config['vae_kwargs'].get('vae_type', 'AutoencoderKL') == 'AutoencoderKLMagvit' and weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): vae.upcast_vae = True if vae_path is not None: @@ -246,16 +247,28 @@ pipeline = EasyAnimateInpaintPipeline( ) if GPU_memory_mode == "sequential_cpu_offload": + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [] + for name, _text_encoder in zip(["text_encoder", "text_encoder_2"], [pipeline.text_encoder, pipeline.text_encoder_2]): + if isinstance(_text_encoder, Qwen2VLForConditionalGeneration): + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_model_weight_to_float8(_text_encoder) + convert_weight_dtype_wrapper(_text_encoder, weight_dtype) + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [name] pipeline.enable_sequential_cpu_offload() elif GPU_memory_mode == "model_cpu_offload_and_qfloat8": - pipeline.enable_model_cpu_offload() + for _text_encoder in [pipeline.text_encoder, pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual convert_weight_dtype_wrapper(transformer, weight_dtype) + pipeline.enable_model_cpu_offload() else: pipeline.enable_model_cpu_offload() -if "v5.1" in config_path and enable_teacache: +coefficients = get_teacache_coefficients(model_name) +if coefficients is not None and enable_teacache: print(f"Enable TeaCache with threshold: {teacache_threshold}.") - pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold) + pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold, coefficients=coefficients) generator = torch.Generator(device="cuda").manual_seed(seed) diff --git a/predict_t2v.py b/predict_t2v.py old mode 100644 new mode 100755 index 42f5b58..8cecb7d --- a/predict_t2v.py +++ b/predict_t2v.py @@ -7,18 +7,19 @@ from diffusers import (DDIMScheduler, DPMSolverMultistepScheduler, FlowMatchEulerDiscreteScheduler, PNDMScheduler) from omegaconf import OmegaConf from PIL import Image -from transformers import (BertModel, BertTokenizer, - CLIPImageProcessor, CLIPVisionModelWithProjection, - Qwen2Tokenizer, Qwen2VLForConditionalGeneration, - T5EncoderModel, T5Tokenizer) +from transformers import (BertModel, BertTokenizer, CLIPImageProcessor, + CLIPVisionModelWithProjection, Qwen2Tokenizer, + Qwen2VLForConditionalGeneration, T5EncoderModel, + T5Tokenizer) from easyanimate.models import (name_to_autoencoder_magvit, name_to_transformer3d) -from easyanimate.pipeline.pipeline_easyanimate import \ - EasyAnimatePipeline +from easyanimate.models.transformer3d import get_teacache_coefficients +from easyanimate.pipeline.pipeline_easyanimate import EasyAnimatePipeline from easyanimate.pipeline.pipeline_easyanimate_inpaint import \ EasyAnimateInpaintPipeline -from easyanimate.utils.fp8_optimization import convert_weight_dtype_wrapper +from easyanimate.utils.fp8_optimization import (convert_model_weight_to_float8, + convert_weight_dtype_wrapper) from easyanimate.utils.lora_utils import merge_lora, unmerge_lora from easyanimate.utils.utils import get_image_to_video_latent, save_videos_grid @@ -32,14 +33,13 @@ from easyanimate.utils.utils import get_image_to_video_latent, save_videos_grid # resulting in slower speeds but saving a large amount of GPU memory. # # EasyAnimateV1, V2 and V3 support "model_cpu_offload" "sequential_cpu_offload" -# EasyAnimateV4, V5 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" -# EasyAnimateV5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" +# EasyAnimateV4, V5 and V5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" GPU_memory_mode = "model_cpu_offload_and_qfloat8" # EasyAnimateV5.1 support TeaCache. enable_teacache = True # Recommended to be set between 0.05 and 0.1. A larger threshold can cache more steps, speeding up the inference process, # but it may cause slight differences between the generated content and the original content. -teacache_threshold = 0.1 +teacache_threshold = 0.08 # Config and model path config_path = "config/easyanimate_video_v5.1_magvit_qwen.yaml" @@ -75,7 +75,7 @@ weight_dtype = torch.bfloat16 # EasyAnimateV4, V5 and V5.1 support English and Chinese. # 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性 # 在neg prompt中添加"安静,固定"等词语可以增加动态性。 -prompt = "一只棕褐色的狗在摇晃脑袋,坐在一个舒适的房间里的浅色沙发上。在狗的后面,架子上有一幅镶框的画,周围是粉红色的花朵。房间里的灯光柔和温暖,营造出舒适的氛围。" +prompt = "一只棕褐色的狗正摇晃着脑袋,坐在一个舒适的房间里的浅色沙发上。沙发看起来柔软而宽敞,为这只活泼的狗狗提供了一个完美的休息地点。在狗的后面,靠墙摆放着一个架子,架子上挂着一幅精美的镶框画,画中描绘着一些美丽的风景或场景。画框周围装饰着粉红色的花朵,这些花朵不仅增添了房间的色彩,还带来了一丝自然和生机。房间里的灯光柔和而温暖,从天花板上的吊灯和角落里的台灯散发出来,营造出一种温馨舒适的氛围。整个空间给人一种宁静和谐的感觉,仿佛时间在这里变得缓慢而美好。" negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。" # # Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability @@ -96,7 +96,7 @@ Choosen_Transformer3DModel = name_to_transformer3d[ ] transformer_additional_kwargs = OmegaConf.to_container(config['transformer_additional_kwargs']) -if weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): transformer_additional_kwargs["upcast_attention"] = True transformer = Choosen_Transformer3DModel.from_pretrained_2d( @@ -140,7 +140,7 @@ vae = Choosen_AutoencoderKL.from_pretrained( subfolder="vae", vae_additional_kwargs=OmegaConf.to_container(config['vae_kwargs']) ).to(weight_dtype) -if config['vae_kwargs'].get('vae_type', 'AutoencoderKL') == 'AutoencoderKLMagvit' and weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): vae.upcast_vae = True if vae_path is not None: @@ -254,16 +254,28 @@ else: ) if GPU_memory_mode == "sequential_cpu_offload": + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [] + for name, _text_encoder in zip(["text_encoder", "text_encoder_2"], [pipeline.text_encoder, pipeline.text_encoder_2]): + if isinstance(_text_encoder, Qwen2VLForConditionalGeneration): + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_model_weight_to_float8(_text_encoder) + convert_weight_dtype_wrapper(_text_encoder, weight_dtype) + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [name] pipeline.enable_sequential_cpu_offload() elif GPU_memory_mode == "model_cpu_offload_and_qfloat8": + for _text_encoder in [pipeline.text_encoder, pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_weight_dtype_wrapper(transformer, weight_dtype) pipeline.enable_model_cpu_offload() - convert_weight_dtype_wrapper(pipeline.transformer, weight_dtype) else: pipeline.enable_model_cpu_offload() -if "v5.1" in config_path and enable_teacache: +coefficients = get_teacache_coefficients(model_name) +if coefficients is not None and enable_teacache: print(f"Enable TeaCache with threshold: {teacache_threshold}.") - pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold) + pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold, coefficients=coefficients) generator = torch.Generator(device="cuda").manual_seed(seed) diff --git a/predict_v2v.py b/predict_v2v.py old mode 100644 new mode 100755 index 44aa569..7474f9b --- a/predict_v2v.py +++ b/predict_v2v.py @@ -14,12 +14,13 @@ from transformers import (BertModel, BertTokenizer, CLIPImageProcessor, from easyanimate.models import (name_to_autoencoder_magvit, name_to_transformer3d) +from easyanimate.models.transformer3d import get_teacache_coefficients from easyanimate.pipeline.pipeline_easyanimate_inpaint import \ EasyAnimateInpaintPipeline -from easyanimate.utils.fp8_optimization import convert_weight_dtype_wrapper +from easyanimate.utils.fp8_optimization import (convert_model_weight_to_float8, + convert_weight_dtype_wrapper) from easyanimate.utils.lora_utils import merge_lora, unmerge_lora -from easyanimate.utils.utils import (get_video_to_video_latent, - save_videos_grid) +from easyanimate.utils.utils import get_video_to_video_latent, save_videos_grid # GPU memory mode, which can be choosen in [model_cpu_offload, model_cpu_offload_and_qfloat8, sequential_cpu_offload]. # model_cpu_offload means that the entire model will be moved to the CPU after use, which can save some GPU memory. @@ -31,14 +32,13 @@ from easyanimate.utils.utils import (get_video_to_video_latent, # resulting in slower speeds but saving a large amount of GPU memory. # # EasyAnimateV3 support "model_cpu_offload" "sequential_cpu_offload" -# EasyAnimateV4, V5 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" -# EasyAnimateV5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" +# EasyAnimateV4, V5 and V5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" GPU_memory_mode = "model_cpu_offload_and_qfloat8" # EasyAnimateV5.1 support TeaCache. enable_teacache = True # Recommended to be set between 0.05 and 0.1. A larger threshold can cache more steps, speeding up the inference process, # but it may cause slight differences between the generated content and the original content. -teacache_threshold = 0.1 +teacache_threshold = 0.08 # Config and model path config_path = "config/easyanimate_video_v5.1_magvit_qwen.yaml" @@ -74,7 +74,7 @@ denoise_strength = 0.70 # 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性 # 在neg prompt中添加"安静,固定"等词语可以增加动态性。 -prompt = "一只穿着小外套的猫咪正在花园秋千上安静地弹吉他。晚霞的余光洒在它柔软的毛皮上,和煦的微风轻轻拂过,周围斑驳的光影随着音乐的旋律轻轻摇曳。" +prompt = "一只穿着小外套的猫咪正安静地坐在花园的秋千上弹吉他。它的小外套精致而合身,增添了几分俏皮与可爱。晚霞的余光洒在它柔软的毛皮上,给它的毛发镀上了一层温暖的金色光辉。和煦的微风轻轻拂过,带来阵阵花香和草木的气息,令人心旷神怡。周围斑驳的光影随着音乐的旋律轻轻摇曳,仿佛整个花园都在为这只小猫咪的演奏伴舞。阳光透过树叶间的缝隙,投下一片片光影交错的图案,与悠扬的吉他声交织在一起,营造出一种梦幻而宁静的氛围。猫咪专注而投入地弹奏着,每一个音符都似乎充满了魔力,让这个傍晚变得更加美好。" negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。" # # Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability @@ -95,7 +95,7 @@ Choosen_Transformer3DModel = name_to_transformer3d[ ] transformer_additional_kwargs = OmegaConf.to_container(config['transformer_additional_kwargs']) -if weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): transformer_additional_kwargs["upcast_attention"] = True transformer = Choosen_Transformer3DModel.from_pretrained_2d( @@ -139,7 +139,7 @@ vae = Choosen_AutoencoderKL.from_pretrained( subfolder="vae", vae_additional_kwargs=OmegaConf.to_container(config['vae_kwargs']) ).to(weight_dtype) -if config['vae_kwargs'].get('vae_type', 'AutoencoderKL') == 'AutoencoderKLMagvit' and weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): vae.upcast_vae = True if vae_path is not None: @@ -241,16 +241,28 @@ pipeline = EasyAnimateInpaintPipeline( ) if GPU_memory_mode == "sequential_cpu_offload": + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [] + for name, _text_encoder in zip(["text_encoder", "text_encoder_2"], [pipeline.text_encoder, pipeline.text_encoder_2]): + if isinstance(_text_encoder, Qwen2VLForConditionalGeneration): + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_model_weight_to_float8(_text_encoder) + convert_weight_dtype_wrapper(_text_encoder, weight_dtype) + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [name] pipeline.enable_sequential_cpu_offload() elif GPU_memory_mode == "model_cpu_offload_and_qfloat8": + for _text_encoder in [pipeline.text_encoder, pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_weight_dtype_wrapper(transformer, weight_dtype) pipeline.enable_model_cpu_offload() - convert_weight_dtype_wrapper(pipeline.transformer, weight_dtype) else: pipeline.enable_model_cpu_offload() -if "v5.1" in config_path and enable_teacache: +coefficients = get_teacache_coefficients(model_name) +if coefficients is not None and enable_teacache: print(f"Enable TeaCache with threshold: {teacache_threshold}.") - pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold) + pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold, coefficients=coefficients) generator = torch.Generator(device="cuda").manual_seed(seed) diff --git a/predict_v2v_control.py b/predict_v2v_control.py old mode 100644 new mode 100755 index d9640e4..9597e6a --- a/predict_v2v_control.py +++ b/predict_v2v_control.py @@ -4,23 +4,26 @@ import numpy as np import torch from diffusers import (DDIMScheduler, DPMSolverMultistepScheduler, EulerAncestralDiscreteScheduler, EulerDiscreteScheduler, - PNDMScheduler) + FlowMatchEulerDiscreteScheduler, PNDMScheduler) from omegaconf import OmegaConf from PIL import Image -from transformers import (BertModel, BertTokenizer, - CLIPImageProcessor, CLIPVisionModelWithProjection, - Qwen2Tokenizer, Qwen2VLForConditionalGeneration, - T5EncoderModel, T5Tokenizer) +from transformers import (BertModel, BertTokenizer, CLIPImageProcessor, + CLIPVisionModelWithProjection, Qwen2Tokenizer, + Qwen2VLForConditionalGeneration, T5EncoderModel, + T5Tokenizer) from easyanimate.data.dataset_image_video import process_pose_file from easyanimate.models import (name_to_autoencoder_magvit, name_to_transformer3d) +from easyanimate.models.transformer3d import get_teacache_coefficients from easyanimate.pipeline.pipeline_easyanimate_control import \ EasyAnimateControlPipeline +from easyanimate.utils.fp8_optimization import (convert_model_weight_to_float8, + convert_weight_dtype_wrapper) from easyanimate.utils.lora_utils import merge_lora, unmerge_lora -from easyanimate.utils.utils import get_video_to_video_latent, save_videos_grid, get_image_latent -from easyanimate.utils.fp8_optimization import convert_weight_dtype_wrapper -from diffusers import FlowMatchEulerDiscreteScheduler +from easyanimate.utils.utils import (get_image_latent, + get_video_to_video_latent, + save_videos_grid) # GPU memory mode, which can be choosen in [model_cpu_offload, model_cpu_offload_and_qfloat8, sequential_cpu_offload]. # model_cpu_offload means that the entire model will be moved to the CPU after use, which can save some GPU memory. @@ -31,14 +34,13 @@ from diffusers import FlowMatchEulerDiscreteScheduler # sequential_cpu_offload means that each layer of the model will be moved to the CPU after use, # resulting in slower speeds but saving a large amount of GPU memory. # -# EasyAnimateV5 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" -# EasyAnimateV5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" +# EasyAnimateV5 and V5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload" GPU_memory_mode = "model_cpu_offload_and_qfloat8" # EasyAnimateV5.1 support TeaCache. enable_teacache = True # Recommended to be set between 0.05 and 0.1. A larger threshold can cache more steps, speeding up the inference process, # but it may cause slight differences between the generated content and the original content. -teacache_threshold = 0.1 +teacache_threshold = 0.08 # Config and model path config_path = "config/easyanimate_video_v5.1_magvit_qwen.yaml" @@ -72,7 +74,7 @@ ref_image = None # 使用更长的neg prompt如"模糊,突变,变形,失真,画面暗,文本字幕,画面固定,连环画,漫画,线稿,没有主体。",可以增加稳定性 # 在neg prompt中添加"安静,固定"等词语可以增加动态性。 -prompt = "一位穿着合身的白色连衣裙,带着细肩带的女人站在一个铺着木地板的房间里。她有一头深色的长发。背景是一个放着各种瓶子的架子。灯光温暖,背景似乎在室内。" +prompt = "在这个阳光明媚的户外花园里,美女身穿一袭及膝的白色无袖连衣裙,裙摆在她轻盈的舞姿中轻柔地摆动,宛如一只翩翩起舞的蝴蝶。阳光透过树叶间洒下斑驳的光影,映衬出她柔和的脸庞和清澈的眼眸,显得格外优雅。仿佛每一个动作都在诉说着青春与活力,她在草地上旋转,裙摆随之飞扬,仿佛整个花园都因她的舞动而欢愉。周围五彩缤纷的花朵在微风中摇曳,玫瑰、菊花、百合,各自释放出阵阵香气,营造出一种轻松而愉快的氛围。" negative_prompt = "扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。" # # Using longer neg prompt such as "Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art." can increase stability @@ -93,7 +95,7 @@ Choosen_Transformer3DModel = name_to_transformer3d[ ] transformer_additional_kwargs = OmegaConf.to_container(config['transformer_additional_kwargs']) -if weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): transformer_additional_kwargs["upcast_attention"] = True transformer = Choosen_Transformer3DModel.from_pretrained_2d( @@ -137,7 +139,7 @@ vae = Choosen_AutoencoderKL.from_pretrained( subfolder="vae", vae_additional_kwargs=OmegaConf.to_container(config['vae_kwargs']) ).to(weight_dtype) -if config['vae_kwargs'].get('vae_type', 'AutoencoderKL') == 'AutoencoderKLMagvit' and weight_dtype == torch.float16: +if weight_dtype == torch.float16 and "v5.1" not in model_name.lower(): vae.upcast_vae = True if vae_path is not None: @@ -226,16 +228,28 @@ pipeline = EasyAnimateControlPipeline( ) if GPU_memory_mode == "sequential_cpu_offload": + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [] + for name, _text_encoder in zip(["text_encoder", "text_encoder_2"], [pipeline.text_encoder, pipeline.text_encoder_2]): + if isinstance(_text_encoder, Qwen2VLForConditionalGeneration): + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_model_weight_to_float8(_text_encoder) + convert_weight_dtype_wrapper(_text_encoder, weight_dtype) + pipeline._manual_cpu_offload_in_sequential_cpu_offload = [name] pipeline.enable_sequential_cpu_offload() elif GPU_memory_mode == "model_cpu_offload_and_qfloat8": + for _text_encoder in [pipeline.text_encoder, pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual + convert_weight_dtype_wrapper(transformer, weight_dtype) pipeline.enable_model_cpu_offload() - convert_weight_dtype_wrapper(pipeline.transformer, weight_dtype) else: pipeline.enable_model_cpu_offload() -if "v5.1" in config_path and enable_teacache: +coefficients = get_teacache_coefficients(model_name) +if coefficients is not None and enable_teacache: print(f"Enable TeaCache with threshold: {teacache_threshold}.") - pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold) + pipeline.transformer.enable_teacache(num_inference_steps, teacache_threshold, coefficients=coefficients) generator = torch.Generator(device="cuda").manual_seed(seed) diff --git a/scripts/README_TRAIN_REWARD.md b/scripts/README_TRAIN_REWARD.md old mode 100644 new mode 100755 index db8085b..596ec49 --- a/scripts/README_TRAIN_REWARD.md +++ b/scripts/README_TRAIN_REWARD.md @@ -228,6 +228,9 @@ if GPU_memory_mode == "sequential_cpu_offload": pipeline.enable_sequential_cpu_offload() elif GPU_memory_mode == "model_cpu_offload_and_qfloat8": pipeline.enable_model_cpu_offload() + for _text_encoder in [pipeline.text_encoder, pipeline.text_encoder_2]: + if hasattr(_text_encoder, "visual"): + del _text_encoder.visual convert_weight_dtype_wrapper(pipeline.transformer, weight_dtype) else: pipeline.enable_model_cpu_offload() diff --git a/scripts/train.py b/scripts/train.py old mode 100644 new mode 100755 index 20199f3..539a2f5 --- a/scripts/train.py +++ b/scripts/train.py @@ -186,8 +186,6 @@ def encode_prompt( texts.append(text) text_inputs = tokenizer( text=texts, - images=None, - videos=None, padding="max_length", max_length=max_length, truncation=True, diff --git a/scripts/train_control.py b/scripts/train_control.py old mode 100644 new mode 100755 index ce215fd..33c4f46 --- a/scripts/train_control.py +++ b/scripts/train_control.py @@ -182,8 +182,6 @@ def encode_prompt( texts.append(text) text_inputs = tokenizer( text=texts, - images=None, - videos=None, padding="max_length", max_length=max_length, truncation=True, diff --git a/scripts/train_lora.py b/scripts/train_lora.py old mode 100644 new mode 100755 index 316f2ed..93e4ca0 --- a/scripts/train_lora.py +++ b/scripts/train_lora.py @@ -186,8 +186,6 @@ def encode_prompt( texts.append(text) text_inputs = tokenizer( text=texts, - images=None, - videos=None, padding="max_length", max_length=max_length, truncation=True,