From 45c298af711b3ac7f5763aa00547d75a54681b44 Mon Sep 17 00:00:00 2001
From: Bubbliiiing <47347516+bubbliiiing@users.noreply.github.com>
Date: Thu, 26 Dec 2024 15:41:21 +0800
Subject: [PATCH 1/3] Update Readme && Update lora cache && Lora merge with
CUDA (#173)
* Update Readme
* Fix bug in training
* Update Readme
* Update Readme && Update lora cache
* Update Readme
* Update error info in sampler
* Fix Memory bug in 12b
---
README.md | 32 ++++++++++---------
README_ja-JP.md | 16 ++++++----
README_zh-CN.md | 20 ++++++------
easyanimate/data/bucket_sampler.py | 6 ++--
..._easyanimate_multi_text_encoder_inpaint.py | 2 +-
easyanimate/ui/ui.py | 28 +++++++++-------
predict_i2v.py | 4 +--
predict_t2v.py | 4 +--
predict_v2v.py | 4 +--
predict_v2v_control.py | 4 +--
scripts/README_TRAIN.md | 17 ++++++----
scripts/README_TRAIN_CONTROL.md | 17 ++++++----
scripts/README_TRAIN_LORA.md | 29 +++++++++--------
scripts/README_TRAIN_REWARD.md | 2 +-
scripts/train.py | 8 ++---
scripts/train_control.py | 4 +--
scripts/train_lora.py | 4 +--
17 files changed, 111 insertions(+), 90 deletions(-)
mode change 100644 => 100755 easyanimate/data/bucket_sampler.py
diff --git a/README.md b/README.md
index 80e54d2..6d3ae90 100644
--- a/README.md
+++ b/README.md
@@ -115,20 +115,20 @@ The detailed of Linux:
We need about 60GB available on disk (for saving weights), please check!
The video size for EasyAnimateV5-12B can be generated by different GPU Memory, including:
-| GPU memory | 384x672x72 | 384x672x49 | 576x1008x25 | 576x1008x49 | 768x1344x25 | 768x1344x49 |
-|------------|------------|------------|------------|------------|------------|------------|
-| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
-| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | ❌ | ❌ |
-| 40GB | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
-| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
+| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+|----------|----------|----------|----------|----------|----------|----------|
+| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
+| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ |
+| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
+| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
The video size for EasyAnimateV5-7B can be generated by different GPU Memory, including:
-| GPU memory | 384x672x72 | 384x672x49 | 576x1008x25 | 576x1008x49 | 768x1344x25 | 768x1344x49 |
-|------------|------------|------------|------------|------------|------------|------------|
-| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
-| 24GB | ✅ | ✅ | 🧡 | 🧡 | ❌ | ❌ |
-| 40GB | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
-| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
+| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+|----------|----------|----------|----------|----------|----------|----------|
+| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
+| 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ |
+| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
+| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
✅ indicates it can run under "model_cpu_offload", 🧡 represents it can run under "model_cpu_offload_and_qfloat8", ⭕️ indicates it can run under "sequential_cpu_offload", ❌ means it can't run. Please note that running with sequential_cpu_offload will be slower.
@@ -137,7 +137,7 @@ Some GPUs that do not support torch.bfloat16, such as 2080ti and V100, require c
The generation time for EasyAnimateV5-12B using different GPUs over 25 steps is as follows:
-| GPU | 384x672x72 | 384x672x49 | 576x1008x25 | 576x1008x49 | 768x1344x25 | 768x1344x49 |
+| GPU | 384x672x25 | 384x672x49 | 576x1008x25 | 576x1008x49 | 768x1344x25 | 768x1344x49 |
|-----------|------------------|------------------|------------------|------------------|------------------|-----------------|
| A10 24GB | ~120s (4.8s/it) | ~240s (9.6s/it) | ~320s (12.7s/it) | ~750s (29.8s/it) | ❌ | ❌ |
| A100 80GB | ~45s (1.75s/it) | ~90s (3.7s/it) | ~120s (4.7s/it) | ~300s (11.4s/it) | ~265s (10.6s/it) | ~710s (28.3s/it) |
@@ -149,7 +149,7 @@ The generation time for EasyAnimateV5-12B using different GPUs over 25 steps is
The video size for EasyAnimateV3 can be generated by different GPU Memory, including:
-| GPU memory | 384x672x72 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
+| GPU memory | 384x672x25 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
|------------|------------|-------------|-------------|--------------|-------------|--------------|
| 12GB | ⭕️ | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 16GB | ✅ | ✅ | ⭕️ | ⭕️ | ⭕️ | ❌ |
@@ -323,7 +323,9 @@ A complete EasyAnimate training pipeline should include data preprocessing, Vide
a. data preprocessing
-We have provided a simple demo of training the Lora model through image data, which can be found in the [wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora) for details.
+We provide two simple demos:
+- Train a Lora model using image data. For more details, you can refer to the [wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora).
+- Perform SFT model training using video data. For more details, you can refer to the [wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-SFT).
A complete data preprocessing link for long video segmentation, cleaning, and description can refer to [README](./easyanimate/video_caption/README.md) in the video captions section.
diff --git a/README_ja-JP.md b/README_ja-JP.md
index 3872bad..dcaaf3c 100644
--- a/README_ja-JP.md
+++ b/README_ja-JP.md
@@ -115,18 +115,18 @@ Linuxの詳細:
ディスクに約60GBの空き容量が必要です(重みを保存するため)、確認してください!
EasyAnimateV5-12Bのビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください:
-| GPUメモリ |384x672x72|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
-| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | ❌ | ❌ |
+| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
EasyAnimateV5-7Bのビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください:
-| GPUメモリ |384x672x72|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
-| 24GB | ✅ | ✅ | 🧡 | 🧡 | ❌ | ❌ |
+| 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
@@ -135,7 +135,7 @@ EasyAnimateV5-7Bのビデオサイズは異なるGPUメモリにより生成で
一部のGPU(例:2080ti、V100)はtorch.bfloat16をサポートしていないため、app.pyおよびpredictファイル内のweight_dtypeをtorch.float16に変更する必要があります。
EasyAnimateV5-12Bは異なるGPUで25ステップ生成する時間は次の通りです:
-| GPU |384x672x72|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+| GPU |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| A10 24GB |約120秒 (4.8s/it)|約240秒 (9.6s/it)|約320秒 (12.7s/it)|約750秒 (29.8s/it)| ❌ | ❌ |
| A100 80GB |約45秒 (1.75s/it)|約90秒 (3.7s/it)|約120秒 (4.7s/it)|約300秒 (11.4s/it)|約265秒 (10.6s/it)| 約710秒 (28.3s/it)|
@@ -145,7 +145,7 @@ EasyAnimateV5-12Bは異なるGPUで25ステップ生成する時間は次の通
(廃止予定) EasyAnimateV3:
EasyAnimateV3のビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください:
-| GPUメモリ | 384x672x72 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
+| GPUメモリ | 384x672x25 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
|----------|----------|----------|----------|----------|----------|----------|
| 12GB | ⭕️ | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 16GB | ✅ | ✅ | ⭕️ | ⭕️ | ⭕️ | ❌ |
@@ -318,7 +318,9 @@ EasyAnimateV5のパラメータが大きいため、メモリを節約するた
a. データ前処理
-画像データを使用してLoraモデルをトレーニングする簡単なデモを提供しています。詳細は[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)を参照してください。
+私たちは2つの簡単なデモを提供します:
+- 画像データを使用してLoraモデルを訓練します。詳細はこちらの[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)をご覧ください。
+- 動画データを使用してSFTモデルを訓練します。詳細はこちらの[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-SFT)をご覧ください。
長時間動画のセグメンテーション、クリーニング、および説明のための完全なデータ前処理リンクは、ビデオキャプションセクションの[README](./easyanimate/video_caption/README.md)を参照してください。
diff --git a/README_zh-CN.md b/README_zh-CN.md
index cd73887..9c98406 100644
--- a/README_zh-CN.md
+++ b/README_zh-CN.md
@@ -113,19 +113,19 @@ Linux 的详细信息:
我们需要大约 60GB 的可用磁盘空间,请检查!
EasyAnimateV5-12B的视频大小可以由不同的GPU Memory生成,包括:
-| GPU memory |384x672x72|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
-| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | ❌ | ❌ |
-| 40GB | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
+| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ |
+| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
EasyAnimateV5-7B的视频大小可以由不同的GPU Memory生成,包括:
-| GPU memory |384x672x72|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | 🧡 | ❌ | ❌ | ❌ | ❌ |
-| 24GB | ✅ | ✅ | 🧡 | 🧡 | ❌ | ❌ |
-| 40GB | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
+| 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ |
+| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
✅ 表示它可以在"model_cpu_offload"的情况下运行,🧡代表它可以在"model_cpu_offload_and_qfloat8"的情况下运行,⭕️ 表示它可以在"sequential_cpu_offload"的情况下运行,❌ 表示它无法运行。请注意,使用sequential_cpu_offload运行会更慢。
@@ -133,7 +133,7 @@ EasyAnimateV5-7B的视频大小可以由不同的GPU Memory生成,包括:
有一些不支持torch.bfloat16的卡型,如2080ti、V100,需要将app.py、predict文件中的weight_dtype修改为torch.float16才可以运行。
EasyAnimateV5-12B使用不同GPU在25个steps中的生成时间如下:
-| GPU |384x672x72|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
+| GPU |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| A10 24GB |约120秒 (4.8s/it)|约240秒 (9.6s/it)|约320秒 (12.7s/it)| 约750秒 (29.8s/it)| ❌ | ❌ |
| A100 80GB |约45秒 (1.75s/it)|约90秒 (3.7s/it)|约120秒 (4.7s/it)|约300秒 (11.4s/it)|约265秒 (10.6s/it)| 约710秒 (28.3s/it)|
@@ -144,7 +144,7 @@ EasyAnimateV5-12B使用不同GPU在25个steps中的生成时间如下:
(Obsolete) EasyAnimateV3:
EasyAnimateV3的视频大小可以由不同的GPU Memory生成,包括:
-| GPU memory | 384x672x72 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
+| GPU memory | 384x672x25 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
|----------|----------|----------|----------|----------|----------|----------|
| 12GB | ⭕️ | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 16GB | ✅ | ✅ | ⭕️ | ⭕️ | ⭕️ | ❌ |
@@ -317,7 +317,9 @@ qfloat8会降低模型的性能,但可以节省更多的显存。如果显存
a.数据预处理
-我们给出了一个简单的demo通过图片数据训练lora模型,详情可以查看[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)。
+我们给出了两个简单的demo:
+- 通过图片数据训练lora模型,详情可以查看[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)。
+- 通过视频数据进行SFT模型,详情可以查看[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-SFT)。
一个完整的长视频切分、清洗、描述的数据预处理链路可以参考video caption部分的[README](easyanimate/video_caption/README.md)进行。
diff --git a/easyanimate/data/bucket_sampler.py b/easyanimate/data/bucket_sampler.py
old mode 100644
new mode 100755
index 2c5fded..24b4160
--- a/easyanimate/data/bucket_sampler.py
+++ b/easyanimate/data/bucket_sampler.py
@@ -254,7 +254,7 @@ class AspectRatioBatchSampler(BatchSampler):
width = int(width)
ratio = height / width # self.dataset[idx]
except Exception as e:
- print(e)
+ print(e, self.dataset[idx], "This item is error, please check it.")
continue
# find the closest aspect ratio
closest_ratio = min(self.aspect_ratios.keys(), key=lambda r: abs(float(r) - ratio))
@@ -330,7 +330,7 @@ class AspectRatioBatchImageVideoSampler(BatchSampler):
width = int(width)
ratio = height / width # self.dataset[idx]
except Exception as e:
- print(e)
+ print(e, self.dataset[idx], "This item is error, please check it.")
continue
# find the closest aspect ratio
closest_ratio = min(self.aspect_ratios.keys(), key=lambda r: abs(float(r) - ratio))
@@ -365,7 +365,7 @@ class AspectRatioBatchImageVideoSampler(BatchSampler):
width = int(width)
ratio = height / width # self.dataset[idx]
except Exception as e:
- print(e)
+ print(e, self.dataset[idx], "This item is error, please check it.")
continue
# find the closest aspect ratio
closest_ratio = min(self.aspect_ratios.keys(), key=lambda r: abs(float(r) - ratio))
diff --git a/easyanimate/pipeline/pipeline_easyanimate_multi_text_encoder_inpaint.py b/easyanimate/pipeline/pipeline_easyanimate_multi_text_encoder_inpaint.py
index cecf5a2..5e55af9 100644
--- a/easyanimate/pipeline/pipeline_easyanimate_multi_text_encoder_inpaint.py
+++ b/easyanimate/pipeline/pipeline_easyanimate_multi_text_encoder_inpaint.py
@@ -172,7 +172,7 @@ class EasyAnimatePipeline_Multi_Text_Encoder_Inpaint(DiffusionPipeline):
The image processor for the CLIP image embedder.
"""
- model_cpu_offload_seq = "text_encoder->text_encoder_2->clip_image_encoder->transformer->vae"
+ model_cpu_offload_seq = "text_encoder->text_encoder_2->transformer->vae"
_optional_components = [
"safety_checker",
"feature_extractor",
diff --git a/easyanimate/ui/ui.py b/easyanimate/ui/ui.py
index f6c253d..d088a4f 100644
--- a/easyanimate/ui/ui.py
+++ b/easyanimate/ui/ui.py
@@ -334,12 +334,16 @@ class EasyAnimateController:
return gr.update(value=None)
else:
base_model_dropdown = os.path.join(self.personalized_model_dir, base_model_dropdown)
- base_model_state_dict = {}
- with safe_open(base_model_dropdown, framework="pt", device="cpu") as f:
- for key in f.keys():
- base_model_state_dict[key] = f.get_tensor(key)
- self.transformer.load_state_dict(base_model_state_dict, strict=False)
- print("Update base done")
+ print(f"From checkpoint: {base_model_dropdown}")
+ if base_model_dropdown.endswith("safetensors"):
+ from safetensors.torch import load_file, safe_open
+ state_dict = load_file(base_model_dropdown)
+ else:
+ state_dict = torch.load(base_model_dropdown, map_location="cpu")
+ state_dict = state_dict["state_dict"] if "state_dict" in state_dict else state_dict
+
+ m, u = self.transformer.load_state_dict(state_dict, strict=False)
+ print(f"Update base done. Missing keys: {len(m)}, unexpected keys: {len(u)}")
return gr.update()
def update_lora_model(self, lora_model_dropdown):
@@ -450,7 +454,7 @@ class EasyAnimateController:
self.pipeline.scheduler = scheduler_dict[sampler_dropdown].from_config(self.pipeline.scheduler.config)
if self.lora_model_path != "none":
# lora part
- self.pipeline = merge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider)
+ self.pipeline = merge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider, device="cuda", dtype=self.weight_dtype)
if int(seed_textbox) != -1 and seed_textbox != "": torch.manual_seed(int(seed_textbox))
else: seed_textbox = np.random.randint(0, 1e10)
@@ -588,7 +592,7 @@ class EasyAnimateController:
torch.cuda.empty_cache()
torch.cuda.ipc_collect()
if self.lora_model_path != "none":
- self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider)
+ self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider, device="cuda", dtype=self.weight_dtype)
if is_api:
return "", f"Error. error information is {str(e)}"
else:
@@ -600,7 +604,7 @@ class EasyAnimateController:
# lora part
if self.lora_model_path != "none":
- self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider)
+ self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider, device="cuda", dtype=self.weight_dtype)
sample_config = {
"prompt": prompt_textbox,
@@ -1256,7 +1260,7 @@ class EasyAnimateController_Modelscope:
self.pipeline.scheduler = scheduler_dict[sampler_dropdown].from_config(self.pipeline.scheduler.config)
if self.lora_model_path != "none":
# lora part
- self.pipeline = merge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider)
+ self.pipeline = merge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider, device="cuda", dtype=self.weight_dtype)
if int(seed_textbox) != -1 and seed_textbox != "": torch.manual_seed(int(seed_textbox))
else: seed_textbox = np.random.randint(0, 1e10)
@@ -1327,7 +1331,7 @@ class EasyAnimateController_Modelscope:
torch.cuda.empty_cache()
torch.cuda.ipc_collect()
if self.lora_model_path != "none":
- self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider)
+ self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider, device="cuda", dtype=self.weight_dtype)
if is_api:
return "", f"Error. error information is {str(e)}"
else:
@@ -1339,7 +1343,7 @@ class EasyAnimateController_Modelscope:
# lora part
if self.lora_model_path != "none":
- self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider)
+ self.pipeline = unmerge_lora(self.pipeline, self.lora_model_path, multiplier=lora_alpha_slider, device="cuda", dtype=self.weight_dtype)
if not os.path.exists(self.savedir_sample):
os.makedirs(self.savedir_sample, exist_ok=True)
diff --git a/predict_i2v.py b/predict_i2v.py
index 745af67..c745459 100644
--- a/predict_i2v.py
+++ b/predict_i2v.py
@@ -239,7 +239,7 @@ else:
generator = torch.Generator(device="cuda").manual_seed(seed)
if lora_path is not None:
- pipeline = merge_lora(pipeline, lora_path, lora_weight, "cuda")
+ pipeline = merge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
if partial_video_length is not None:
init_frames = 0
@@ -327,7 +327,7 @@ else:
).videos
if lora_path is not None:
- pipeline = unmerge_lora(pipeline, lora_path, lora_weight, "cuda")
+ pipeline = unmerge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
if not os.path.exists(save_path):
os.makedirs(save_path, exist_ok=True)
diff --git a/predict_t2v.py b/predict_t2v.py
index c4f3acc..78db586 100644
--- a/predict_t2v.py
+++ b/predict_t2v.py
@@ -259,7 +259,7 @@ else:
generator = torch.Generator(device="cuda").manual_seed(seed)
if lora_path is not None:
- pipeline = merge_lora(pipeline, lora_path, lora_weight)
+ pipeline = merge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
with torch.no_grad():
if transformer.config.in_channels != vae.config.latent_channels:
@@ -296,7 +296,7 @@ with torch.no_grad():
).videos
if lora_path is not None:
- pipeline = unmerge_lora(pipeline, lora_path, lora_weight)
+ pipeline = unmerge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
if not os.path.exists(save_path):
os.makedirs(save_path, exist_ok=True)
diff --git a/predict_v2v.py b/predict_v2v.py
index 2331882..2e9a22e 100644
--- a/predict_v2v.py
+++ b/predict_v2v.py
@@ -237,7 +237,7 @@ else:
generator = torch.Generator(device="cuda").manual_seed(seed)
if lora_path is not None:
- pipeline = merge_lora(pipeline, lora_path, lora_weight, "cuda")
+ pipeline = merge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
if vae.cache_mag_vae:
video_length = int((video_length - 1) // vae.mini_batch_encoder * vae.mini_batch_encoder) + 1 if video_length != 1 else 1
@@ -263,7 +263,7 @@ with torch.no_grad():
).videos
if lora_path is not None:
- pipeline = unmerge_lora(pipeline, lora_path, lora_weight, "cuda")
+ pipeline = unmerge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
if not os.path.exists(save_path):
os.makedirs(save_path, exist_ok=True)
diff --git a/predict_v2v_control.py b/predict_v2v_control.py
index 1871e49..553658a 100644
--- a/predict_v2v_control.py
+++ b/predict_v2v_control.py
@@ -219,7 +219,7 @@ else:
generator = torch.Generator(device="cuda").manual_seed(seed)
if lora_path is not None:
- pipeline = merge_lora(pipeline, lora_path, lora_weight)
+ pipeline = merge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
with torch.no_grad():
if vae.cache_mag_vae:
@@ -242,7 +242,7 @@ with torch.no_grad():
).videos
if lora_path is not None:
- pipeline = unmerge_lora(pipeline, lora_path, lora_weight)
+ pipeline = unmerge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
if not os.path.exists(save_path):
os.makedirs(save_path, exist_ok=True)
diff --git a/scripts/README_TRAIN.md b/scripts/README_TRAIN.md
index e5a590c..286aecf 100644
--- a/scripts/README_TRAIN.md
+++ b/scripts/README_TRAIN.md
@@ -8,12 +8,17 @@ Some parameters in the sh file can be confusing, and they are explained in this
- `enable_bucket` is used to enable bucket training. When enabled, the model does not crop the images and videos at the center, but instead, it trains the entire images and videos after grouping them into buckets based on resolution.
- `random_frame_crop` is used for random cropping on video frames to simulate videos with different frame counts.
-- `random_hw_adapt` is used to enable automatic height and width scaling for images and videos. When random_hw_adapt is enabled, the training images will have their height and width set to image_sample_size as the maximum and video_sample_size as the minimum. For training videos, the height and width will be set to video_sample_size as the maximum and min(video_sample_size, 512) as the minimum.
-- `training_with_video_token_length` specifies training the model according to token length. The token length for a video with dimensions 512x512 and 49 frames is 13,312.
- - At 512x512 resolution, the number of video frames is 49;
- - At 768x768 resolution, the number of video frames is 21;
- - At 1024x1024 resolution, the number of video frames is 9;
- - These resolutions combined with their corresponding lengths allow the model to generate videos of different sizes.
+- `random_hw_adapt` is used to enable automatic height and width scaling for images and videos. When `random_hw_adapt` is enabled, the training images will have their height and width set to `image_sample_size` as the maximum and `min(video_sample_size, 512)` as the minimum. For training videos, the height and width will be set to `video_sample_size` as the maximum and `min(video_sample_size, 512)` as the minimum.
+ - For example, when `random_hw_adapt` is enabled, with `video_sample_n_frames=49`, `video_sample_size=1024`, and `image_sample_size=1024`, the resolution of image inputs for training is `512x512` to `1024x1024`, and the shape of video inputs for training is `512x512x49` to `1024x1024x49`.
+ - For example, when `random_hw_adapt` is enabled, with `video_sample_n_frames=49`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the shape of video inputs for training is `256x256x49`.
+- `training_with_video_token_length` specifies training the model according to token length. For training images and videos, the height and width will be set to `image_sample_size` as the maximum and `video_sample_size` as the minimum.
+ - For example, when `training_with_video_token_length` is enabled, with `video_sample_n_frames=49`, `token_sample_size=1024`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the shape of video inputs for training is `256x256x49` to `1024x1024x49`.
+ - For example, when `training_with_video_token_length` is enabled, with `video_sample_n_frames=49`, `token_sample_size=512`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the shape of video inputs for training is `256x256x49` to `1024x1024x9`.
+ - The token length for a video with dimensions 512x512 and 49 frames is 13,312. We need to set the `token_sample_size = 512`.
+ - At 512x512 resolution, the number of video frames is 49 (~= 512 * 512 * 49 / 512 / 512).
+ - At 768x768 resolution, the number of video frames is 21 (~= 512 * 512 * 49 / 768 / 768).
+ - At 1024x1024 resolution, the number of video frames is 9 (~= 512 * 512 * 49 / 1024 / 1024).
+ - These resolutions combined with their corresponding lengths allow the model to generate videos of different sizes.
- `train_mode` is used to specify the training mode, which can be either normal or inpaint. Since EasyAnimate uses the Inpaint model to achieve image-to-video generation, the default is set to inpaint mode. If you only wish to achieve text-to-video generation, you can remove this line, and it will default to the text-to-video mode.
- `uniform_sampling` is used to ensure that each batch can be uniformly sampled from 0 to 1000.
- The default parameter for training is the Inpaint model. If you only want to train the T2V model, please set train_made="normal" and use the EasyAnimateV5-12b-zh model.
diff --git a/scripts/README_TRAIN_CONTROL.md b/scripts/README_TRAIN_CONTROL.md
index d77c6e3..83b3b72 100644
--- a/scripts/README_TRAIN_CONTROL.md
+++ b/scripts/README_TRAIN_CONTROL.md
@@ -28,12 +28,17 @@ Some parameters in the sh file can be confusing, and they are explained in this
- `enable_bucket` is used to enable bucket training. When enabled, the model does not crop the images and videos at the center, but instead, it trains the entire images and videos after grouping them into buckets based on resolution.
- `random_frame_crop` is used for random cropping on video frames to simulate videos with different frame counts.
-- `random_hw_adapt` is used to enable automatic height and width scaling for images and videos. When random_hw_adapt is enabled, the training images will have their height and width set to image_sample_size as the maximum and video_sample_size as the minimum. For training videos, the height and width will be set to video_sample_size as the maximum and min(video_sample_size, 512) as the minimum.
-- `training_with_video_token_length` specifies training the model according to token length. The token length for a video with dimensions 512x512 and 49 frames is 13,312.
- - At 512x512 resolution, the number of video frames is 49;
- - At 768x768 resolution, the number of video frames is 21;
- - At 1024x1024 resolution, the number of video frames is 9;
- - These resolutions combined with their corresponding lengths allow the model to generate videos of different sizes.
+- `random_hw_adapt` is used to enable automatic height and width scaling for images and videos. When `random_hw_adapt` is enabled, the training images will have their height and width set to `image_sample_size` as the maximum and `min(video_sample_size, 512)` as the minimum. For training videos, the height and width will be set to `video_sample_size` as the maximum and `min(video_sample_size, 512)` as the minimum.
+ - For example, when `random_hw_adapt` is enabled, with `video_sample_n_frames=49`, `video_sample_size=1024`, and `image_sample_size=1024`, the resolution of image inputs for training is `512x512` to `1024x1024`, and the resolution of video inputs for training is `512x512x49` to `1024x1024x49`.
+ - For example, when `random_hw_adapt` is enabled, with `video_sample_n_frames=49`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the resolution of video inputs for training is `256x256x49`.
+- `training_with_video_token_length` specifies training the model according to token length. For training images and videos, the height and width will be set to `image_sample_size` as the maximum and `video_sample_size` as the minimum.
+ - For example, when `training_with_video_token_length` is enabled, with `video_sample_n_frames=49`, `token_sample_size=1024`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the resolution of video inputs for training is `256x256x49` to `1024x1024x49`.
+ - For example, when `training_with_video_token_length` is enabled, with `video_sample_n_frames=49`, `token_sample_size=512`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the resolution of video inputs for training is `256x256x49` to `1024x1024x9`.
+ - The token length for a video with dimensions 512x512 and 49 frames is 13,312. We need to set the `token_sample_size = 512`.
+ - At 512x512 resolution, the number of video frames is 49 (~= 512 * 512 * 49 / 512 / 512).
+ - At 768x768 resolution, the number of video frames is 21 (~= 512 * 512 * 49 / 768 / 768).
+ - At 1024x1024 resolution, the number of video frames is 9 (~= 512 * 512 * 49 / 1024 / 1024).
+ - These resolutions combined with their corresponding lengths allow the model to generate videos of different sizes.
EasyAnimateV5 without deepspeed:
```sh
diff --git a/scripts/README_TRAIN_LORA.md b/scripts/README_TRAIN_LORA.md
index db026ab..c9a5711 100644
--- a/scripts/README_TRAIN_LORA.md
+++ b/scripts/README_TRAIN_LORA.md
@@ -8,12 +8,17 @@ Some parameters in the sh file can be confusing, and they are explained in this
- `enable_bucket` is used to enable bucket training. When enabled, the model does not crop the images and videos at the center, but instead, it trains the entire images and videos after grouping them into buckets based on resolution.
- `random_frame_crop` is used for random cropping on video frames to simulate videos with different frame counts.
-- `random_hw_adapt` is used to enable automatic height and width scaling for images and videos. When random_hw_adapt is enabled, the training images will have their height and width set to image_sample_size as the maximum and video_sample_size as the minimum. For training videos, the height and width will be set to video_sample_size as the maximum and min(video_sample_size, 512) as the minimum.
-- `training_with_video_token_length` specifies training the model according to token length. The token length for a video with dimensions 512x512 and 49 frames is 13,312.
- - At 512x512 resolution, the number of video frames is 49;
- - At 768x768 resolution, the number of video frames is 21;
- - At 1024x1024 resolution, the number of video frames is 9;
- - These resolutions combined with their corresponding lengths allow the model to generate videos of different sizes.
+- `random_hw_adapt` is used to enable automatic height and width scaling for images and videos. When `random_hw_adapt` is enabled, the training images will have their height and width set to `image_sample_size` as the maximum and `min(video_sample_size, 512)` as the minimum. For training videos, the height and width will be set to `video_sample_size` as the maximum and `min(video_sample_size, 512)` as the minimum.
+ - For example, when `random_hw_adapt` is enabled, with `video_sample_n_frames=49`, `video_sample_size=1024`, and `image_sample_size=1024`, the resolution of image inputs for training is `512x512` to `1024x1024`, and the resolution of video inputs for training is `512x512x49` to `1024x1024x49`.
+ - For example, when `random_hw_adapt` is enabled, with `video_sample_n_frames=49`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the resolution of video inputs for training is `256x256x49`.
+- `training_with_video_token_length` specifies training the model according to token length. For training images and videos, the height and width will be set to `image_sample_size` as the maximum and `video_sample_size` as the minimum.
+ - For example, when `training_with_video_token_length` is enabled, with `video_sample_n_frames=49`, `token_sample_size=1024`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the resolution of video inputs for training is `256x256x49` to `1024x1024x49`.
+ - For example, when `training_with_video_token_length` is enabled, with `video_sample_n_frames=49`, `token_sample_size=512`, `video_sample_size=1024`, and `image_sample_size=256`, the resolution of image inputs for training is `256x256` to `1024x1024`, and the resolution of video inputs for training is `256x256x49` to `1024x1024x9`.
+ - The token length for a video with dimensions 512x512 and 49 frames is 13,312. We need to set the `token_sample_size = 512`.
+ - At 512x512 resolution, the number of video frames is 49 (~= 512 * 512 * 49 / 512 / 512).
+ - At 768x768 resolution, the number of video frames is 21 (~= 512 * 512 * 49 / 768 / 768).
+ - At 1024x1024 resolution, the number of video frames is 9 (~= 512 * 512 * 49 / 1024 / 1024).
+ - These resolutions combined with their corresponding lengths allow the model to generate videos of different sizes.
- `train_mode` is used to specify the training mode, which can be either normal or inpaint. Since EasyAnimate uses the Inpaint model to achieve image-to-video generation, the default is set to inpaint mode. If you only wish to achieve text-to-video generation, you can remove this line, and it will default to the text-to-video mode.
- `uniform_sampling` is used to ensure that each batch can be uniformly sampled from 0 to 1000.
- The default parameter for training is the Inpaint model. If you only want to train the T2V model, please set train_made="normal" and use the EasyAnimateV5-12b-zh model.
@@ -150,8 +155,7 @@ accelerate launch --mixed_precision="bf16" scripts/train_lora.py \
--motion_sub_loss \
--not_sigma_loss \
--enable_bucket \
- --train_mode="inpaint" \
- --random_frame_crop
+ --train_mode="inpaint"
```
EasyAnimateV4 with deepspeed:
@@ -195,8 +199,7 @@ accelerate launch --use_deepspeed --deepspeed_config_file config/zero_stage2_con
--not_sigma_loss \
--enable_bucket \
--use_deepspeed \
- --train_mode="inpaint" \
- --random_frame_crop
+ --train_mode="inpaint"
```
@@ -239,8 +242,7 @@ accelerate launch --mixed_precision="bf16" scripts/train_lora.py \
--max_grad_norm=1 \
--vae_mini_batch=1 \
--enable_bucket \
- --train_mode="inpaint" \
- --random_frame_crop
+ --train_mode="inpaint"
```
@@ -282,7 +284,6 @@ accelerate launch --mixed_precision="bf16" scripts/train_t2iv_lora.py \
--adam_epsilon=1e-10 \
--max_grad_norm=1 \
--vae_mini_batch=1 \
- --enable_bucket \
- --random_frame_crop
+ --enable_bucket
```
\ No newline at end of file
diff --git a/scripts/README_TRAIN_REWARD.md b/scripts/README_TRAIN_REWARD.md
index cd05f37..e9c2873 100644
--- a/scripts/README_TRAIN_REWARD.md
+++ b/scripts/README_TRAIN_REWARD.md
@@ -228,7 +228,7 @@ elif GPU_memory_mode == "model_cpu_offload_and_qfloat8":
convert_weight_dtype_wrapper(pipeline.transformer, weight_dtype)
else:
pipeline.enable_model_cpu_offload()
-pipeline = merge_lora(pipeline, lora_path, lora_weight)
+pipeline = merge_lora(pipeline, lora_path, lora_weight, device="cuda", dtype=weight_dtype)
generator = torch.Generator(device="cuda").manual_seed(42)
input_video, input_video_mask, _ = get_image_to_video_latent(None, None, video_length=video_length, sample_size=sample_size)
diff --git a/scripts/train.py b/scripts/train.py
index b5d038f..10b2ca7 100644
--- a/scripts/train.py
+++ b/scripts/train.py
@@ -1042,9 +1042,9 @@ def main():
def save_model_hook(models, weights, output_dir):
if accelerator.is_main_process:
if args.use_ema:
- ema_transformer3d.save_pretrained(os.path.join(output_dir, "transformer_ema"))
+ ema_transformer3d.save_pretrained(os.path.join(output_dir, "transformer_ema"), max_shard_size="30GB")
- models[0].save_pretrained(os.path.join(output_dir, "transformer"))
+ models[0].save_pretrained(os.path.join(output_dir, "transformer"), max_shard_size="30GB")
if not args.use_deepspeed:
weights.pop()
@@ -1671,9 +1671,9 @@ def main():
torch.cuda.empty_cache()
vae.to(accelerator.device)
if not args.enable_text_encoder_in_dataloader:
- text_encoder.to(accelerator.device)
+ text_encoder.to('cpu')
if text_encoder_2 is not None:
- text_encoder_2.to(accelerator.device)
+ text_encoder_2.to('cpu')
with torch.no_grad():
video_length = pixel_values.shape[1]
diff --git a/scripts/train_control.py b/scripts/train_control.py
index 21f189e..22c8fd4 100644
--- a/scripts/train_control.py
+++ b/scripts/train_control.py
@@ -951,9 +951,9 @@ def main():
def save_model_hook(models, weights, output_dir):
if accelerator.is_main_process:
if args.use_ema:
- ema_transformer3d.save_pretrained(os.path.join(output_dir, "transformer_ema"))
+ ema_transformer3d.save_pretrained(os.path.join(output_dir, "transformer_ema"), max_shard_size="30GB")
- models[0].save_pretrained(os.path.join(output_dir, "transformer"))
+ models[0].save_pretrained(os.path.join(output_dir, "transformer"), max_shard_size="30GB")
if not args.use_deepspeed:
weights.pop()
diff --git a/scripts/train_lora.py b/scripts/train_lora.py
index 7331346..038dd2d 100644
--- a/scripts/train_lora.py
+++ b/scripts/train_lora.py
@@ -1649,9 +1649,9 @@ def main():
torch.cuda.empty_cache()
vae.to(accelerator.device)
if not args.enable_text_encoder_in_dataloader:
- text_encoder.to(accelerator.device)
+ text_encoder.to('cpu')
if text_encoder_2 is not None:
- text_encoder_2.to(accelerator.device)
+ text_encoder_2.to('cpu')
with torch.no_grad():
video_length = pixel_values.shape[1]
From 5b9777d364b7c01cdef05f3604e836db0472111a Mon Sep 17 00:00:00 2001
From: Robin Huang
Date: Sun, 12 Jan 2025 18:20:11 -0800
Subject: [PATCH 2/3] chore(publish): Add Github Action for Publishing to Comfy
Registry (#178)
Co-authored-by: snomiao
---
.github/workflows/publish.yml | 25 +++++++++++++++++++++++++
1 file changed, 25 insertions(+)
create mode 100644 .github/workflows/publish.yml
diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml
new file mode 100644
index 0000000..00b01d4
--- /dev/null
+++ b/.github/workflows/publish.yml
@@ -0,0 +1,25 @@
+name: Publish to Comfy registry
+on:
+ workflow_dispatch:
+ push:
+ branches:
+ - main
+ - master
+ paths:
+ - "pyproject.toml"
+
+jobs:
+ publish-node:
+ name: Publish Custom Node to registry
+ runs-on: ubuntu-latest
+ if: ${{ github.repository_owner == 'aigc-apps' }}
+ steps:
+ - name: Check out code
+ uses: actions/checkout@v4
+ with:
+ submodules: true
+ - name: Publish Custom Node
+ uses: Comfy-Org/publish-node-action@main
+ with:
+ ## Add your own personal access token to your Github Repository secrets and reference it here.
+ personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
From 78415be88246aa5eca2d88171360a8023af4d786 Mon Sep 17 00:00:00 2001
From: Robin Huang
Date: Sun, 12 Jan 2025 18:20:52 -0800
Subject: [PATCH 3/3] Add pyproject.toml for Custom Node Registry (#177)
* chore(pyproject): Add pyproject.toml for Custom Node Registry
* Update pyproject.toml
Add Publisher id
---------
Co-authored-by: snomiao
Co-authored-by: Bubbliiiing <47347516+bubbliiiing@users.noreply.github.com>
---
pyproject.toml | 15 +++++++++++++++
1 file changed, 15 insertions(+)
create mode 100644 pyproject.toml
diff --git a/pyproject.toml b/pyproject.toml
new file mode 100644
index 0000000..5b59328
--- /dev/null
+++ b/pyproject.toml
@@ -0,0 +1,15 @@
+[project]
+name = "easyanimate"
+description = "Video Generation Nodes for EasyAnimate, which suppors text-to-video, image-to-video, video-to-video and different controls."
+version = "1.0.0"
+license = {file = "LICENSE"}
+dependencies = ["Pillow", "einops", "safetensors", "timm", "tomesd", "torch>=2.1.2", "torchdiffeq", "torchsde", "decord", "datasets", "numpy", "scikit-image", "opencv-python", "omegaconf", "SentencePiece", "albumentations", "imageio[ffmpeg]", "imageio[pyav]", "tensorboard", "beautifulsoup4", "ftfy", "func_timeout", "accelerate>=0.25.0", "gradio>=3.41.2,<=3.48.0", "diffusers>=0.30.1", "transformers>=4.37.2"]
+
+[project.urls]
+Repository = "https://github.com/aigc-apps/EasyAnimate"
+# Used by Comfy Registry https://comfyregistry.org
+
+[tool.comfy]
+PublisherId = "bubbliiiing"
+DisplayName = "EasyAnimate"
+Icon = ""