From 35e134c2e2cd55563895dc1033024d8cc18a5b01 Mon Sep 17 00:00:00 2001 From: bubbliiiing <3323290568@qq.com> Date: Wed, 29 May 2024 10:52:17 +0800 Subject: [PATCH] update lots of readme --- README.md | 43 ++++++++----- README_zh-CN.md | 48 +++++++++------ easyanimate/vae/README.md | 60 +++++++++++++++++++ .../autoencoder_kl_32x32x4_mag.yaml | 2 + .../autoencoder_kl_32x32x4_slice.yaml | 4 +- ...encoder_kl_32x32x4_slice_decoder_only.yaml | 4 +- ...coder_kl_32x32x4_slice_t_downsample_8.yaml | 4 +- .../vae/ldm/data/dataset_image_video.py | 13 +++- 8 files changed, 141 insertions(+), 37 deletions(-) create mode 100644 easyanimate/vae/README.md diff --git a/README.md b/README.md index 75f0da5..d7aea72 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,7 @@ -# 📷 EasyAnimate | Integrated generation of baseline scheme for videos and images. -😊 EasyAnimate is a repo for generating long videos and images, training transformer based diffusion generators. +# 📷 EasyAnimate | An End-to-End Solution for High-Resolution and Long Video Generation +😊 EasyAnimate is an end-to-end solution for generating high-resolution and long videos. We can train transformer based diffusion generators, train VAEs for processing long videos, and preprocess metadata. -😊 Based on Sora like structure and DIT, we use transformer as a diffuser for video generation. In order to ensure good expansibility, we built easyanimate based on motion module. In the future, we will try more training programs to improve the effect. +😊 Based on Sora like structure and DIT, we use transformer as a diffuser for video generation. We built easyanimate based on motion module, u-vit and slice-vae. In the future, we will try more training programs to improve the effect. 😊 Welcome! @@ -68,7 +68,11 @@ We show some results in the [GALLERY](scripts/Result%20Gallery.md). # Quick Start ### 1. Cloud usage: AliyunDSW/Docker #### a. From AliyunDSW -Stay tuned. +DSW has free GPU time, which can be applied once by a user and is valid for 3 months after applying. + +Aliyun provide free GPU time in [Freetier](https://free.aliyun.com/?product=9602825&crowd=enterprise&spm=5176.28055625.J_5831864660.1.e939154aRgha4e&scm=20140722.M_9974135.P_110.MO_1806-ID_9974135-MID_9974135-CID_30683-ST_8512-V_1), get it and use in Aliyun PAI-DSW to start EasyAnimate within 5min! + +[![DSW Notebook](images/dsw.png)](https://gallery.pai-ml.com/#/preview/deepLearning/cv/easyanimate) #### b. From docker If you are using docker, please make sure that the graphics card driver and CUDA environment have been installed correctly in your machine. @@ -186,7 +190,13 @@ EasyAnimateV2: - Step 3: Select the generated model based on the page, fill in prompt, neg_prompt, guidance_scale, and seed, click on generate, wait for the generated result, and save the result in the samples folder. ### 2. Model Training -We have provided a simple demo of training the Lora model through image data, which can be found in the [wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora) for details. +A complete EasyAnimate training pipeline should include data preprocessing, Video VAE training, and Video DiT training. Among these, Video VAE training is optional because we have already provided a pre-trained Video VAE. + +#### a. data preprocessing +We have provided a simple demo of training the Lora model through image data, which can be found in the [wiki](https://github.com/aigc-apps/ +EasyAnimate/wiki/Training-Lora) for details. + +A complete data preprocessing link for long video segmentation, cleaning, and description can refer to [README](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora) in the video captions section. If you want to train a text to image and video generation model. You need to arrange the dataset in this format. @@ -201,7 +211,7 @@ If you want to train a text to image and video generation model. You need to arr │ └── 📄 json_of_internal_datasets.json ``` -The json_of_internal_datasets.json is a standard JSON file, as shown in below: +The json_of_internal_datasets.json is a standard JSON file. The file_path in the json can to be set as relative path, as shown in below: ```json [ { @@ -217,13 +227,6 @@ The json_of_internal_datasets.json is a standard JSON file, as shown in below: ..... ] ``` -The file_path in the json can to be set as relative path. - -Then, set scripts/train_t2iv.sh. -``` -export DATASET_NAME="datasets/internal_datasets/" -export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.json" -``` You can also set the path as absolute path as follow: ```json @@ -241,7 +244,19 @@ You can also set the path as absolute path as follow: ..... ] ``` -The scripts/train_t2iv.sh should be set as follow: + +#### b. Video VAE training (optional) +Video VAE training is an optional option as we have already provided pre trained Video VAEs. +If you want to train video vae, you can refer to [README] (easyanimate/vae/README. md) in the video vae section. + +#### c. Video VAE training +If the data format is relative path during data preprocessing, please set ```scripts/train_t2iv.sh``` as follow. +``` +export DATASET_NAME="datasets/internal_datasets/" +export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.json" +``` + +If the data format is absolute path during data preprocessing, please set ```scripts/train_t2iv.sh``` as follow. ``` export DATASET_NAME="" export DATASET_META_NAME="/mnt/data/json_of_internal_datasets.json" diff --git a/README_zh-CN.md b/README_zh-CN.md index c3d07f8..56eefb7 100644 --- a/README_zh-CN.md +++ b/README_zh-CN.md @@ -1,7 +1,7 @@ # EasyAnimate | 视频与图片一体化生成基线方案。 -😊 EasyAnimate是一个用于生成长视频和图片 和 训练基于transformer的扩散生成器的repo。 +😊 EasyAnimate是一个用于生成高分辨率和长视频的端到端解决方案。我们可以训练基于转换器的扩散生成器,训练用于处理长视频的VAE,以及预处理元数据。 -😊 我们基于类SORA结构与DIT,使用transformer进行作为扩散器进行视频与图片生成。为了保证良好的拓展性,我们基于motion module构建了EasyAnimate,未来我们也会尝试更多的训练方案一提高效果。 +😊 我们基于类SORA结构与DIT,使用transformer进行作为扩散器进行视频与图片生成。我们基于motion module、u-vit和slice-vae构建了EasyAnimate,未来我们也会尝试更多的训练方案一提高效果。 😊 Welcome! @@ -39,7 +39,6 @@ EasyAnimate是一个基于transformer结构的pipeline,可用于生成AI图片 - 支持视频inpaint模型。 # Model zoo - EasyAnimateV2: | 名称 | 种类 | 存储空间 | 下载地址 | 描述 | |--|--|--|--|--| @@ -69,7 +68,11 @@ EasyAnimateV2: # 快速启动 ### 1. 云使用: AliyunDSW/Docker #### a. 通过阿里云 DSW -敬请期待。 +DSW 有免费 GPU 时间,用户可申请一次,申请后3个月内有效。 + +阿里云在[Freetier](https://free.aliyun.com/?product=9602825&crowd=enterprise&spm=5176.28055625.J_5831864660.1.e939154aRgha4e&scm=20140722.M_9974135.P_110.MO_1806-ID_9974135-MID_9974135-CID_30683-ST_8512-V_1)提供免费GPU时间,获取并在阿里云PAI-DSW中使用,5分钟内即可启动EasyAnimate + +[![DSW Notebook](images/dsw.png)](https://gallery.pai-ml.com/#/preview/deepLearning/cv/easyanimate) #### b. 通过docker 使用docker的情况下,请保证机器中已经正确安装显卡驱动与CUDA环境,然后以此执行以下命令: @@ -186,8 +189,13 @@ EasyAnimateV2: - 步骤3:根据页面选择生成模型,填入prompt、neg_prompt、guidance_scale和seed等,点击生成,等待生成结果,结果保存在sample文件夹中。 ### 2. 模型训练 +一个完整的EasyAnimate训练链路应该包括数据预处理、Video VAE训练、Video DiT训练。其中Video VAE训练是一个可选项,因为我们已经提供了训练好的Video VAE。 + +#### a. 数据预处理 我们给出了一个简单的demo通过图片数据训练lora模型,详情可以查看[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)。 +一个完整的长视频切分、清洗、描述的数据预处理链路可以参考video caption部分的[README](easyanimate/video_caption/README.md)进行。 + 如果期望训练一个文生图视频的生成模型,您需要以这种格式排列数据集。 ``` 📦 project/ @@ -200,7 +208,7 @@ EasyAnimateV2: │ └── 📄 json_of_internal_datasets.json ``` -json_of_internal_datasets.json是一个标准的json文件,如下所示: +json_of_internal_datasets.json是一个标准的json文件。json中的file_path可以被设置为相对路径,如下所示: ```json [ { @@ -216,17 +224,6 @@ json_of_internal_datasets.json是一个标准的json文件,如下所示: ..... ] ``` -json中的file_path可以被设置为相对路径。 - -然后,进入scripts/train_t2iv.sh进行设置。 -``` -export DATASET_NAME="datasets/internal_datasets/" -export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.json" - -... - -train_data_format="normal" -``` 你也可以将路径设置为绝对路径: ```json @@ -244,7 +241,24 @@ train_data_format="normal" ..... ] ``` -此时 scripts/train_t2iv.sh 可以被设置为如下: + +#### b. Video VAE训练 (可选) +Video VAE训练是一个可选项,因为我们已经提供了训练好的Video VAE。 + +如果想要进行训练,可以参考video vae部分的[README](easyanimate/vae/README.md)进行。 + +#### c. Video DiT训练 +如果数据预处理时,数据的格式为相对路径,则进入scripts/train_t2iv.sh进行如下设置。 +``` +export DATASET_NAME="datasets/internal_datasets/" +export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.json" + +... + +train_data_format="normal" +``` + +如果数据的格式为绝对路径,则进入scripts/train_t2iv.sh进行如下设置。 ``` export DATASET_NAME="" export DATASET_META_NAME="/mnt/data/json_of_internal_datasets.json" diff --git a/easyanimate/vae/README.md b/easyanimate/vae/README.md new file mode 100644 index 0000000..2eef328 --- /dev/null +++ b/easyanimate/vae/README.md @@ -0,0 +1,60 @@ +## Data preprocessing +After completing data preprocessing, we can obtain the following dataset: + +``` +📦 project/ +├── 📂 datasets/ +│ ├── 📂 internal_datasets/ +│ ├── 📂 videos/ +│ │ ├── 📄 00000001.mp4 +│ │ ├── 📄 00000001.jpg +│ │ └── 📄 ..... +│ └── 📄 json_of_internal_datasets.json +``` + +The json_of_internal_datasets.json is a standard JSON file. The file_path in the json can to be set as relative path, as shown in below: +```json +[ + { + "file_path": "videos/00000001.mp4", + "text": "A group of young men in suits and sunglasses are walking down a city street.", + "type": "video" + }, + { + "file_path": "train/00000001.jpg", + "text": "A group of young men in suits and sunglasses are walking down a city street.", + "type": "image" + }, + ..... +] +``` + +You can also set the path as absolute path as follow: +```json +[ + { + "file_path": "/mnt/data/videos/00000001.mp4", + "text": "A group of young men in suits and sunglasses are walking down a city street.", + "type": "video" + }, + { + "file_path": "/mnt/data/train/00000001.jpg", + "text": "A group of young men in suits and sunglasses are walking down a city street.", + "type": "image" + }, + ..... +] +``` + +## Train Video VAE +We need to set config in ```easyanimate/vae/configs/autoencoder``` at first. The default config is ```autoencoder_kl_32x32x4_slice.yaml```. We need to set the some params in yaml file. + +- ```data_json_path``` corresponds to the JSON file of the dataset. +- ```data_root``` corresponds to the root path of the dataset. If you want to use absolute path in json file, please delete this line. +- ```ckpt_path``` corresponds to the pretrained weights of the vae. +- ```gpus``` and num_nodes need to be set as the actual situation of your machine. + +The we run shell file as follow: +``` +sh scripts/train_vae.sh +``` \ No newline at end of file diff --git a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_mag.yaml b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_mag.yaml index b818b56..cfd0e0a 100644 --- a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_mag.yaml +++ b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_mag.yaml @@ -29,6 +29,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRTrain params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 128 degradation: pil_nearest video_size: 128 @@ -38,6 +39,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRValidation params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 128 degradation: pil_nearest video_size: 128 diff --git a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice.yaml b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice.yaml index 9239a9a..8fe607c 100644 --- a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice.yaml +++ b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice.yaml @@ -6,7 +6,7 @@ model: mini_batch_encoder: 8 mini_batch_decoder: 2 monitor: train/rec_loss - ckpt_path: models/Diffusion_Transformer/new_vae_1888/vae-share-decoder/diffusion_pytorch_model.safetensors + ckpt_path: models/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512/vae/diffusion_pytorch_model.safetensors down_block_types: ("SpatialDownBlock3D", "SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D",) up_block_types: ("SpatialUpBlock3D", "SpatialTemporalUpBlock3D", "SpatialTemporalUpBlock3D", @@ -32,6 +32,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRTrain params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 256 degradation: pil_nearest video_size: 256 @@ -41,6 +42,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRValidation params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 256 degradation: pil_nearest video_size: 256 diff --git a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_decoder_only.yaml b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_decoder_only.yaml index 4e06a87..32fa32b 100644 --- a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_decoder_only.yaml +++ b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_decoder_only.yaml @@ -7,7 +7,7 @@ model: mini_batch_encoder: 8 mini_batch_decoder: 2 monitor: train/rec_loss - ckpt_path: models/Diffusion_Transformer/new_vae_1888/vae-share-decoder/diffusion_pytorch_model.safetensors + ckpt_path: models/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512/vae/diffusion_pytorch_model.safetensors down_block_types: ("SpatialDownBlock3D", "SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D",) up_block_types: ("SpatialUpBlock3D", "SpatialTemporalUpBlock3D", "SpatialTemporalUpBlock3D", @@ -33,6 +33,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRTrain params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 256 degradation: pil_nearest video_size: 256 @@ -42,6 +43,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRValidation params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 256 degradation: pil_nearest video_size: 256 diff --git a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_t_downsample_8.yaml b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_t_downsample_8.yaml index ef01ae2..25f05f2 100644 --- a/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_t_downsample_8.yaml +++ b/easyanimate/vae/configs/autoencoder/autoencoder_kl_32x32x4_slice_t_downsample_8.yaml @@ -6,7 +6,7 @@ model: mini_batch_encoder: 8 mini_batch_decoder: 1 monitor: train/rec_loss - ckpt_path: models/Diffusion_Transformer/new_vae_1888/vae-share-decoder/diffusion_pytorch_model.safetensors + ckpt_path: models/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512/vae/diffusion_pytorch_model.safetensors down_block_types: ("SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D",) up_block_types: ("SpatialTemporalUpBlock3D", "SpatialTemporalUpBlock3D", "SpatialTemporalUpBlock3D", @@ -33,6 +33,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRTrain params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 256 degradation: pil_nearest video_size: 256 @@ -42,6 +43,7 @@ data: target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRValidation params: data_json_path: pretrain.json + data_root: /your_data_root # This is used in relative path size: 256 degradation: pil_nearest video_size: 256 diff --git a/easyanimate/vae/ldm/data/dataset_image_video.py b/easyanimate/vae/ldm/data/dataset_image_video.py index 0953d22..bd0c4f6 100644 --- a/easyanimate/vae/ldm/data/dataset_image_video.py +++ b/easyanimate/vae/ldm/data/dataset_image_video.py @@ -90,7 +90,7 @@ class ImageVideoDataset(Dataset): # If caught exception(timeout or others), try another index until successful and return. def __init__(self, size=None, video_size=128, video_len=25, degradation=None, downscale_f=4, random_crop=True, min_crop_f=0.25, max_crop_f=1., - s_t=None, slice_interval=None, + s_t=None, slice_interval=None, data_root=None ): """ Imagenet Superresolution Dataloader @@ -124,6 +124,7 @@ class ImageVideoDataset(Dataset): self.video_rescaler = albumentations.SmallestMaxSize(max_size=video_size, interpolation=cv2.INTER_AREA) self.video_len = video_len self.video_size = video_size + self.data_root = data_root self.pil_interpolation = False # gets reset later if incase interp_op is from pillow @@ -165,7 +166,10 @@ class ImageVideoDataset(Dataset): def __getitem__(self, i): @func_set_timeout(3) # time wait 3 seconds def get_video_item(example): - video_reader = VideoReader(example['file_path']) + if self.data_root is not None: + video_reader = VideoReader(os.path.join(self.data_root, example['file_path'])) + else: + video_reader = VideoReader(example['file_path']) video_length = len(video_reader) clip_length = min(video_length, (self.video_len - 1) * self.slice_interval + 1) @@ -221,7 +225,10 @@ class ImageVideoDataset(Dataset): while True: try: example = self.base[i] - image = Image.open(example['file_path']) + if self.data_root is not None: + image = Image.open(os.path.join(self.data_root, example['file_path'])) + else: + image = Image.open(example['file_path']) image = image.convert("RGB") image = np.array(image).astype(np.uint8)