Author SHA1 Message Date
zouxinyi0625 b3e4b5ffc7 update readme 2024-06-04 10:43:00 +08:00
bubbliiiing cd2117cdac update huggingface url 2024-06-04 10:13:27 +08:00
bubbliiiing ff9dfb852f update huggingface url 2024-06-04 10:07:55 +08:00
bubbliiiing a43fafaaba update CN README 2024-06-03 14:38:35 +08:00
bubbliiiing c562b3eb9c update 768 2024-06-03 14:25:37 +08:00
bubbliiiing 840ed19dc2 merge main 2024-06-03 13:11:42 +08:00
yunkchen 9cae83066e Update requirements.txt
Add torch version limit
2024-05-31 11:19:14 +08:00
zouxinyi0625 6e75e236b6 arxiv link 2024-05-31 10:50:19 +08:00
zouxinyi0625 cd64d31945 update requirements 2024-05-31 10:42:32 +08:00
zouxinyi0625 1e6cca5324 update vae readme 2024-05-31 10:31:09 +08:00
zouxinyi0625 3303184516 link fix 2024-05-30 20:34:50 +08:00
zouxinyi0625 93dbedec4f update readme 2024-05-30 20:28:48 +08:00
zouxinyi0625 fdf6fdc443 Merge branch 'v2' of https://github.com/aigc-apps/EasyAnimate into v2 2024-05-30 20:28:13 +08:00
zouxinyi0625 d832775526 update readme 2024-05-30 20:25:59 +08:00
bubbliiiing 10246bb426 update arxiv 2024-05-30 19:35:54 +08:00
bubbliiiing 874a17304b update gallery 2024-05-30 19:31:22 +08:00
bubbliiiing 81eee13bc9 Merge branch 'v2' of github.com:aigc-apps/EasyAnimate into v2 2024-05-30 17:26:20 +08:00
bubbliiiing a2524bbf26 delete IDDPM 2024-05-30 17:26:01 +08:00
zouxinyi0625 7a202c95e0 add arxiv 2024-05-30 16:46:06 +08:00
bubbliiiing 92fa867dcc update text box 2024-05-29 17:09:43 +08:00
bubbliiiing 4f908317aa provide example for video cut 2024-05-29 16:52:07 +08:00
bubbliiiing b10557c31e fix bug in validation while training 2024-05-29 15:02:07 +08:00
bubbliiiing 5e1041a104 update readme bans 2024-05-29 11:11:18 +08:00
bubbliiiing 35e134c2e2 update lots of readme 2024-05-29 10:52:17 +08:00
bubbliiiing bc3bde68e6 complete data preprocess pipeline. 2024-05-29 00:34:04 +08:00
bubbliiiing b790734afd merge main 2024-05-28 20:37:17 +08:00
bubbliiiing c321dc3d67 update fast api 2024-05-28 20:22:26 +08:00
bubbliiiing 96f619c1c3 update datasets loader 2024-05-27 13:54:41 +08:00
bubbliiiing e6336652aa update datasets loader 2024-05-27 13:54:29 +08:00
bubbliiiing f4286edee0 update EasyAnimateV2 2024-05-26 21:05:24 +08:00
187 changed files with 7597 additions and 42788 deletions
-25
View File
@@ -1,25 +0,0 @@
name: Publish to Comfy registry
on:
workflow_dispatch:
push:
branches:
- main
- master
paths:
- "pyproject.toml"
jobs:
publish-node:
name: Publish Custom Node to registry
runs-on: ubuntu-latest
if: ${{ github.repository_owner == 'aigc-apps' }}
steps:
- name: Check out code
uses: actions/checkout@v4
with:
submodules: true
- name: Publish Custom Node
uses: Comfy-Org/publish-node-action@main
with:
## Add your own personal access token to your Github Repository secrets and reference it here.
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
+1 -4
View File
@@ -2,13 +2,10 @@
models*
output*
samples*
datasets*
asset*
_*
datasets
__pycache__/
*.py[cod]
*$py.class
scripts_demo*
# C extensions
*.so
+14 -27
View File
@@ -1,4 +1,4 @@
FROM nvidia/cuda:12.1.0-cudnn8-devel-ubuntu22.04
FROM nvidia/cuda:11.8.0-devel-ubuntu22.04
ENV DEBIAN_FRONTEND noninteractive
RUN rm -r /etc/apt/sources.list.d/
@@ -6,46 +6,33 @@ RUN rm -r /etc/apt/sources.list.d/
RUN apt-get update -y && apt-get install -y \
libgl1 libglib2.0-0 google-perftools \
sudo wget git git-lfs vim tig pkg-config libcairo2-dev \
aria2 telnet curl net-tools iputils-ping jq \
python3-pip python-is-python3 python3.10-venv tzdata lsof zip tmux
RUN apt-get update && \
apt-get install -y software-properties-common && \
add-apt-repository ppa:ubuntuhandbook1/ffmpeg6 && \
apt-get update && \
apt-get install -y ffmpeg
telnet curl net-tools iputils-ping wget jq \
python3-pip python-is-python3 python3.10-venv tzdata lsof && \
rm -rf /var/lib/apt/lists/*
RUN pip3 install --upgrade pip -i https://mirrors.aliyun.com/pypi/simple/
# add all extensions
RUN apt-get update -y && apt-get install -y zip && \
rm -rf /var/lib/apt/lists/*
RUN pip install wandb tqdm GitPython==3.1.32 Pillow==9.5.0 setuptools --upgrade -i https://mirrors.aliyun.com/pypi/simple/
RUN pip install torch==2.4.0 torchvision==0.19.0 torchaudio==2.4.0 --index-url https://download.pytorch.org/whl/cu118
RUN pip install xformers==0.0.27.post2 --index-url https://download.pytorch.org/whl/cu118
# install vllm (video-caption)
RUN pip install vllm==0.6.3
# install requirements (video-caption)
WORKDIR /root/
COPY easyanimate/video_caption/requirements.txt /root/requirements-video_caption.txt
RUN pip install -r /root/requirements-video_caption.txt
RUN rm /root/requirements-video_caption.txt
# reinstall torch to keep compatible with xformers
RUN pip uninstall -qy torch torchvision && \
pip install torch==2.1.2 torchvision==0.16.2 torchaudio==2.1.2 --index-url https://download.pytorch.org/whl/cu118
RUN pip uninstall -qy xfromers && pip install xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118
RUN apt-get update && apt-get install -y aria2
RUN pip install -U http://eas-data.oss-cn-shanghai.aliyuncs.com/sdk/allspark-0.15-py2.py3-none-any.whl
RUN pip install https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/package/vllm-0.3.3%2Bcu118-cp310-cp310-manylinux1_x86_64.whl
RUN pip install pandas>=2.0.0 auto_gptq sglang[srt] func_timeout -i https://mirrors.aliyun.com/pypi/simple/
RUN pip install -e git+https://github.com/CompVis/taming-transformers.git@master#egg=taming-transformers
RUN pip install came-pytorch deepspeed pytorch_lightning==1.9.4 func_timeout -i https://mirrors.aliyun.com/pypi/simple/
RUN pip install pytorch_lightning==1.9.4 func_timeout -i https://mirrors.aliyun.com/pypi/simple/
# install requirements
RUN pip install bitsandbytes mamba-ssm causal-conv1d>=1.4.0 -i https://mirrors.aliyun.com/pypi/simple/
RUN pip install ipykernel -i https://mirrors.aliyun.com/pypi/simple/
COPY ./requirements.txt /root/requirements.txt
RUN pip install -r /root/requirements.txt -i https://mirrors.aliyun.com/pypi/simple/
RUN rm -rf /root/requirements.txt
# install package patches (video-caption)
COPY easyanimate/video_caption/package_patches/easyocr_detection_patched.py /usr/local/lib/python3.10/dist-packages/easyocr/detection.py
COPY easyanimate/video_caption/package_patches/vila_siglip_encoder_patched.py /usr/local/lib/python3.10/dist-packages/llava/model/multimodal_encoder/siglip_encoder.py
ENV PYTHONUNBUFFERED 1
ENV NVIDIA_DISABLE_REQUIRE 1
Executable → Regular
+129 -394
View File
@@ -1,44 +1,38 @@
# 📷 EasyAnimate | An End-to-End Solution for High-Resolution and Long Video Generation
😊 EasyAnimate is an end-to-end solution for generating high-resolution and long videos. We can train transformer based diffusion generators, train VAEs for processing long videos, and preprocess metadata.
😊 We use DIT and transformer as a diffuser for video and image generation.
😊 Based on Sora like structure and DIT, we use transformer as a diffuser for video generation. We built easyanimate based on motion module, u-vit and slice-vae. In the future, we will try more training programs to improve the effect.
😊 Welcome!
[![Arxiv Page](https://img.shields.io/badge/Arxiv-Page-red)](https://arxiv.org/abs/2405.18991)
[![Project Page](https://img.shields.io/badge/Project-Website-green)](https://easyanimate.github.io/)
[![Modelscope Studio](https://img.shields.io/badge/Modelscope-Studio-blue)](https://modelscope.cn/studios/PAI/EasyAnimate/summary)
[![Hugging Face Spaces](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-yellow)](https://huggingface.co/spaces/alibaba-pai/EasyAnimate)
[![Discord Page](https://img.shields.io/badge/Discord-Page-blue)](https://discord.gg/UzkpB4Bn)
English | [简体中文](./README_zh-CN.md) | [日本語](./README_ja-JP.md)
English | [简体中文](./README_zh-CN.md)
# Table of Contents
- [Table of Contents](#table-of-contents)
- [Introduction](#introduction)
- [Quick Start](#quick-start)
- [Video Result](#video-result)
- [How to use](#how-to-use)
- [Model zoo](#model-zoo)
- [Algorithm Detailed](#algorithm-detailed)
- [TODO List](#todo-list)
- [Contact Us](#contact-us)
- [Reference](#reference)
- [License](#license)
# Introduction
EasyAnimate is a pipeline based on the transformer architecture, designed for generating AI images and videos, and for training baseline models and Lora models for Diffusion Transformer. We support direct prediction from pre-trained EasyAnimate models, allowing for the generation of videos with various resolutions, approximately 6 seconds in length, at 8fps (EasyAnimateV5, 1 to 49 frames). Additionally, users can train their own baseline and Lora models for specific style transformations.
EasyAnimate is a pipeline based on the transformer architecture that can be used to generate AI photos and videos, train baseline models and Lora models for the Diffusion Transformer. We support making predictions directly from the pre-trained EasyAnimate model to generate videos of about different resolutions, 6 seconds with 24 fps (1 ~ 144 frames, in the future, we will support longer videos). Users are also supported to train their own baseline models and Lora models to perform certain style transformations.
We will support quick pull-ups from different platforms, refer to [Quick Start](#quick-start).
**New Features:**
- **Updated to version v5.1**, the Qwen2 VL is used as the text encoder, and Flow is used as the sampling method. It supports bilingual prediction in both Chinese and English. In addition to common controls such as Canny and Pose, it also supports trajectory control, camera control. [2025.01.21]
- Use reward backpropagation to train Lora and optimize the video, aligning it better with human preferences, detailes in [here](scripts/README_TRAIN_REWARD.md). EasyAnimateV5-7b is released now. [2024.11.27]
- **Updated to v5**, supporting video generation up to 1024x1024, 49 frames, 6s, 8fps, with expanded model scale to 12B, incorporating the MMDIT structure, and enabling control models with diverse inputs; supports bilingual predictions in Chinese and English. [2024.11.08]
- **Updated to v4**, allowing for video generation up to 1024x1024, 144 frames, 6s, 24fps; supports video generation from text, image, and video, with a single model handling resolutions from 512 to 1280; bilingual predictions in Chinese and English enabled. [2024.08.15]
- **Updated to v3**, supporting video generation up to 960x960, 144 frames, 6s, 24fps, from text and image. [2024.07.01]
- **ModelScope-Sora “Data Director” Creative Race** — The third Data-Juicer Big Model Data Challenge is now officially launched! Utilizing EasyAnimate as the base model, it explores the impact of data processing on model training. Visit the [competition website](https://tianchi.aliyun.com/competition/entrance/532219) for details. [2024.06.17]
- **Updated to v2**, supporting video generation up to 768x768, 144 frames, 6s, 24fps. [2024.05.26]
- **Code Created!** Now supporting Windows and Linux. [2024.04.12]
What's New:
- Updated to v2, supports a maximum of 144 frames (768x768, 6s, 24fps) for generation. [ 2024.05.26 ]
- Create Code! Support for Windows and Linux Now. [ 2024.04.12 ]
Function:
- [Data Preprocessing](#data-preprocess)
@@ -46,8 +40,11 @@ Function:
- [Train DiT](#dit-train)
- [Video Generation](#video-gen)
These are our generated results [GALLERY](scripts/Result_Gallery.md):
![Combine_512](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v2/Combine_512.jpg)
Our UI interface is as follows:
![ui](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/ui_v3.jpg)
![ui](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/ui.png)
# Quick Start
### 1. Cloud usage: AliyunDSW/Docker
@@ -56,16 +53,14 @@ DSW has free GPU time, which can be applied once by a user and is valid for 3 mo
Aliyun provide free GPU time in [Freetier](https://free.aliyun.com/?product=9602825&crowd=enterprise&spm=5176.28055625.J_5831864660.1.e939154aRgha4e&scm=20140722.M_9974135.P_110.MO_1806-ID_9974135-MID_9974135-CID_30683-ST_8512-V_1), get it and use in Aliyun PAI-DSW to start EasyAnimate within 5min!
[![DSW Notebook](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/dsw.png)](https://gallery.pai-ml.com/#/preview/deepLearning/cv/easyanimate_v5)
[![DSW Notebook](images/dsw.png)](https://gallery.pai-ml.com/#/preview/deepLearning/cv/easyanimate)
#### b. From ComfyUI
Our ComfyUI is as follows, please refer to [ComfyUI README](comfyui/README.md) for details.
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v3/comfyui_i2v.jpg)
#### c. From docker
#### b. From docker
If you are using docker, please make sure that the graphics card driver and CUDA environment have been installed correctly in your machine.
Then execute the following commands in this way:
EasyAnimateV2:
```
# pull image
docker pull mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
@@ -84,334 +79,107 @@ mkdir models/Diffusion_Transformer
mkdir models/Motion_Module
mkdir models/Personalized_Model
# Please use the hugginface link or modelscope link to download the EasyAnimateV5.1 model.
# https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP
# https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512.tar -O models/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512.tar
# https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh
# https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh
cd models/Diffusion_Transformer/
tar -xvf EasyAnimateV2-XL-2-512x512.tar
cd ../../
```
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
```
# pull image
docker pull mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
# enter image
docker run -it -p 7860:7860 --network host --gpus all --security-opt seccomp:unconfined --shm-size 200g mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
# clone code
git clone https://github.com/aigc-apps/EasyAnimate.git
# enter EasyAnimate's dir
cd EasyAnimate
# download weights
mkdir models/Diffusion_Transformer
mkdir models/Motion_Module
mkdir models/Personalized_Model
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Motion_Module/easyanimate_v1_mm.safetensors -O models/Motion_Module/easyanimate_v1_mm.safetensors
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait.safetensors -O models/Personalized_Model/easyanimate_portrait.safetensors
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait_lora.safetensors -O models/Personalized_Model/easyanimate_portrait_lora.safetensors
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/PixArt-XL-2-512x512.tar -O models/Diffusion_Transformer/PixArt-XL-2-512x512.tar
cd models/Diffusion_Transformer/
tar -xvf PixArt-XL-2-512x512.tar
cd ../../
```
</details>
### 2. Local install: Environment Check/Downloading/Installation
#### a. Environment Check
We have verified EasyAnimate execution on the following environment:
The detailed of Windows:
- OS: Windows 10
- python: python3.10 & python3.11
- pytorch: torch2.2.0
- CUDA: 11.8 & 12.1
- CUDNN: 8+
- GPU: Nvidia-3060 12G
The detailed of Linux:
- OS: Ubuntu 20.04, CentOS
- python: python3.10 & python3.11
- python: py3.10 & py3.11
- pytorch: torch2.2.0
- CUDA: 11.8 & 12.1
- CUDA: 11.8
- CUDNN: 8+
- GPU:Nvidia-V100 16G & Nvidia-A10 24G & Nvidia-A100 40G & Nvidia-A100 80G
- GPU: Nvidia-A10 24G & Nvidia-A100 40G & Nvidia-A100 80G
We need about 60GB available on disk (for saving weights), please check!
The video size for EasyAnimateV5.1-12B can be generated by different GPU Memory, including:
| GPU memory | 384x672x25 | 384x672x49 | 576x1008x25 | 576x1008x49 | 768x1344x25 | 768x1344x49 |
|------------|------------|------------|------------|------------|------------|------------|
| 16GB | 🧡 | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
The video size for EasyAnimateV5.1-7B can be generated by different GPU Memory, including:
| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | 🧡 | ⭕️ | ⭕️ | ❌ | ❌ |
| 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
✅ indicates it can run under "model_cpu_offload", 🧡 represents it can run under "model_cpu_offload_and_qfloat8", ⭕️ indicates it can run under "sequential_cpu_offload", ❌ means it can't run. Please note that running with sequential_cpu_offload will be slower.
Some GPUs that do not support torch.bfloat16, such as 2080ti and V100, require changing the weight_dtype in app.py and predict files to torch.float16 in order to run.
The generation time for EasyAnimateV5.1-12B using different GPUs over 25 steps is as follows:
| GPU | 384x672x25 | 384x672x49 | 576x1008x25 | 576x1008x49 | 768x1344x25 | 768x1344x49 |
|-----------|------------------|------------------|------------------|------------------|------------------|-----------------|
| A10 24GB | ~120s (4.8s/it) | ~240s (9.6s/it) | ~320s (12.7s/it) | ~750s (29.8s/it) | ❌ | ❌ |
| A100 80GB | ~45s (1.75s/it) | ~90s (3.7s/it) | ~120s (4.7s/it) | ~300s (11.4s/it) | ~265s (10.6s/it) | ~710s (28.3s/it) |
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
The video size for EasyAnimateV3 can be generated by different GPU Memory, including:
| GPU memory | 384x672x72 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
|------------|------------|-------------|-------------|--------------|-------------|--------------|
| 12GB | ⭕️ | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 16GB | ✅ | ✅ | ⭕️ | ⭕️ | ⭕️ | ❌ |
| 24GB | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
(⭕️) indicates it can run with low_gpu_memory_mode=True, but at a slower speed, and ❌ means it can't run.
</details>
#### b. Weights
We'd better place the [weights](#model-zoo) along the specified path:
EasyAnimateV5.1:
EasyAnimateV2:
```
📦 models/
├── 📂 Diffusion_Transformer/
│ ├── 📂 EasyAnimateV5.1-12b-zh-InP/
│ └── 📂 EasyAnimateV5.1-12b-zh/
│ └── 📂 EasyAnimateV2-XL-2-512x512/
├── 📂 Personalized_Model/
│ └── your trained trainformer model / your trained lora model (for UI load)
```
# Video Result
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
### Image to Video with EasyAnimateV5.1-12b-zh-InP
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/74a23109-f555-4026-a3d8-1ac27bb3884c" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/ab5aab27-fbd7-4f55-add9-29644125bde7" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/238043c2-cdbd-4288-9857-a273d96f021f" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/48881a0e-5513-4482-ae49-13a0ad7a2557" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/3e7aba7f-6232-4f39-80a8-6cfae968f38c" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/986d9f77-8dc3-45fa-bc9d-8b26023fffbc" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/7f62795a-2b3b-4c14-aeb1-1230cb818067" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b581df84-ade1-4605-a7a8-fd735ce3e222" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/eab1db91-1082-4de2-bb0a-d97fd25ceea1" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/3fda0e96-c1a8-4186-9c4c-043e11420f05" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4b53145d-7e98-493a-83c9-4ea4f5b58289" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/75f7935f-17a8-4e20-b24c-b61479cf07fc" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### Text to Video with EasyAnimateV5.1-12b-zh
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/8818dae8-e329-4b08-94fa-00d923f38fd2" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/d3e483c3-c710-47d2-9fac-89f732f2260a" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4dfa2067-d5d4-4741-a52c-97483de1050d" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/fb44c2db-82c6-427e-9297-97dcce9a4948" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/dc6b8eaf-f21b-4576-a139-0e10438f20e4" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b3f8fd5b-c5c8-44ee-9b27-49105a08fbff" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/a68ed61b-eed3-41d2-b208-5f039bf2788e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4e33f512-0126-4412-9ae8-236ff08bcd21" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### Control Video with EasyAnimateV5.1-12b-zh-Control
Trajectory Control:
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/bf3b8970-ca7b-447f-8301-72dfe028055b" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/63a7057b-573e-4f73-9d7b-8f8001245af4" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/090ac2f3-1a76-45cf-abe5-4e326113389b" width="100%" controls autoplay loop></video>
</td>
<tr>
</table>
Generic Control Video (Canny, Pose, Depth, etc.):
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/53002ce2-dd18-4d4f-8135-b6f68364cabd" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/fce43c0b-81fa-4ab2-9ca7-78d786f520e6" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b208b92c-5add-4ece-a200-3dbbe47b93c3" width="100%" controls autoplay loop></video>
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/3aec95d5-d240-49fb-a9e9-914446c7a4cf" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/60fa063b-5c1f-485f-b663-09bd6669de3f" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4adde728-8397-42f3-8a2a-23f7b39e9a1e" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### Camera Control with EasyAnimateV5.1-12b-zh-Control-Camera
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
Pan Up
</td>
<td>
Pan Left
</td>
<td>
Pan Right
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/a88f81da-e263-4038-a5b3-77b26f79719e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/e346c59d-7bca-4253-97fb-8cbabc484afb" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4de470d4-47b7-46e3-82d3-b714a2f6aef6" width="100%" controls autoplay loop></video>
</td>
<tr>
<td>
Pan Down
</td>
<td>
Pan Up + Pan Left
</td>
<td>
Pan Up + Pan Right
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/7a3fecc2-d41a-4de3-86cd-5e19aea34a0d" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/cb281259-28b6-448e-a76f-643c3465672e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/44faf5b6-d83c-4646-9436-971b2b9c7216" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
```
📦 models/
├── 📂 Diffusion_Transformer/
│ └── 📂 PixArt-XL-2-512x512/
├── 📂 Motion_Module/
│ └── 📄 easyanimate_v1_mm.safetensors
├── 📂 Personalized_Model/
│ ├── 📄 easyanimate_portrait.safetensors
│ └── 📄 easyanimate_portrait_lora.safetensors
```
</details>
# How to use
<h3 id="video-gen">1. Inference </h3>
#### a. Memory-Saving Options
Since EasyAnimateV5 and V5.1 have very large parameters, we need to consider memory-saving options to adapt to consumer-grade graphics cards. We provide GPU_memory_mode for each prediction file, allowing you to choose from model_cpu_offload, model_cpu_offload_and_qfloat8, or sequential_cpu_offload.
- model_cpu_offload means the entire model will move to the CPU after use, saving some memory.
- model_cpu_offload_and_qfloat8 means the entire model will move to the CPU after use and applies float8 quantization to the transformer model, saving more memory.
- sequential_cpu_offload means each layer of the model moves to CPU after use, which is slower but saves a lot of memory.
qfloat8 may reduce model performance but saves more memory. If memory is sufficient, it's recommended to use model_cpu_offload.
#### b. Via ComfyUI
For more details, see the [ComfyUI README](comfyui/README.md).
#### c. Run Python Files
#### a. Using Python Code
- Step 1: Download the corresponding [weights](#model-zoo) and place them in the models folder.
- Step 2: Use different files for predictions based on the weights and prediction goals.
- Text-to-Video:
- Modify the prompt, neg_prompt, guidance_scale, and seed in the predict_t2v.py file.
- Then run the predict_t2v.py file and wait for the results, which are stored in the samples/easyanimate-videos folder.
- Image-to-Video:
- Modify validation_image_start, validation_image_end, prompt, neg_prompt, guidance_scale, and seed in the predict_i2v.py file.
- validation_image_start is the starting image, and validation_image_end is the ending image of the video.
- Then run the predict_i2v.py file and wait for the results, which are stored in the samples/easyanimate-videos_i2v folder.
- Video-to-Video:
- Modify validation_video, validation_image_end, prompt, neg_prompt, guidance_scale, and seed in the predict_v2v.py file.
- validation_video is the reference video for video-to-video. You can run a demo with the following video: [Demo Video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/play_guitar.mp4)
- Then run the predict_v2v.py file and wait for the results, which are stored in samples/easyanimate-videos_v2v folder.
- Generic Control Video (Canny, Pose, Depth, etc.):
- Modify control_video, validation_image_end, prompt, neg_prompt, guidance_scale, and seed in the predict_v2v_control.py file.
- control_video is the control video for video generation, extracted using Canny, Pose, Depth, etc. You can run a demo with the following video: [Demo Video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1.1/pose.mp4)
- Then run the predict_v2v_control.py file and wait for the results, which are stored in samples/easyanimate-videos_v2v_control folder.
- Trajectory Control Video:
- Modify control_video, ref_image, validation_image_end, prompt, neg_prompt, guidance_scale, and seed in the predict_v2v_control.py file.
- control_video is the control video, and ref_image is the reference first frame image. You can run a demo with the following image and video: [Demo Image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/dog.png), [Demo Video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/trajectory_demo.mp4)
- Then run the predict_v2v_control.py file and wait for the results, which are stored in samples/easyanimate-videos_v2v_control folder.
- Interaction via ComfyUI is recommended.
- Camera Control Video:
- Modify control_video, ref_image, validation_image_end, prompt, neg_prompt, guidance_scale, and seed in the predict_v2v_control.py file.
- control_camera_txt is the control file for camera control video, and ref_image is the reference first frame image. You can run a demo with the following image and control file: [Demo Image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png), [Demo File (from CameraCtrl)](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/0a3b5fb184936a83.txt)
- Then run the predict_v2v_control.py file and wait for the results, which are stored in samples/easyanimate-videos_v2v_control folder.
- Interaction via ComfyUI is recommended.
- Step 3: To combine with other backbones and Lora trained by yourself, modify predict_t2v.py and lora_path accordingly in the predict_t2v.py file.
#### d. Via WebUI Interface
WebUI supports text-to-video, image-to-video, video-to-video, and control-based video generation (such as Canny, Pose, Depth, etc.).
- Step 2: Modify prompt, neg_prompt, guidance_scale, and seed in the predict_t2v.py file.
- Step 3: Run the predict_t2v.py file, wait for the generated results, and save the results in the samples/easyanimate-videos folder.
- Step 4: If you want to combine other backbones you have trained with Lora, modify the predict_t2v.py and Lora_path in predict_t2v.py depending on the situation.
#### b. Using webui
- Step 1: Download the corresponding [weights](#model-zoo) and place them in the models folder.
- Step 2: Run the app.py file to enter the Gradio page.
- Step 3: Choose the generation model from the page, fill in prompt, neg_prompt, guidance_scale, seed, etc., click generate, and wait for the results, which are stored in the sample folder.
- Step 2: Run the app.py file to enter the graph page.
- Step 3: Select the generated model based on the page, fill in prompt, neg_prompt, guidance_scale, and seed, click on generate, wait for the generated result, and save the result in the samples folder.
### 2. Model Training
A complete EasyAnimate training pipeline should include data preprocessing, Video VAE training, and Video DiT training. Among these, Video VAE training is optional because we have already provided a pre-trained Video VAE.
<h4 id="data-preprocess">a. data preprocessing</h4>
We provide two simple demos:
- Train a Lora model using image data. For more details, you can refer to the [wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora).
- Perform SFT model training using video data. For more details, you can refer to the [wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-SFT).
We have provided a simple demo of training the Lora model through image data, which can be found in the [wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora) for details.
A complete data preprocessing link for long video segmentation, cleaning, and description can refer to [README](./easyanimate/video_caption/README.md) in the video captions section.
@@ -421,9 +189,9 @@ If you want to train a text to image and video generation model. You need to arr
📦 project/
├── 📂 datasets/
│ ├── 📂 internal_datasets/
│ ├── 📂 train/
│ ├── 📂 videos/
│ │ ├── 📄 00000001.mp4
│ │ ├── 📄 00000002.jpg
│ │ ├── 📄 00000001.jpg
│ │ └── 📄 .....
│ └── 📄 json_of_internal_datasets.json
```
@@ -432,12 +200,12 @@ The json_of_internal_datasets.json is a standard JSON file. The file_path in the
```json
[
{
"file_path": "train/00000001.mp4",
"file_path": "videos/00000001.mp4",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "video"
},
{
"file_path": "train/00000002.jpg",
"file_path": "train/00000001.jpg",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "image"
},
@@ -469,25 +237,23 @@ If you want to train video vae, you can refer to [README](easyanimate/vae/README
<h4 id="dit-train">c. Video DiT training </h4>
If the data format is relative path during data preprocessing, please set ```scripts/train.sh``` as follow.
If the data format is relative path during data preprocessing, please set ```scripts/train_t2iv.sh``` as follow.
```
export DATASET_NAME="datasets/internal_datasets/"
export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.json"
```
If the data format is absolute path during data preprocessing, please set ```scripts/train.sh``` as follow.
If the data format is absolute path during data preprocessing, please set ```scripts/train_t2iv.sh``` as follow.
```
export DATASET_NAME=""
export DATASET_META_NAME="/mnt/data/json_of_internal_datasets.json"
```
Then, we run scripts/train.sh.
Then, we run scripts/train_t2iv.sh.
```sh
sh scripts/train.sh
sh scripts/train_t2iv.sh
```
For details on setting some parameters, please refer to [Readme Train](scripts/README_TRAIN.md) and [Readme Lora](scripts/README_TRAIN_LORA.md).
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
If you want to train EasyAnimateV1. Please switch to the git branch v1.
@@ -496,70 +262,12 @@ For details on setting some parameters, please refer to [Readme Train](scripts/R
# Model zoo
EasyAnimateV5.1:
7B:
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
EasyAnimateV2:
| Name | Type | Storage Space | Url | Hugging Face | Description |
|--|--|--|--|--|--|
| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, and trajectory control. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera) | Official video camera control weights, supporting direction generation control by inputting camera motion trajectories. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
12B:
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV5.1-12b-zh-InP | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-12b-zh-Control | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, and trajectory control. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-12b-zh-Control-Camera | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control-Camera) | Official video camera control weights, supporting direction generation control by inputting camera motion trajectories. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-12b-zh | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
<details>
<summary>(Obsolete) EasyAnimateV5:</summary>
7B:
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV5-7b-zh-InP | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-7b-zh-InP) | Official 7B image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports bilingual prediction in Chinese and English. |
| EasyAnimateV5-7b-zh | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-7b-zh) | Official 7B text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports bilingual prediction in Chinese and English. |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | The official reward backpropagation technology model optimizes the videos generated by EasyAnimateV5-12b to better match human preferences. |
12B:
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV5-12b-zh-InP | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports bilingual prediction in Chinese and English. |
| EasyAnimateV5-12b-zh-Control | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, etc. Supports video prediction at multiple resolutions (512, 768, 1024) and is trained with 49 frames at 8 frames per second. Bilingual prediction in Chinese and English is supported. |
| EasyAnimateV5-12b-zh | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports bilingual prediction in Chinese and English. |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | The official reward backpropagation technology model optimizes the videos generated by EasyAnimateV5-12b to better match human preferences. |
</details>
<details>
<summary>(Obsolete) EasyAnimateV4:</summary>
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV4-XL-2-InP | EasyAnimateV4 | Before extraction: 8.9 GB \/ After extraction: 14.0 GB |[🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV4-XL-2-InP)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV4-XL-2-InP)| | Our official graph-generated video model is capable of predicting videos at multiple resolutions (512, 768, 1024, 1280) and has been trained on 144 frames at a rate of 24 frames per second. |
</details>
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV3-XL-2-InP-512x512 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-512x512) | EasyAnimateV3 official weights for 512x512 text and image to video resolution. Training with 144 frames and fps 24 |
| EasyAnimateV3-XL-2-InP-768x768 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-768x768) | EasyAnimateV3 official weights for 768x768 text and image to video resolution. Training with 144 frames and fps 24 |
| EasyAnimateV3-XL-2-InP-960x960 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-960x960) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-960x960) | EasyAnimateV3 official weights for 960x960 text and image to video resolution. Training with 144 frames and fps 24 |
</details>
<details>
<summary>(Obsolete) EasyAnimateV2:</summary>
| Name | Type | Storage Space | Url | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|--|
| EasyAnimateV2-XL-2-512x512 | EasyAnimateV2 | 16.2GB | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV2-XL-2-512x512)| EasyAnimateV2 official weights for 512x512 resolution. Training with 144 frames and fps 24 |
| EasyAnimateV2-XL-2-768x768 | EasyAnimateV2 | 16.2GB | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV2-XL-2-768x768)| EasyAnimateV2 official weights for 768x768 resolution. Training with 144 frames and fps 24 |
| easyanimatev2_minimalism_lora.safetensors | Lora of Pixart | 485.1MB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimatev2_minimalism_lora.safetensors)| - | - | A lora training with a specifial type images. Images can be downloaded from [Url](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v2/Minimalism.zip). |
</details>
| EasyAnimateV2-XL-2-512x512.tar | EasyAnimateV2 | 16.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-512x512) | EasyAnimateV2 official weights for 512x512 resolution. Training with 144 frames and fps 24 |
| EasyAnimateV2-XL-2-768x768.tar | EasyAnimateV2 | 16.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV2-XL-2-768x768.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-768x768) | EasyAnimateV2 official weights for 768x768 resolution. Training with 144 frames and fps 24 |
| easyanimatev2_minimalism_lora.safetensors | Lora of Pixart | 485.1MB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimatev2_minimalism_lora.safetensors) | - | A lora training with a specifial type images. Images can be downloaded from [Url](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/webui/Minimalism.zip). |
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
@@ -577,8 +285,42 @@ EasyAnimateV5.1:
| easyanimate_portrait_lora.safetensors | Lora of Pixart | 654.0MB | [download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait_lora.safetensors)| Training with internal portrait datasets |
</details>
# Algorithm Detailed
### 1. Data Preprocessing
**Video Cut**
For long video cut, EasyAnimate utilizes PySceneDetect to identify scene changes within the video and performs scene cutting based on certain threshold values to ensure consistency in the themes of the video segments. After cutting, we only keep segments with lengths ranging from 3 to 10 seconds for model training.
**Video Cleaning and Description**
Following SVD's data preparation process, EasyAnimate provides a simple yet effective data processing pipeline for high-quality data filtering and labeling. It also supports distributed processing to accelerate the speed of data preprocessing. The overall process is as follows:
- Duration filtering: Analyze the basic information of the video to filter out low-quality videos that are short in duration or low in resolution.
- Aesthetic filtering: Filter out videos with poor content (blurry, dim, etc.) by calculating the average aesthetic score of uniformly distributed 4 frames.
- Text filtering: Use easyocr to calculate the text proportion of middle frames to filter out videos with a large proportion of text.
- Motion filtering: Calculate interframe optical flow differences to filter out videos that move too slowly or too quickly.
- Text description: Recaption video frames using videochat2 and vila. PAI is also developing a higher quality video recaption model, which will be released for use as soon as possible.
### 2. Model Architecture
We have adopted [PixArt-alpha](https://github.com/PixArt-alpha/PixArt-alpha) as the base model and modified the VAE and DiT model structures on this basis to better support video generation. The overall structure of EasyAnimate is as follows:
The diagram below outlines the pipeline of EasyAnimate. It includes the Text Encoder, Video VAE (video encoder and decoder), and Diffusion Transformer (DiT). The T5 Encoder is used as the text encoder. Other components are detailed in the sections below.
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/pipeline_v2.jpg" alt="ui" style="zoom:50%;" />
To introduce feature information along the temporal axis, EasyAnimate incorporates the Motion Module to achieve the expansion from 2D images to 3D videos. For better generation effects, it jointly finetunes the Backbone together with the Motion Module, thereby achieving image generation and video generation within a single Pipeline.
Additionally, referencing U-ViT, it introduces a skip connection structure into EasyAnimate to further optimize deeper features by incorporating shallow features. A fully connected layer is also zero-initialized for each skip connection structure, allowing it to be applied as a plug-in module to previously trained and well-performing DiTs.
Moreover, it proposes Slice VAE, which addresses the memory difficulties encountered by MagViT when dealing with long and large videos, while also achieving greater compression in the temporal dimension during video encoding and decoding stages compared to MagViT.
For more details, please refer to [arxiv](https://arxiv.org/abs/2405.18991).
# TODO List
- Support model with larger params.
- Support model with larger resolution.
- Support video inpaint model.
# Contact Us
1. Use Dingding to search group 77450006752 or Scan to join
@@ -589,20 +331,13 @@ EasyAnimateV5.1:
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/group/person.jpg" alt="Person" width="30%"/>
# Reference
- CogVideo: https://github.com/THUDM/CogVideo/
- Flux: https://github.com/black-forest-labs/flux
- magvit: https://github.com/google-research/magvit
- PixArt: https://github.com/PixArt-alpha/PixArt-alpha
- Open-Sora-Plan: https://github.com/PKU-YuanGroup/Open-Sora-Plan
- Open-Sora: https://github.com/hpcaitech/Open-Sora
- Animatediff: https://github.com/guoyww/AnimateDiff
- HunYuan DiT: https://github.com/tencent/HunyuanDiT
- ComfyUI-KJNodes: https://github.com/kijai/ComfyUI-KJNodes
- ComfyUI-EasyAnimateWrapper: https://github.com/kijai/ComfyUI-EasyAnimateWrapper
- ComfyUI-CameraCtrl-Wrapper: https://github.com/chaojie/ComfyUI-CameraCtrl-Wrapper
- CameraCtrl: https://github.com/hehao13/CameraCtrl
- DragAnything: https://github.com/showlab/DragAnything
# License
This project is licensed under the [Apache License (Version 2.0)](https://github.com/modelscope/modelscope/blob/master/LICENSE).
This project is licensed under the [Apache License (Version 2.0)](https://github.com/modelscope/modelscope/blob/master/LICENSE).
-593
View File
@@ -1,593 +0,0 @@
# 📷 EasyAnimate | 高解像度および長時間動画生成のためのエンドツーエンドソリューション
😊 EasyAnimateは、高解像度および長時間動画を生成するためのエンドツーエンドソリューションです。トランスフォーマーベースの拡散生成器をトレーニングし、長時間動画を処理するためのVAEをトレーニングし、メタデータを前処理することができます。
😊 DITをベースに、トランスフォーマーを拡散器として使用して動画や画像を生成します。
😊 ようこそ!
[![Arxiv Page](https://img.shields.io/badge/Arxiv-Page-red)](https://arxiv.org/abs/2405.18991)
[![Project Page](https://img.shields.io/badge/Project-Website-green)](https://easyanimate.github.io/)
[![Modelscope Studio](https://img.shields.io/badge/Modelscope-Studio-blue)](https://modelscope.cn/studios/PAI/EasyAnimate/summary)
[![Hugging Face Spaces](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-yellow)](https://huggingface.co/spaces/alibaba-pai/EasyAnimate)
[![Discord Page](https://img.shields.io/badge/Discord-Page-blue)](https://discord.gg/UzkpB4Bn)
English | [简体中文](./README_zh-CN.md) | 日本語
# 目次
- [目次](#目次)
- [紹介](#紹介)
- [クイックスタート](#クイックスタート)
- [ビデオ結果](#ビデオ結果)
- [使い方](#使い方)
- [モデルズー](#モデルズー)
- [TODOリスト](#todoリスト)
- [お問い合わせ](#お問い合わせ)
- [参考文献](#参考文献)
- [ライセンス](#ライセンス)
# 紹介
EasyAnimateは、トランスフォーマーアーキテクチャに基づいたパイプラインで、AI画像および動画の生成、Diffusion TransformerのベースラインモデルおよびLoraモデルのトレーニングに使用されます。事前トレーニング済みのEasyAnimateモデルから直接予測を行い、さまざまな解像度で約6秒間、8fpsの動画を生成できます(EasyAnimateV5、1〜49フレーム)。さらに、ユーザーは特定のスタイル変換のために独自のベースラインおよびLoraモデルをトレーニングできます。
異なるプラットフォームからのクイックプルアップをサポートします。詳細は[クイックスタート](#クイックスタート)を参照してください。
**新機能:**
- **バージョンv5.1に更新**、Qwen2 VLがテキストエンコーダーとして使用され、Flowがサンプリング方法として使用されます。中国語と英語の両方でバイリンガル予測をサポートしています。CannyやPoseといった一般的なコントロールに加えて、軌道制御やカメラ制御もサポートしています。[2025.01.21]
- インセンティブ逆伝播を使用してLoraを訓練し、人間の好みに合うようにビデオを最適化します。詳細は、[ここ](scripts/README _ train _ REVARD.md)を参照してください。EasyAnimateV 5-7 bがリリースされました。[2024.11.27]
- **v5に更新**、1024x1024までの動画生成をサポート、49フレーム、6秒、8fps、モデルスケールを12Bに拡張、MMDIT構造を組み込み、さまざまな入力を持つ制御モデルをサポート。中国語と英語のバイリンガル予測をサポート。[2024.11.08]
- **v4に更新**、1024x1024までの動画生成をサポート、144フレーム、6秒、24fps、テキスト、画像、動画からの動画生成をサポート、512から1280までの解像度を単一モデルで処理。中国語と英語のバイリンガル予測をサポート。[2024.08.15]
- **v3に更新**、960x960までの動画生成をサポート、144フレーム、6秒、24fps、テキストと画像からの動画生成をサポート。[2024.07.01]
- **ModelScope-Sora “データディレクター” クリエイティブレース** — 第三回Data-Juicerビッグモデルデータチャレンジが正式に開始されました!EasyAnimateをベースモデルとして使用し、データ処理がモデルトレーニングに与える影響を探ります。詳細は[競技ウェブサイト](https://tianchi.aliyun.com/competition/entrance/532219)をご覧ください。[2024.06.17]
- **v2に更新**、768x768までの動画生成をサポート、144フレーム、6秒、24fps。[2024.05.26]
- **コード作成!** 現在、WindowsおよびLinuxをサポート。[2024.04.12]
機能:
- [データ前処理](#data-preprocess)
- [VAEのトレーニング](#vae-train)
- [DiTのトレーニング](#dit-train)
- [動画生成](#video-gen)
私たちのUIインターフェースは次のとおりです:
![ui](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/ui_v3.jpg)
# クイックスタート
### 1. クラウド使用: AliyunDSW/Docker
#### a. AliyunDSWから
DSWには無料のGPU時間があり、ユーザーは一度申請でき、申請後3ヶ月間有効です。
Aliyunは[Freetier](https://free.aliyun.com/?product=9602825&crowd=enterprise&spm=5176.28055625.J_5831864660.1.e939154aRgha4e&scm=20140722.M_9974135.P_110.MO_1806-ID_9974135-MID_9974135-CID_30683-ST_8512-V_1)で無料のGPU時間を提供しており、取得してAliyun PAI-DSWで使用し、5分以内にEasyAnimateを開始できます!
[![DSW Notebook](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/dsw.png)](https://gallery.pai-ml.com/#/preview/deepLearning/cv/easyanimate_v5)
#### b. ComfyUIから
私たちのComfyUIは次のとおりです。詳細は[ComfyUI README](comfyui/README.md)を参照してください。
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v3/comfyui_i2v.jpg)
#### c. Dockerから
Dockerを使用している場合は、マシンにグラフィックスカードドライバとCUDA環境が正しくインストールされていることを確認してください。
次のコマンドを実行します:
```
# イメージをプル
docker pull mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
# イメージに入る
docker run -it -p 7860:7860 --network host --gpus all --security-opt seccomp:unconfined --shm-size 200g mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
# コードをクローン
git clone https://github.com/aigc-apps/EasyAnimate.git
# EasyAnimateのディレクトリに入る
cd EasyAnimate
# 重みをダウンロード
mkdir models/Diffusion_Transformer
mkdir models/Motion_Module
mkdir models/Personalized_Model
# EasyAnimateV5モデルをダウンロードするには、hugginfaceリンクまたはmodelscopeリンクを使用してください。
# I2Vモデル
# https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP
# https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP
# T2Vモデル
# https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh
# https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh
```
### 2. ローカルインストール: 環境チェック/ダウンロード/インストール
#### a. 環境チェック
次の環境でEasyAnimateの実行を確認しました:
Windowsの詳細:
- OS: Windows 10
- python: python3.10 & python3.11
- pytorch: torch2.2.0
- CUDA: 11.8 & 12.1
- CUDNN: 8+
- GPU: Nvidia-3060 12G
Linuxの詳細:
- OS: Ubuntu 20.04, CentOS
- python: python3.10 & python3.11
- pytorch: torch2.2.0
- CUDA: 11.8 & 12.1
- CUDNN: 8+
- GPU:Nvidia-V100 16G & Nvidia-A10 24G & Nvidia-A100 40G & Nvidia-A100 80G
ディスクに約60GBの空き容量が必要です(重みを保存するため)、確認してください!
EasyAnimateV5.1-12Bのビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください:
| GPUメモリ |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
EasyAnimateV5.1-7Bのビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください:
| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | 🧡 | ⭕️ | ⭕️ | ❌ | ❌ |
| 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
✅ は"model_cpu_offload"の条件で実行可能であることを示し、🧡は"model_cpu_offload_and_qfloat8"の条件で実行可能を示し、⭕️ は"sequential_cpu_offload"の条件では実行可能であることを示しています。❌は実行できないことを示します。sequential_cpu_offloadにより実行する場合は遅くなります。
一部のGPU(例:2080ti、V100)はtorch.bfloat16をサポートしていないため、app.pyおよびpredictファイル内のweight_dtypeをtorch.float16に変更する必要があります。
EasyAnimateV5-12Bは異なるGPUで25ステップ生成する時間は次の通りです:
| GPU |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| A10 24GB |約120秒 (4.8s/it)|約240秒 (9.6s/it)|約320秒 (12.7s/it)|約750秒 (29.8s/it)| ❌ | ❌ |
| A100 80GB |約45秒 (1.75s/it)|約90秒 (3.7s/it)|約120秒 (4.7s/it)|約300秒 (11.4s/it)|約265秒 (10.6s/it)| 約710秒 (28.3s/it)|
<details>
<summary>(廃止予定) EasyAnimateV3:</summary>
EasyAnimateV3のビデオサイズは異なるGPUメモリにより生成できます。以下の表をご覧ください:
| GPUメモリ | 384x672x72 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
|----------|----------|----------|----------|----------|----------|----------|
| 12GB | ⭕️ | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 16GB | ✅ | ✅ | ⭕️ | ⭕️ | ⭕️ | ❌ |
| 24GB | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
(⭕️) はlow_gpu_memory_mode=Trueの条件で実行可能であるが、速度が遅くなることを示しています。また、❌は実行できないことを示します。
</details>
#### b. 重み
[重み](#model-zoo)を指定されたパスに配置することをお勧めします:
EasyAnimateV5:
```
📦 models/
├── 📂 Diffusion_Transformer/
│ ├── 📂 EasyAnimateV5.1-12b-zh-InP/
│ └── 📂 EasyAnimateV5.1-12b-zh/
├── 📂 Personalized_Model/
│ └── あなたのトレーニング済みのトランスフォーマーモデル / あなたのトレーニング済みのLoraモデル(UIロード用)
```
# ビデオ結果
### Image to Video with EasyAnimateV5.1-12b-zh-InP
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/74a23109-f555-4026-a3d8-1ac27bb3884c" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/ab5aab27-fbd7-4f55-add9-29644125bde7" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/238043c2-cdbd-4288-9857-a273d96f021f" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/48881a0e-5513-4482-ae49-13a0ad7a2557" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/3e7aba7f-6232-4f39-80a8-6cfae968f38c" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/986d9f77-8dc3-45fa-bc9d-8b26023fffbc" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/7f62795a-2b3b-4c14-aeb1-1230cb818067" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b581df84-ade1-4605-a7a8-fd735ce3e222" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/eab1db91-1082-4de2-bb0a-d97fd25ceea1" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/3fda0e96-c1a8-4186-9c4c-043e11420f05" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4b53145d-7e98-493a-83c9-4ea4f5b58289" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/75f7935f-17a8-4e20-b24c-b61479cf07fc" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### Text to Video with EasyAnimateV5.1-12b-zh
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/8818dae8-e329-4b08-94fa-00d923f38fd2" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/d3e483c3-c710-47d2-9fac-89f732f2260a" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4dfa2067-d5d4-4741-a52c-97483de1050d" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/fb44c2db-82c6-427e-9297-97dcce9a4948" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/dc6b8eaf-f21b-4576-a139-0e10438f20e4" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b3f8fd5b-c5c8-44ee-9b27-49105a08fbff" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/a68ed61b-eed3-41d2-b208-5f039bf2788e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4e33f512-0126-4412-9ae8-236ff08bcd21" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### Control Video with EasyAnimateV5.1-12b-zh-Control
Trajectory Control:
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/bf3b8970-ca7b-447f-8301-72dfe028055b" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/63a7057b-573e-4f73-9d7b-8f8001245af4" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/090ac2f3-1a76-45cf-abe5-4e326113389b" width="100%" controls autoplay loop></video>
</td>
<tr>
</table>
Generic Control Video (Canny, Pose, Depth, etc.):
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/53002ce2-dd18-4d4f-8135-b6f68364cabd" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/fce43c0b-81fa-4ab2-9ca7-78d786f520e6" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b208b92c-5add-4ece-a200-3dbbe47b93c3" width="100%" controls autoplay loop></video>
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/3aec95d5-d240-49fb-a9e9-914446c7a4cf" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/60fa063b-5c1f-485f-b663-09bd6669de3f" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4adde728-8397-42f3-8a2a-23f7b39e9a1e" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### Camera Control with EasyAnimateV5.1-12b-zh-Control-Camera
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
Pan Up
</td>
<td>
Pan Left
</td>
<td>
Pan Right
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/a88f81da-e263-4038-a5b3-77b26f79719e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/e346c59d-7bca-4253-97fb-8cbabc484afb" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4de470d4-47b7-46e3-82d3-b714a2f6aef6" width="100%" controls autoplay loop></video>
</td>
<tr>
<td>
Pan Down
</td>
<td>
Pan Up + Pan Left
</td>
<td>
Pan Up + Pan Right
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/7a3fecc2-d41a-4de3-86cd-5e19aea34a0d" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/cb281259-28b6-448e-a76f-643c3465672e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/44faf5b6-d83c-4646-9436-971b2b9c7216" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
# 使い方
<h3 id="video-gen">1. 推論 </h3>
#### a、メモリ節約策
EasyAnimateV5およびV5.1のパラメータが非常に大きいため、消費者向けグラフィックスカードに適応させるためにメモリの節約策を考慮する必要があります。各予測ファイルにはGPU_memory_modeを提供しており、model_cpu_offload、model_cpu_offload_and_qfloat8、sequential_cpu_offloadから選択することができます。
- model_cpu_offloadは、使用後にモデル全体がCPUに移動することを示し、メモリの一部を節約できます。
- model_cpu_offload_and_qfloat8は、使用後にモデル全体がCPUに移動し、トランスフォーマーモデルをfloat8に量子化することを示し、さらに多くのメモリを節約できます。
- sequential_cpu_offloadは、使用後に各レイヤーが順次CPUに移動することを示し、速度は遅くなりますが、大量のメモリを節約できます。
qfloat8はモデルの性能を低下させますが、さらに多くのメモリを節約できます。メモリが十分にある場合は、model_cpu_offloadを使用することをお勧めします。
#### b、ComfyUIを使用する
詳細は[ComfyUI README](comfyui/README.md)をご覧ください。
#### c、pythonファイルを実行する
- ステップ1:対応する[重み](#model-zoo)をダウンロードし、modelsフォルダに入れます。
- ステップ2:異なる重みと予測目標に応じて異なるファイルを使用して予測を行います。
- テキストからビデオの生成:
- predict_t2v.pyファイルでprompt、neg_prompt、guidance_scale、seedを変更します。
- 次にpredict_t2v.pyファイルを実行し、生成結果を待ちます。結果はsamples/easyanimate-videosフォルダに保存されます。
- 画像からビデオの生成:
- predict_i2v.pyファイルでvalidation_image_start、validation_image_end、prompt、neg_prompt、guidance_scale、seedを変更します。
- validation_image_startはビデオの開始画像、validation_image_endはビデオの終了画像です。
- 次にpredict_i2v.pyファイルを実行し、生成結果を待ちます。結果はsamples/easyanimate-videos_i2vフォルダに保存されます。
- ビデオからビデオの生成:
- predict_v2v.pyファイルでvalidation_video、validation_image_end、prompt、neg_prompt、guidance_scale、seedを変更します。
- validation_videoはビデオの参照ビデオです。以下のビデオを使用してデモを実行できます:[デモビデオ](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/play_guitar.mp4)
- 次にpredict_v2v.pyファイルを実行し、生成結果を待ちます。結果はsamples/easyanimate-videos_v2vフォルダに保存されます。
- 通常のコントロールビデオ生成(Canny、Pose、Depthなど):
- predict_v2v_control.pyファイルでcontrol_video、validation_image_end、prompt、neg_prompt、guidance_scale、seedを変更します。
- control_videoはCanny、Pose、Depthなどのフィルタを適用した後のビデオです。以下のビデオを使用してデモを実行できます:[デモビデオ](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1.1/pose.mp4)
- 次にpredict_v2v_control.pyファイルを実行し、生成結果を待ちます。結果はsamples/easyanimate-videos_v2v_controlフォルダに保存されます。
- トラジェクトリーコントロールビデオ:
- predict_v2v_control.pyファイルでcontrol_video、ref_image、validation_image_end、prompt、neg_prompt、guidance_scale、seedを変更します。
- control_videoはトラジェクトリーコントロールビデオのコントロールビデオ、ref_imageは参照の初期フレーム画像です。以下の画像とコントロールビデオを使用してデモを実行できます:[デモ画像](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/dog.png)、[デモビデオ](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/trajectory_demo.mp4)
- 次にpredict_v2v_control.pyファイルを実行し、生成結果を待ちます。結果はsamples/easyanimate-videos_v2v_controlフォルダに保存されます。
- 交互利用にComfyUIの使用を推奨します。
- カメラコントロールビデオ:
- predict_v2v_control.pyファイルでcontrol_video、ref_image、validation_image_end、prompt、neg_prompt、guidance_scale、seedを変更します。
- control_camera_txtはカメラコントロールビデオのコントロールファイル、ref_imageは参照の初期フレーム画像です。以下の画像とコントロールビデオを使用してデモを実行できます:[デモ画像](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png)、[デモファイル(CameraCtrlから)](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/0a3b5fb184936a83.txt)
- 次にpredict_v2v_control.pyファイルを実行し、生成結果を待ちます。結果はsamples/easyanimate-videos_v2v_controlフォルダに保存されます。
- 交互利用にComfyUIの使用を推奨します。
- ステップ3:他のトレーニング済みバックボーンとLoraを組み合わせたい場合、predict_t2v.pyでpredict_t2v.pyとlora_pathを適宜変更してください。
#### d、UIインターフェイスを使用する
webuiはテキストからビデオ、画像からビデオ、ビデオからビデオ、および通常のコントロールビデオ(Canny、Pose、Depthなど)の生成をサポートしています。
- ステップ1:対応する[重み](#model-zoo)をダウンロードし、modelsフォルダに入れます。
- ステップ2:app.pyファイルを実行し、gradioページに入ります。
- ステップ3:ページで生成モデルを選択し、prompt、neg_prompt、guidance_scale、seedなどを入力して生成をクリックし、生成結果を待ちます。結果はsampleフォルダに保存されます。
### 2. モデルトレーニング
完全なEasyAnimateトレーニングパイプラインには、データ前処理、Video VAEトレーニング、およびVideo DiTトレーニングが含まれる必要があります。これらの中で、Video VAEトレーニングはオプションです。すでにトレーニング済みのVideo VAEを提供しているためです。
<h4 id="data-preprocess">a. データ前処理</h4>
私たちは2つの簡単なデモを提供します:
- 画像データを使用してLoraモデルを訓練します。詳細はこちらの[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)をご覧ください。
- 動画データを使用してSFTモデルを訓練します。詳細はこちらの[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-SFT)をご覧ください。
長時間動画のセグメンテーション、クリーニング、および説明のための完全なデータ前処理リンクは、ビデオキャプションセクションの[README](./easyanimate/video_caption/README.md)を参照してください。
テキストから画像および動画生成モデルをトレーニングする場合は、データセットを次の形式で配置する必要があります。
```
📦 project/
├── 📂 datasets/
│ ├── 📂 internal_datasets/
│ ├── 📂 train/
│ │ ├── 📄 00000001.mp4
│ │ ├── 📄 00000002.jpg
│ │ └── 📄 .....
│ └── 📄 json_of_internal_datasets.json
```
json_of_internal_datasets.jsonは標準のJSONファイルです。json内のfile_pathは相対パスとして設定できます。以下のように:
```json
[
{
"file_path": "train/00000001.mp4",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "video"
},
{
"file_path": "train/00000002.jpg",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "image"
},
.....
]
```
パスを絶対パスとして設定することもできます:
```json
[
{
"file_path": "/mnt/data/videos/00000001.mp4",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "video"
},
{
"file_path": "/mnt/data/train/00000001.jpg",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "image"
},
.....
]
```
<h4 id="vae-train">b. Video VAEトレーニング(オプション)</h4>
Video VAEトレーニングはオプションです。すでにトレーニング済みのVideo VAEを提供しているためです。
Video VAEをトレーニングする場合は、ビデオVAEセクションの[README](easyanimate/vae/README.md)を参照してください。
<h4 id="dit-train">c. Video DiTトレーニング </h4>
データ前処理時にデータ形式が相対パスの場合、```scripts/train.sh```を次のように設定します。
```
export DATASET_NAME="datasets/internal_datasets/"
export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.json"
```
データ前処理時にデータ形式が絶対パスの場合、```scripts/train.sh```を次のように設定します。
```
export DATASET_NAME=""
export DATASET_META_NAME="/mnt/data/json_of_internal_datasets.json"
```
次に、scripts/train.shを実行します。
```sh
sh scripts/train.sh
```
一部のパラメータの設定の詳細については、[Readme Train](scripts/README_TRAIN.md)および[Readme Lora](scripts/README_TRAIN_LORA.md)を参照してください。
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
EasyAnimateV1をトレーニングする場合は、gitブランチv1に切り替えてください。
</details>
# モデルズー
12B:
| 名前 | タイプ | ストレージスペース | Hugging Face | モデルスコープ | 説明 |
|--|--|--|--|--|--|
| EasyAnimateV5.1-12b-zh-InP | EasyAnimateV5.1 | 39 GB | [🤗リンク](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP) | [😄リンク](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP) | 公式の画像からビデオへの変換用の重み。支持多解像度(512、768、1024)的ビデオ予測、49フレームで毎秒8フレームの訓練、多言語予測をサポート |
| EasyAnimateV5.1-12b-zh-Control | EasyAnimateV5.1 | 39 GB | [🤗リンク](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control) | [😄リンク](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control) | 公式のビデオ制御用の重み。Canny、Depth、Pose、MLSD、および軌道制御などのさまざまな制御条件をサポートします。支持多解像度(512、768、1024)的ビデオ予測、49フレームで毎秒8フレームの訓練、多言語予測をサポート |
| EasyAnimateV5.1-12b-zh-Control-Camera | EasyAnimateV5.1 | 39 GB | [🤗リンク](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control-Camera) | [😄リンク](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control-Camera) | 公式のビデオカメラ制御用の重み。カメラの動きの軌跡を入力することで方向生成を制御します。支持多解像度(512、768、1024)的ビデオ予測、49フレームで毎秒8フレームの訓練、多言語予測をサポート |
| EasyAnimateV5.1-12b-zh | EasyAnimateV5.1 | 39 GB | [🤗リンク](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh) | [😄リンク](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh) | 公式のテキストからビデオへの変換用の重み。支持多解像度(512、768、1024)的ビデオ予測、49フレームで毎秒8フレームの訓練、多言語予測をサポート |
<details>
<summary>(Obsolete) EasyAnimateV5:</summary>
7B:
| 名前 | 種類 | ストレージスペース | Hugging Face | Model Scope | 説明 |
|--|--|--|--|--|--|
| EasyAnimateV5-7b-zh-InP | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-7b-zh-InP) | 公式の画像から動画への重み。複数の解像度(512、768、1024)での動画予測をサポートし、49フレーム、毎秒8フレームでトレーニングされ、中国語と英語のバイリンガル予測をサポートします。 |
| EasyAnimateV5-7b-zh | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-7b-zh) | 公式のテキストから動画への重み。複数の解像度(512、768、1024)での動画予測をサポートし、49フレーム、毎秒8フレームでトレーニングされ、中国語と英語のバイリンガル予測をサポートします。 |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | 公式インバース伝播技術モデルによるEasyAnimateV 5-12 b生成ビデオの最適化によるヒト選好の最適化|
12B:
| 名前 | 種類 | ストレージスペース | Hugging Face | Model Scope | 説明 |
|--|--|--|--|--|--|
| EasyAnimateV5-12b-zh-InP | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-InP) | 公式の画像から動画への重み。複数の解像度(512、768、1024)での動画予測をサポートし、49フレーム、毎秒8フレームでトレーニングされ、中国語と英語のバイリンガル予測をサポートします。 |
| EasyAnimateV5-12b-zh-Control | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-Control) | 公式の動画制御重み。Canny、Depth、Pose、MLSDなどのさまざまな制御条件をサポートします。複数の解像度(512、768、1024)での動画予測をサポートし、49フレーム、毎秒8フレームでトレーニングされ、中国語と英語のバイリンガル予測をサポートします。 |
| EasyAnimateV5-12b-zh | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh) | 公式のテキストから動画への重み。複数の解像度(512、768、1024)での動画予測をサポートし、49フレーム、毎秒8フレームでトレーニングされ、中国語と英語のバイリンガル予測をサポートします。 |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | 公式インバース伝播技術モデルによるEasyAnimateV 5-12 b生成ビデオの最適化によるヒト選好の最適化|
</details>
<details>
<summary>(Obsolete) EasyAnimateV4:</summary>
| 名前 | 種類 | ストレージスペース | Hugging Face | Model Scope | 説明 |
|--|--|--|--|--|--|
| EasyAnimateV4-XL-2-InP | EasyAnimateV4 | 解凍前: 8.9 GB / 解凍後: 14.0 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV4-XL-2-InP)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV4-XL-2-InP) | 公式のグラフ生成動画モデル。複数の解像度(512、768、1024、1280)での動画予測をサポートし、144フレーム、毎秒24フレームでトレーニングされています。 |
</details>
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
| 名前 | 種類 | ストレージスペース | Hugging Face | Model Scope | 説明 |
|--|--|--|--|--|--|
| EasyAnimateV3-XL-2-InP-512x512 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-512x512) | EasyAnimateV3公式の512x512テキストおよび画像から動画への重み。144フレーム、毎秒24フレームでトレーニングされています。 |
| EasyAnimateV3-XL-2-InP-768x768 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-768x768) | EasyAnimateV3公式の768x768テキストおよび画像から動画への重み。144フレーム、毎秒24フレームでトレーニングされています。 |
| EasyAnimateV3-XL-2-InP-960x960 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-960x960) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-960x960) | EasyAnimateV3公式の960x960テキストおよび画像から動画への重み。144フレーム、毎秒24フレームでトレーニングされています。 |
</details>
<details>
<summary>(Obsolete) EasyAnimateV2:</summary>
| 名前 | 種類 | ストレージスペース | URL | Hugging Face | Model Scope | 説明 |
|--|--|--|--|--|--|--|
| EasyAnimateV2-XL-2-512x512 | EasyAnimateV2 | 16.2GB | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV2-XL-2-512x512) | EasyAnimateV2公式の512x512解像度の重み。144フレーム、毎秒24フレームでトレーニングされています。 |
| EasyAnimateV2-XL-2-768x768 | EasyAnimateV2 | 16.2GB | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV2-XL-2-768x768) | EasyAnimateV2公式の768x768解像度の重み。144フレーム、毎秒24フレームでトレーニングされています。 |
| easyanimatev2_minimalism_lora.safetensors | Lora of Pixart | 485.1MB | [ダウンロード](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimatev2_minimalism_lora.safetensors) | - | - | 特定のタイプの画像でトレーニングされたLora。画像は[URL](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v2/Minimalism.zip)からダウンロードできます。 |
</details>
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
### 1、モーション重み
| 名前 | 種類 | ストレージスペース | URL | 説明 |
|--|--|--|--|--|
| easyanimate_v1_mm.safetensors | モーションモジュール | 4.1GB | [ダウンロード](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Motion_Module/easyanimate_v1_mm.safetensors) | 80フレーム、毎秒12フレームでトレーニングされています。 |
### 2、その他の重み
| 名前 | 種類 | ストレージスペース | URL | 説明 |
|--|--|--|--|--|
| PixArt-XL-2-512x512.tar | Pixart | 11.4GB | [ダウンロード](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/PixArt-XL-2-512x512.tar)| Pixart-Alpha公式の重み。 |
| easyanimate_portrait.safetensors | Pixartのチェックポイント | 2.3GB | [ダウンロード](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait.safetensors) | 内部のポートレートデータセットでトレーニングされています。 |
| easyanimate_portrait_lora.safetensors | PixartのLora | 654.0MB | [ダウンロード](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait_lora.safetensors)| 内部のポートレートデータセットでトレーニングされています。 |
</details>
# TODOリスト
- より大きなパラメータを持つモデルをサポートします。
# お問い合わせ
1. Dingdingを使用してグループ77450006752を検索するか、スキャンして参加します。
2. WeChatグループに参加するには画像をスキャンするか、期限切れの場合はこの学生を友達として追加して招待します。
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/group/dd.png" alt="ding group" width="30%"/>
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/group/wechat.jpg" alt="Wechat group" width="30%"/>
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/group/person.jpg" alt="Person" width="30%"/>
# 参考文献
- CogVideo: https://github.com/THUDM/CogVideo/
- Flux: https://github.com/black-forest-labs/flux
- magvit: https://github.com/google-research/magvit
- PixArt: https://github.com/PixArt-alpha/PixArt-alpha
- Open-Sora-Plan: https://github.com/PKU-YuanGroup/Open-Sora-Plan
- Open-Sora: https://github.com/hpcaitech/Open-Sora
- Animatediff: https://github.com/guoyww/AnimateDiff
- HunYuan DiT: https://github.com/tencent/HunyuanDiT
- ComfyUI-KJNodes: https://github.com/kijai/ComfyUI-KJNodes
- ComfyUI-EasyAnimateWrapper: https://github.com/kijai/ComfyUI-EasyAnimateWrapper
- ComfyUI-CameraCtrl-Wrapper: https://github.com/chaojie/ComfyUI-CameraCtrl-Wrapper
- CameraCtrl: https://github.com/hehao13/CameraCtrl
- DragAnything: https://github.com/showlab/DragAnything
# ライセンス
このプロジェクトは[Apache License (Version 2.0)](https://github.com/modelscope/modelscope/blob/master/LICENSE)の下でライセンスされています。
Executable → Regular
+126 -387
View File
@@ -1,43 +1,36 @@
# EasyAnimate | 高分辨率长视频生成的端到端解决方案
😊 EasyAnimate是一个用于生成高分辨率和长视频的端到端解决方案。我们可以训练基于转换器的扩散生成器,训练用于处理长视频的VAE,以及预处理元数据。
😊 我们基于DIT,使用transformer进行作为扩散器进行视频与图片生成。
😊 我们基于类SORA结构与DIT,使用transformer进行作为扩散器进行视频与图片生成。我们基于motion module、u-vit和slice-vae构建了EasyAnimate,未来我们也会尝试更多的训练方案一提高效果。
😊 Welcome!
[![Arxiv Page](https://img.shields.io/badge/Arxiv-Page-red)](https://arxiv.org/abs/2405.18991)
[![Project Page](https://img.shields.io/badge/Project-Website-green)](https://easyanimate.github.io/)
[![Modelscope Studio](https://img.shields.io/badge/Modelscope-Studio-blue)](https://modelscope.cn/studios/PAI/EasyAnimate/summary)
[![Hugging Face Spaces](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-yellow)](https://huggingface.co/spaces/alibaba-pai/EasyAnimate)
[![Discord Page](https://img.shields.io/badge/Discord-Page-blue)](https://discord.gg/UzkpB4Bn)
[English](./README.md) | 简体中文 | [日本語](./README_ja-JP.md)
[English](./README.md) | 简体中文
# 目录
- [目录](#目录)
- [简介](#简介)
- [快速启动](#快速启动)
- [视频作品](#视频作品)
- [如何使用](#如何使用)
- [模型地址](#模型地址)
- [算法细节](#算法细节)
- [未来计划](#未来计划)
- [联系我们](#联系我们)
- [参考文献](#参考文献)
- [许可证](#许可证)
# 简介
EasyAnimate是一个基于transformer结构的pipeline,可用于生成AI图片与视频、训练Diffusion Transformer的基线模型与Lora模型,我们支持从已经训练好的EasyAnimate模型直接进行预测,生成不同分辨率,6秒左右、fps8的视频(EasyAnimateV5.1,1 ~ 49帧),也支持用户训练自己的基线模型与Lora模型,进行一定的风格变换。
EasyAnimate是一个基于transformer结构的pipeline,可用于生成AI图片与视频、训练Diffusion Transformer的基线模型与Lora模型,我们支持从已经训练好的EasyAnimate模型直接进行预测,生成不同分辨率,6秒左右、fps24的视频(1 ~ 144帧, 未来会支持更长的视频),也支持用户训练自己的基线模型与Lora模型,进行一定的风格变换。
我们会逐渐支持从不同平台快速启动,请参阅 [快速启动](#快速启动)。
新特性:
- 更新到v5.1版本,应用Qwen2 VL作为文本编码器,支持多语言预测,使用Flow作为采样方式,除去常见控制如Canny、Pose外,还支持轨迹控制,相机控制等。[ 2025.01.21 ]
- 使用奖励反向传播来训练Lora并优化视频,使其更好地符合人类偏好,详细信息请参见[此处](scripts/README_train_REVARD.md)。EasyAnimateV5-7b现已发布。[ 2024.11.27 ]
- 更新到v5版本,最大支持1024x1024,49帧, 6s, 8fps视频生成,拓展模型规模到12B,应用MMDIT结构,支持不同输入的控制模型,支持中文与英文双语预测。[ 2024.11.08 ]
- 更新到v4版本,最大支持1024x1024,144帧, 6s, 24fps视频生成,支持文、图、视频生视频,单个模型可支持512到1280任意分辨率,支持中文与英文双语预测。[ 2024.08.15 ]
- 更新到v3版本,最大支持960x960,144帧,6s, 24fps视频生成,支持文与图生视频模型。[ 2024.07.01 ]
- ModelScope-Sora“数据导演”创意竞速——第三届Data-Juicer大模型数据挑战赛已经正式启动!其使用EasyAnimate作为基础模型,探究数据处理对于模型训练的作用。立即访问[竞赛官网](https://tianchi.aliyun.com/competition/entrance/532219),了解赛事详情。[ 2024.06.17 ]
- 更新到v2版本,最大支持768x768,144帧,6s, 24fps视频生成。[ 2024.05.26 ]
- 更新到v2版本,最大支持144帧(768x768, 6s, 24fps)生成。[ 2024.05.26 ]
- 创建代码!现在支持 Windows 和 Linux。[ 2024.04.12 ]
功能概览:
@@ -46,8 +39,12 @@ EasyAnimate是一个基于transformer结构的pipeline,可用于生成AI图片
- [训练DiT](#dit-train)
- [模型生成](#video-gen)
这些是我们的生成结果 [GALLERY](scripts/Result_Gallery.md):
![Combine_512](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v2/Combine_512.jpg)
我们的ui界面如下:
![ui](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/ui_v3.jpg)
![ui](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/ui.png)
# 快速启动
### 1. 云使用: AliyunDSW/Docker
@@ -56,14 +53,12 @@ DSW 有免费 GPU 时间,用户可申请一次,申请后3个月内有效。
阿里云在[Freetier](https://free.aliyun.com/?product=9602825&crowd=enterprise&spm=5176.28055625.J_5831864660.1.e939154aRgha4e&scm=20140722.M_9974135.P_110.MO_1806-ID_9974135-MID_9974135-CID_30683-ST_8512-V_1)提供免费GPU时间,获取并在阿里云PAI-DSW中使用,5分钟内即可启动EasyAnimate
[![DSW Notebook](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/dsw.png)](https://gallery.pai-ml.com/#/preview/deepLearning/cv/easyanimate_v5)
[![DSW Notebook](images/dsw.png)](https://gallery.pai-ml.com/#/preview/deepLearning/cv/easyanimate)
#### b. 通过ComfyUI
我们的ComfyUI界面如下,具体查看[ComfyUI README](comfyui/README.md)。
![workflow graph](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v3/comfyui_i2v.jpg)
#### c. 通过docker
#### b. 通过docker
使用docker的情况下,请保证机器中已经正确安装显卡驱动与CUDA环境,然后以此执行以下命令:
EasyAnimateV2:
```
# pull image
docker pull mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
@@ -82,319 +77,98 @@ mkdir models/Diffusion_Transformer
mkdir models/Motion_Module
mkdir models/Personalized_Model
# Please use the hugginface link or modelscope link to download the EasyAnimateV5.1 model.
# https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP
# https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512.tar -O models/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512.tar
# https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh
# https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh
cd models/Diffusion_Transformer/
tar -xvf EasyAnimateV2-XL-2-512x512.tar
cd ../../
```
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
```
# 拉取镜像
docker pull mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
# 进入镜像
docker run -it -p 7860:7860 --network host --gpus all --security-opt seccomp:unconfined --shm-size 200g mybigpai-public-registry.cn-beijing.cr.aliyuncs.com/easycv/torch_cuda:easyanimate
# clone 代码
git clone https://github.com/aigc-apps/EasyAnimate.git
# 进入EasyAnimate文件夹
cd EasyAnimate
# 下载权重
mkdir models/Diffusion_Transformer
mkdir models/Motion_Module
mkdir models/Personalized_Model
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Motion_Module/easyanimate_v1_mm.safetensors -O models/Motion_Module/easyanimate_v1_mm.safetensors
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait.safetensors -O models/Personalized_Model/easyanimate_portrait.safetensors
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait_lora.safetensors -O models/Personalized_Model/easyanimate_portrait_lora.safetensors
wget https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/PixArt-XL-2-512x512.tar -O models/Diffusion_Transformer/PixArt-XL-2-512x512.tar
cd models/Diffusion_Transformer/
tar -xvf PixArt-XL-2-512x512.tar
cd ../../
```
</details>
### 2. 本地安装: 环境检查/下载/安装
#### a. 环境检查
我们已验证EasyAnimate可在以下环境中执行:
Windows 的详细信息:
- 操作系统 Windows 10
- python: python3.10 & python3.11
- pytorch: torch2.2.0
- CUDA: 11.8 & 12.1
- CUDNN: 8+
- GPU: Nvidia-3060 12G
我们已验证EasyPhoto可在以下环境中执行:
Linux 的详细信息:
- 操作系统 Ubuntu 20.04, CentOS
- python: python3.10 & python3.11
- pytorch: torch2.2.0
- CUDA: 11.8 & 12.1
- CUDA: 11.8
- CUDNN: 8+
- GPU:Nvidia-V100 16G & Nvidia-A10 24G & Nvidia-A100 40G & Nvidia-A100 80G
- GPU: Nvidia-A10 24G & Nvidia-A100 40G & Nvidia-A100 80G
我们需要大约 60GB 的可用磁盘空间,请检查!
EasyAnimateV5.1-12B的视频大小可以由不同的GPU Memory生成,包括:
| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 24GB | 🧡 | 🧡 | 🧡 | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
EasyAnimateV5.1-7B的视频大小可以由不同的GPU Memory生成,包括:
| GPU memory |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| 16GB | 🧡 | 🧡 | ⭕️ | ⭕️ | ❌ | ❌ |
| 24GB | ✅ | ✅ | ✅ | 🧡 | 🧡 | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
✅ 表示它可以在"model_cpu_offload"的情况下运行,🧡代表它可以在"model_cpu_offload_and_qfloat8"的情况下运行,⭕️ 表示它可以在"sequential_cpu_offload"的情况下运行,❌ 表示它无法运行。请注意,使用sequential_cpu_offload运行会更慢。
有一些不支持torch.bfloat16的卡型,如2080ti、V100,需要将app.py、predict文件中的weight_dtype修改为torch.float16才可以运行。
EasyAnimateV5.1-12B使用不同GPU在25个steps中的生成时间如下:
| GPU |384x672x25|384x672x49|576x1008x25|576x1008x49|768x1344x25|768x1344x49|
|----------|----------|----------|----------|----------|----------|----------|
| A10 24GB |约120秒 (4.8s/it)|约240秒 (9.6s/it)|约320秒 (12.7s/it)| 约750秒 (29.8s/it)| ❌ | ❌ |
| A100 80GB |约45秒 (1.75s/it)|约90秒 (3.7s/it)|约120秒 (4.7s/it)|约300秒 (11.4s/it)|约265秒 (10.6s/it)| 约710秒 (28.3s/it)|
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
EasyAnimateV3的视频大小可以由不同的GPU Memory生成,包括:
| GPU memory | 384x672x72 | 384x672x144 | 576x1008x72 | 576x1008x144 | 720x1280x72 | 720x1280x144 |
|----------|----------|----------|----------|----------|----------|----------|
| 12GB | ⭕️ | ⭕️ | ⭕️ | ⭕️ | ❌ | ❌ |
| 16GB | ✅ | ✅ | ⭕️ | ⭕️ | ⭕️ | ❌ |
| 24GB | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ |
| 40GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
| 80GB | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ |
(⭕️) 表示它可以在low_gpu_memory_mode=True的情况下运行,但速度较慢,同时❌ 表示它无法运行。
</details>
#### b. 权重放置
我们最好将[权重](#model-zoo)按照指定路径进行放置:
EasyAnimateV5.1:
EasyAnimateV2:
```
📦 models/
├── 📂 Diffusion_Transformer/
│ ├── 📂 EasyAnimateV5.1-12b-zh-InP/
│ └── 📂 EasyAnimateV5.1-12b-zh/
│ └── 📂 EasyAnimateV2-XL-2-512x512/
├── 📂 Personalized_Model/
│ └── your trained trainformer model / your trained lora model (for UI load)
```
# 视频作品
### 图生视频 EasyAnimateV5.1-12b-zh-InP
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/74a23109-f555-4026-a3d8-1ac27bb3884c" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/ab5aab27-fbd7-4f55-add9-29644125bde7" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/238043c2-cdbd-4288-9857-a273d96f021f" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/48881a0e-5513-4482-ae49-13a0ad7a2557" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/3e7aba7f-6232-4f39-80a8-6cfae968f38c" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/986d9f77-8dc3-45fa-bc9d-8b26023fffbc" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/7f62795a-2b3b-4c14-aeb1-1230cb818067" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b581df84-ade1-4605-a7a8-fd735ce3e222" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/eab1db91-1082-4de2-bb0a-d97fd25ceea1" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/3fda0e96-c1a8-4186-9c4c-043e11420f05" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4b53145d-7e98-493a-83c9-4ea4f5b58289" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/75f7935f-17a8-4e20-b24c-b61479cf07fc" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### 文生视频 EasyAnimateV5.1-12b-zh
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/8818dae8-e329-4b08-94fa-00d923f38fd2" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/d3e483c3-c710-47d2-9fac-89f732f2260a" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4dfa2067-d5d4-4741-a52c-97483de1050d" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/fb44c2db-82c6-427e-9297-97dcce9a4948" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/dc6b8eaf-f21b-4576-a139-0e10438f20e4" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b3f8fd5b-c5c8-44ee-9b27-49105a08fbff" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/a68ed61b-eed3-41d2-b208-5f039bf2788e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4e33f512-0126-4412-9ae8-236ff08bcd21" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### 控制生视频 EasyAnimateV5.1-12b-zh-Control
轨迹控制
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/bf3b8970-ca7b-447f-8301-72dfe028055b" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/63a7057b-573e-4f73-9d7b-8f8001245af4" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/090ac2f3-1a76-45cf-abe5-4e326113389b" width="100%" controls autoplay loop></video>
</td>
<tr>
</table>
普通控制生视频(Canny、Pose、Depth等)
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
<video src="https://github.com/user-attachments/assets/53002ce2-dd18-4d4f-8135-b6f68364cabd" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/fce43c0b-81fa-4ab2-9ca7-78d786f520e6" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/b208b92c-5add-4ece-a200-3dbbe47b93c3" width="100%" controls autoplay loop></video>
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/3aec95d5-d240-49fb-a9e9-914446c7a4cf" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/60fa063b-5c1f-485f-b663-09bd6669de3f" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4adde728-8397-42f3-8a2a-23f7b39e9a1e" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
### 相机镜头控制 EasyAnimateV5.1-12b-zh-Control-Camera
<table border="0" style="width: 100%; text-align: left; margin-top: 20px;">
<tr>
<td>
Pan Up
</td>
<td>
Pan Left
</td>
<td>
Pan Right
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/a88f81da-e263-4038-a5b3-77b26f79719e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/e346c59d-7bca-4253-97fb-8cbabc484afb" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/4de470d4-47b7-46e3-82d3-b714a2f6aef6" width="100%" controls autoplay loop></video>
</td>
<tr>
<td>
Pan Down
</td>
<td>
Pan Up + Pan Left
</td>
<td>
Pan Up + Pan Right
</td>
<tr>
<td>
<video src="https://github.com/user-attachments/assets/7a3fecc2-d41a-4de3-86cd-5e19aea34a0d" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/cb281259-28b6-448e-a76f-643c3465672e" width="100%" controls autoplay loop></video>
</td>
<td>
<video src="https://github.com/user-attachments/assets/44faf5b6-d83c-4646-9436-971b2b9c7216" width="100%" controls autoplay loop></video>
</td>
</tr>
</table>
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
```
📦 models/
├── 📂 Diffusion_Transformer/
│ └── 📂 PixArt-XL-2-512x512/
├── 📂 Motion_Module/
│ └── 📄 easyanimate_v1_mm.safetensors
├── 📂 Personalized_Model/
│ ├── 📄 easyanimate_portrait.safetensors
│ └── 📄 easyanimate_portrait_lora.safetensors
```
</details>
# 如何使用
<h3 id="video-gen">1. 生成 </h3>
#### a、显存节省方案
由于EasyAnimateV5和V5.1的参数非常大,我们需要考虑显存节省方案,以节省显存适应消费级显卡。我们给每个预测文件都提供了GPU_memory_mode,可以在model_cpu_offload,model_cpu_offload_and_qfloat8,sequential_cpu_offload中进行选择。
- model_cpu_offload代表整个模型在使用后会进入cpu,可以节省部分显存。
- model_cpu_offload_and_qfloat8代表整个模型在使用后会进入cpu,并且对transformer模型进行了float8的量化,可以节省更多的显存。
- sequential_cpu_offload代表模型的每一层在使用后会进入cpu,速度较慢,节省大量显存。
qfloat8会降低模型的性能,但可以节省更多的显存。如果显存足够,推荐使用model_cpu_offload。
#### b、通过comfyui
具体查看[ComfyUI README](comfyui/README.md)。
#### c、运行python文件
#### a. 视频生成
##### i、运行python文件
- 步骤1:下载对应[权重](#model-zoo)放入models文件夹。
- 步骤2:根据不同的权重与预测目标使用不同的文件进行预测。
- 文生视频:
- 使用predict_t2v.py文件中修改prompt、neg_prompt、guidance_scale和seed。
- 而后运行predict_t2v.py文件,等待生成结果,结果保存在samples/easyanimate-videos文件夹中。
- 图生视频:
- 使用predict_i2v.py文件中修改validation_image_start、validation_image_end、prompt、neg_prompt、guidance_scale和seed。
- validation_image_start是视频的开始图片,validation_image_end是视频的结尾图片。
- 而后运行predict_i2v.py文件,等待生成结果,结果保存在samples/easyanimate-videos_i2v文件夹中。
- 视频生视频:
- 使用predict_v2v.py文件中修改validation_video、validation_image_end、prompt、neg_prompt、guidance_scale和seed。
- validation_video是视频生视频的参考视频。您可以使用以下视频运行演示:[演示视频](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/play_guitar.mp4)
- 而后运行predict_v2v.py文件,等待生成结果,结果保存在samples/easyanimate-videos_v2v文件夹中。
- 普通控制生视频(Canny、Pose、Depth等):
- 使用predict_v2v_control.py文件中修改control_video、validation_image_end、prompt、neg_prompt、guidance_scale和seed。
- control_video是控制生视频的控制视频,是使用Canny、Pose、Depth等算子提取后的视频。您可以使用以下视频运行演示:[演示视频](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1.1/pose.mp4)
- 而后运行predict_v2v_control.py文件,等待生成结果,结果保存在samples/easyanimate-videos_v2v_control文件夹中。
- 轨迹控制视频:
- 使用predict_v2v_control.py文件中修改control_video、ref_image、validation_image_end、prompt、neg_prompt、guidance_scale和seed。
- control_video是轨迹控制视频的控制视频,ref_image是参考的首帧图片。您可以使用以下图片和控制视频运行演示:[演示图像](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/dog.png),[演示视频](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/trajectory_demo.mp4)
- 而后运行predict_v2v_control.py文件,等待生成结果,结果保存在samples/easyanimate-videos_v2v_control文件夹中。
- 推荐使用ComfyUI进行交互。
- 相机控制视频:
- 使用predict_v2v_control.py文件中修改control_video、ref_image、validation_image_end、prompt、neg_prompt、guidance_scale和seed。
- control_camera_txt是相机控制视频的控制文件,ref_image是参考的首帧图片。您可以使用以下图片和控制视频运行演示:[演示图像](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png),[演示文件(来自于CameraCtrl)](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/0a3b5fb184936a83.txt)
- 而后运行predict_v2v_control.py文件,等待生成结果,结果保存在samples/easyanimate-videos_v2v_control文件夹中。
- 推荐使用ComfyUI进行交互。
- 步骤3:如果想结合自己训练的其他backbone与Lora,则看情况修改predict_t2v.py中的predict_t2v.py和lora_path。
#### d、通过ui界面
webui支持文生视频、图生视频、视频生视频和普通控制生视频(Canny、Pose、Depth等)
- 步骤2:在predict_t2v.py文件中修改prompt、neg_prompt、guidance_scale和seed。
- 步骤3:运行predict_t2v.py文件,等待生成结果,结果保存在samples/easyanimate-videos文件夹中。
- 步骤4:如果想结合自己训练的其他backbone与Lora,则看情况修改predict_t2v.py中的predict_t2v.py和lora_path。
##### ii、通过ui界面
- 步骤1:下载对应[权重](#model-zoo)放入models文件夹。
- 步骤2:运行app.py文件,进入gradio页面。
- 步骤3:根据页面选择生成模型,填入prompt、neg_prompt、guidance_scale和seed等,点击生成,等待生成结果,结果保存在sample文件夹中。
@@ -403,10 +177,7 @@ webui支持文生视频、图生视频、视频生视频和普通控制生视频
一个完整的EasyAnimate训练链路应该包括数据预处理、Video VAE训练、Video DiT训练。其中Video VAE训练是一个可选项,因为我们已经提供了训练好的Video VAE。
<h4 id="data-preprocess">a.数据预处理</h4>
我们给出了两个简单的demo:
- 通过图片数据训练lora模型,详情可以查看[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)。
- 通过视频数据进行SFT模型,详情可以查看[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-SFT)。
我们给出了一个简单的demo通过图片数据训练lora模型,详情可以查看[wiki](https://github.com/aigc-apps/EasyAnimate/wiki/Training-Lora)。
一个完整的长视频切分、清洗、描述的数据预处理链路可以参考video caption部分的[README](easyanimate/video_caption/README.md)进行。
@@ -415,9 +186,9 @@ webui支持文生视频、图生视频、视频生视频和普通控制生视频
📦 project/
├── 📂 datasets/
│ ├── 📂 internal_datasets/
│ ├── 📂 train/
│ ├── 📂 videos/
│ │ ├── 📄 00000001.mp4
│ │ ├── 📄 00000002.jpg
│ │ ├── 📄 00000001.jpg
│ │ └── 📄 .....
│ └── 📄 json_of_internal_datasets.json
```
@@ -426,12 +197,12 @@ json_of_internal_datasets.json是一个标准的json文件。json中的file_path
```json
[
{
"file_path": "train/00000001.mp4",
"file_path": "videos/00000001.mp4",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "video"
},
{
"file_path": "train/00000002.jpg",
"file_path": "train/00000001.jpg",
"text": "A group of young men in suits and sunglasses are walking down a city street.",
"type": "image"
},
@@ -462,7 +233,7 @@ Video VAE训练是一个可选项,因为我们已经提供了训练好的Video
<h4 id="dit-train">c. Video DiT训练 </h4>
如果数据预处理时,数据的格式为相对路径,则进入scripts/train.sh进行如下设置。
如果数据预处理时,数据的格式为相对路径,则进入scripts/train_t2iv.sh进行如下设置。
```
export DATASET_NAME="datasets/internal_datasets/"
export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.json"
@@ -472,89 +243,30 @@ export DATASET_META_NAME="datasets/internal_datasets/json_of_internal_datasets.j
train_data_format="normal"
```
如果数据的格式为绝对路径,则进入scripts/train.sh进行如下设置。
如果数据的格式为绝对路径,则进入scripts/train_t2iv.sh进行如下设置。
```
export DATASET_NAME=""
export DATASET_META_NAME="/mnt/data/json_of_internal_datasets.json"
```
最后运行scripts/train.sh。
最后运行scripts/train_t2iv.sh。
```sh
sh scripts/train.sh
sh scripts/train_t2iv.sh
```
关于一些参数的设置细节,可以查看[Readme Train](scripts/README_TRAIN.md)与[Readme Lora](scripts/README_TRAIN_LORA.md)
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
如果你想训练EasyAnimateV1。请切换到git分支v1。
</details>
# 模型地址
EasyAnimateV5.1:
7B:
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
EasyAnimateV2:
| 名称 | 种类 | 存储空间 | 下载地址 | Hugging Face | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control)| 官方的视频控制权重,支持不同的控制条件,如Canny、Depth、Pose、MLSD等,同时支持使用轨迹控制。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera)| 官方的视频相机控制权重,支持通过输入相机运动轨迹控制生成方向。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh)| 官方的文生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
12B:
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV5.1-12b-zh-InP | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-12b-zh-Control | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control)| 官方的视频控制权重,支持不同的控制条件,如Canny、Depth、Pose、MLSD等,同时支持使用轨迹控制。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-12b-zh-Control-Camera | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control-Camera)| 官方的视频相机控制权重,支持通过输入相机运动轨迹控制生成方向。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-12b-zh | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh)| 官方的文生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
<details>
<summary>(Obsolete) EasyAnimateV5:</summary>
7B:
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV5-7b-zh-InP | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-7b-zh-InP)| 官方的7B图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-7b-zh | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh)| 官方的7B文生视频权重。可用于进行下游任务的fientune。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | 通过奖励反向传播技术,优化了EasyAnimateV5-12b生成的视频,以更好地匹配人类偏好|
12B:
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV5-12b-zh-InP | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-12b-zh-Control | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-Control)| 官方的视频控制权重,支持不同的控制条件,如Canny、Depth、Pose、MLSD等。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-12b-zh | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh)| 官方的文生视频权重。可用于进行下游任务的fientune。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | 通过奖励反向传播技术,优化了EasyAnimateV5-12b生成的视频,以更好地匹配人类偏好|
</details>
<details>
<summary>(Obsolete) EasyAnimateV4:</summary>
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV4-XL-2-InP | EasyAnimateV4 | 解压前 8.9 GB / 解压后 14.0 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV4-XL-2-InP)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV4-XL-2-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024,1280)的视频预测,以144帧、每秒24帧进行训练 |
</details>
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV3-XL-2-InP-512x512 | EasyAnimateV3 | 18.2GB| [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-512x512)| 官方的512x512分辨率的图生视频权重。以144帧、每秒24帧进行训练 |
| EasyAnimateV3-XL-2-InP-768x768 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-768x768)| 官方的768x768分辨率的图生视频权重。以144帧、每秒24帧进行训练 |
| EasyAnimateV3-XL-2-InP-960x960 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-960x960) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-960x960)| 官方的960x960(720P)分辨率的图生视频权重。以144帧、每秒24帧进行训练 |
</details>
<details>
<summary>(Obsolete) EasyAnimateV2:</summary>
| 名称 | 种类 | 存储空间 | 下载地址 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|--|
| EasyAnimateV2-XL-2-512x512 | EasyAnimateV2 | 16.2GB | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV2-XL-2-512x512)| 官方的512x512分辨率的重量。以144帧、每秒24帧进行训练 |
| EasyAnimateV2-XL-2-768x768 | EasyAnimateV2 | 16.2GB | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV2-XL-2-768x768)| 官方的768x768分辨率的重量。以144帧、每秒24帧进行训练 |
| easyanimatev2_minimalism_lora.safetensors | Lora of Pixart | 485.1MB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimatev2_minimalism_lora.safetensors)| - | - | 使用特定类型的图像进行lora训练的结果。图片可从这里[下载](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/webui/Minimalism.zip). |
</details>
| EasyAnimateV2-XL-2-512x512.tar | EasyAnimateV2 | 16.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-512x512)| 官方的512x512分辨率的重量。以144帧、每秒24帧进行训练 |
| EasyAnimateV2-XL-2-768x768.tar | EasyAnimateV2 | 16.2GB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Diffusion_Transformer/EasyAnimateV2-XL-2-768x768.tar) | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV2-XL-2-768x768) | 官方的768x768分辨率的重量。以144帧、每秒24帧进行训练 |
| easyanimatev2_minimalism_lora.safetensors | Lora of Pixart | 485.1MB | [Download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimatev2_minimalism_lora.safetensors)| - | 使用特定类型的图像进行lora训练的结果。图片可从这里[下载](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/webui/Minimalism.zip). |
<details>
<summary>(Obsolete) EasyAnimateV1:</summary>
@@ -572,30 +284,57 @@ EasyAnimateV5.1:
| easyanimate_portrait_lora.safetensors | Lora of Pixart | 654.0MB | [download](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Personalized_Model/easyanimate_portrait_lora.safetensors)| Training with internal portrait datasets |
</details>
# 算法细节
### 1. 数据预处理
**视频分割**
对于较长的视频分割,EasyAnimate使用PySceneDetect以识别视频内的场景变化并基于这些转换,根据一定的门限值来执行场景剪切,以确保视频片段的主题一致性。切割后,我们只保留长度在3到10秒之间的片段用于模型训练。
**视频清洗与描述**
参考SVD的数据准备流程,EasyAnimate提供了一条简单但有效的数据处理链路来进行高质量的数据筛选与打标。并且支持了分布式处理来提升数据预处理的速度,其整体流程如下:
- 时长过滤: 统计视频基本信息,来过滤时间短/分辨率低的低质量视频
- 美学过滤: 通过计算视频均匀4帧的美学得分均值,来过滤内容较差的视频(模糊、昏暗等)
- 文本过滤: 通过easyocr计算中间帧的文本占比,来过滤文本占比过大的视频
- 运动过滤: 计算帧间光流差异来过滤运动过慢或过快的视频。
- 文本描述: 通过videochat2和vila对视频帧进行recaption。PAI也在自研质量更高的视频recaption模型,将在第一时间放出供大家使用。
### 2. 模型结构
我们使用了[PixArt-alpha](https://github.com/PixArt-alpha/PixArt-alpha)作为基础模型,并在此基础上修改了VAE和DiT的模型结构来更好地支持视频的生成。EasyAnimate的整体结构如下:
下图概述了EasyAnimate的管道。它包括Text Encoder、Video VAE(视频编码器和视频解码器)和Diffusion Transformer(DiT)。T5 Encoder用作文本编码器。其他组件将在以下部分中详细说明。
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/pipeline_v2.jpg" alt="ui" style="zoom:50%;" />
为了引入特征点在时间轴上的特征信息,EasyAnimate引入了运动模块(Motion Module),以实现从2D图像到3D视频的扩展。为了更好的生成效果,其联合图片和视频将Backbone连同Motion Module一起Finetune。在一个Pipeline中即实现了图片的生成,也实现了视频的生成。
另外,参考U-ViT,其将跳连接结构引入到EasyAnimate当中,通过引入浅层特征进一步优化深层特征,并且0初始化了一个全连接层给每一个跳连接结构,使其可以作为一个可插入模块应用到之前已经训练的还不错的DIT中。
同时,其提出了Slice VAE,用于解决MagViT在面对长、大视频时编解码上的显存困难,同时相比于MagViT在视频编解码阶段进行了时间维度更大的压缩。
更多细节可以看查看[arxiv](https://arxiv.org/abs/2405.18991)。
# 未来计划
- 支持更大规模参数量的文视频生成模型。
- 支持更大分辨率的文视频生成模型。
- 支持视频inpaint模型。
# 联系我们
1. 扫描下方二维码或搜索群号:77450006752 来加入钉钉群。
2. 扫描下方二维码来加入微信群(如果二维码失效,可扫描最右边同学的微信,邀请您入群)
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/group/dd.png" alt="ding group" width="30%"/>
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/group/wechat.jpg" alt="Wechat group" width="30%"/>
<img src="https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/group/person.jpg" alt="Person" width="30%"/>
# 参考文献
- CogVideo: https://github.com/THUDM/CogVideo/
- Flux: https://github.com/black-forest-labs/flux
- magvit: https://github.com/google-research/magvit
- PixArt: https://github.com/PixArt-alpha/PixArt-alpha
- Open-Sora-Plan: https://github.com/PKU-YuanGroup/Open-Sora-Plan
- Open-Sora: https://github.com/hpcaitech/Open-Sora
- Animatediff: https://github.com/guoyww/AnimateDiff
- HunYuan DiT: https://github.com/tencent/HunyuanDiT
- ComfyUI-KJNodes: https://github.com/kijai/ComfyUI-KJNodes
- ComfyUI-EasyAnimateWrapper: https://github.com/kijai/ComfyUI-EasyAnimateWrapper
- ComfyUI-CameraCtrl-Wrapper: https://github.com/chaojie/ComfyUI-CameraCtrl-Wrapper
- CameraCtrl: https://github.com/hehao13/CameraCtrl
- DragAnything: https://github.com/showlab/DragAnything
# 许可证
本项目采用 [Apache License (Version 2.0)](https://github.com/modelscope/modelscope/blob/master/LICENSE).
-3
View File
@@ -1,3 +0,0 @@
from .comfyui.comfyui_nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
Executable → Regular
+9 -42
View File
@@ -1,58 +1,25 @@
import time
import time
import torch
from easyanimate.api.api import (infer_forward_api,
update_diffusion_transformer_api,
update_edition_api)
from easyanimate.ui.ui import ui, ui_eas, ui_modelscope
from easyanimate.api.api import infer_forward_api, update_diffusion_transformer_api, update_edition_api
from easyanimate.ui.ui import ui_modelscope, ui
if __name__ == "__main__":
# Choose the ui mode
ui_mode = "normal"
# GPU memory mode, which can be choosen in ["model_cpu_offload", "model_cpu_offload_and_qfloat8", "sequential_cpu_offload"].
# "model_cpu_offload" means that the entire model will be moved to the CPU after use, which can save some GPU memory.
#
# "model_cpu_offload_and_qfloat8" indicates that the entire model will be moved to the CPU after use,
# and the transformer model has been quantized to float8, which can save more GPU memory.
#
# "sequential_cpu_offload" means that each layer of the model will be moved to the CPU after use,
# resulting in slower speeds but saving a large amount of GPU memory.
#
# EasyAnimateV1, V2 and V3 support "model_cpu_offload" "sequential_cpu_offload"
# EasyAnimateV4, V5 and V5.1 support "model_cpu_offload" "model_cpu_offload_and_qfloat8" "sequential_cpu_offload"
GPU_memory_mode = "model_cpu_offload_and_qfloat8"
# EasyAnimateV5.1 support TeaCache.
enable_teacache = True
# Recommended to be set between 0.05 and 0.1. A larger threshold can cache more steps, speeding up the inference process,
# but it may cause slight differences between the generated content and the original content.
teacache_threshold = 0.08
# Use torch.float16 if GPU does not support torch.bfloat16
# ome graphics cards, such as v100, 2080ti, do not support torch.bfloat16
weight_dtype = torch.bfloat16
# Server ip
server_name = "0.0.0.0"
server_port = 7860
# Params below is used when ui_mode = "modelscope"
edition = "v5.1"
# Config
config_path = "config/easyanimate_video_v5.1_magvit_qwen.yaml"
# Model path of the pretrained model
model_name = "models/Diffusion_Transformer/EasyAnimateV5.1-12b-zh-InP"
# "Inpaint" or "Control"
model_type = "Inpaint"
# Save dir
edition = "v2"
config_path = "config/easyanimate_video_magvit_motion_module_v2.yaml"
model_name = "models/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512"
savedir_sample = "samples"
if ui_mode == "modelscope":
demo, controller = ui_modelscope(model_type, edition, config_path, model_name, savedir_sample, GPU_memory_mode, enable_teacache, teacache_threshold, weight_dtype)
elif ui_mode == "eas":
demo, controller = ui_eas(edition, config_path, model_name, savedir_sample)
if ui_mode != "modelscope":
demo, controller = ui()
else:
demo, controller = ui(GPU_memory_mode, enable_teacache, teacache_threshold, weight_dtype)
demo, controller = ui_modelscope(edition, config_path, model_name, savedir_sample)
# launch gradio
app, _, _ = demo.queue(status_update_rate=1).launch(
-175
View File
@@ -1,175 +0,0 @@
https://www.youtube.com/watch?v=jQRHwqNC_0U
308341367 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.978989959 -0.010294991 -0.203648433 -0.000762398 -0.007398812 0.996273518 -0.085932352 -0.031535059 0.203774214 0.085633665 0.975265563 -0.153683138
308374733 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.977586806 -0.011497887 -0.210218683 0.002481976 -0.007218716 0.996089876 -0.088050455 -0.033528951 0.210409090 0.087594472 0.973681271 -0.161050474
308408100 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.976103604 -0.012630552 -0.216938347 0.005673227 -0.007174566 0.995891988 -0.090264283 -0.034554699 0.217187256 0.089663729 0.972003162 -0.168504957
308441467 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.974509835 -0.013677491 -0.223927394 0.008982419 -0.007251760 0.995697796 -0.092376187 -0.035320741 0.224227488 0.091645375 0.970218122 -0.175504380
308474833 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.972881317 -0.014926891 -0.230822623 0.012480344 -0.007060312 0.995534182 -0.094137549 -0.036223930 0.231196985 0.093214348 0.968431234 -0.182418105
308508200 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.971116245 -0.015677260 -0.238091335 0.016362104 -0.007245381 0.995441616 -0.095097564 -0.037379468 0.238496885 0.094075851 0.966575921 -0.188874438
308541567 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.969348371 -0.016915115 -0.245107263 0.019519189 -0.007186836 0.995248139 -0.097105585 -0.037570019 0.245585099 0.095890686 0.964620590 -0.194434889
308608300 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.965482831 -0.019274237 -0.259752661 0.026318709 -0.007045615 0.994960845 -0.100016415 -0.039387193 0.260371476 0.098394245 0.960481763 -0.206582088
308641667 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.963333905 -0.020359756 -0.267531812 0.029517279 -0.007176768 0.994804621 -0.101549059 -0.039716119 0.268209398 0.099745661 0.958182931 -0.212002640
308675033 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.961007357 -0.021468673 -0.275688171 0.032829264 -0.007240514 0.994686186 -0.102698565 -0.040377398 0.276428014 0.100690201 0.955745280 -0.217063216
308708400 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.958558679 -0.022696253 -0.283989727 0.036218875 -0.007334619 0.994525313 -0.104238495 -0.041102245 0.284800768 0.102001667 0.953144372 -0.221993067
308741767 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.955935478 -0.023643453 -0.292623252 0.039282347 -0.007694451 0.994391501 -0.105481185 -0.040878463 0.293476015 0.103084780 0.950392187 -0.226182594
308775133 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.953253388 -0.024266239 -0.301196188 0.041521166 -0.008146173 0.994344234 -0.105892323 -0.041338713 0.302062303 0.103395812 0.947664320 -0.229231401
308808500 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.950509310 -0.025124749 -0.309678584 0.044002359 -0.008489858 0.994252503 -0.106723674 -0.041918199 0.310580105 0.104070969 0.944832921 -0.232545524
308875233 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.944740891 -0.027402855 -0.326670647 0.049495607 -0.008745467 0.994038582 -0.108677343 -0.043024623 0.327701300 0.105528817 0.938869298 -0.238573853
308908600 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.941721439 -0.028621495 -0.335173875 0.052471544 -0.008895036 0.993906736 -0.109864593 -0.043136616 0.336276084 0.106443226 0.935728729 -0.241383039
308941967 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.938513279 -0.029339867 -0.343994170 0.055158821 -0.009417908 0.993835866 -0.110460714 -0.042776764 0.345114648 0.106908552 0.932451844 -0.243680639
308975333 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.935223758 -0.030297186 -0.352758616 0.057885988 -0.009723495 0.993758440 -0.111129038 -0.043273624 0.353923738 0.107360564 0.929091871 -0.245835520
309008700 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.931820571 -0.030954622 -0.361596853 0.060782047 -0.010214265 0.993724287 -0.111389861 -0.043572254 0.362775594 0.107488804 0.925656557 -0.247905504
309042067 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.928349078 -0.031636182 -0.370360881 0.063609461 -0.010624910 0.993705988 -0.111514710 -0.043950611 0.371557742 0.107459627 0.922169864 -0.249844265
309075433 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.924848616 -0.032647923 -0.378931642 0.066337543 -0.010918945 0.993619144 -0.112257645 -0.044287495 0.380178690 0.107958861 0.918590784 -0.251641562
309108800 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.921171725 -0.033373199 -0.387722611 0.069763984 -0.011345040 0.993589520 -0.112477288 -0.045101364 0.388990849 0.108009629 0.914887965 -0.254094049
309142167 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.917566240 -0.034442890 -0.396088243 0.072676557 -0.011466603 0.993533552 -0.112958498 -0.045261007 0.397417575 0.108188689 0.911237895 -0.255692348
309208900 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.910201550 -0.035699584 -0.412624091 0.078700093 -0.012179646 0.993540049 -0.112826422 -0.046712792 0.413986415 0.107720405 0.903886914 -0.259251707
309242267 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.906504273 -0.036408246 -0.420623928 0.081863254 -0.012601309 0.993497729 -0.113152504 -0.047517248 0.422008604 0.107873634 0.900151134 -0.260734715
309275633 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.902823627 -0.037298322 -0.428390414 0.085099882 -0.012867143 0.993441820 -0.113612421 -0.048487376 0.429818511 0.108084142 0.896422803 -0.262961863
309309000 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.899192631 -0.037919387 -0.435906827 0.088240571 -0.013205825 0.993431985 -0.113659412 -0.050131037 0.437353671 0.107958212 0.892785966 -0.264952805
309342367 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.895653009 -0.038608752 -0.443074495 0.091415472 -0.013500394 0.993405759 -0.113854058 -0.051581030 0.444548517 0.107955411 0.889225662 -0.267106652
309409100 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.889143944 -0.039823636 -0.455891609 0.096863763 -0.014223916 0.993320107 -0.114511266 -0.055615774 0.457406580 0.108301558 0.882638097 -0.271985441
309442467 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.886038363 -0.040410291 -0.461847425 0.099748874 -0.014641317 0.993258059 -0.114996016 -0.057240949 0.463380694 0.108652942 0.879473090 -0.275233826
309475833 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.883017659 -0.040438525 -0.467594445 0.102214218 -0.015467658 0.993232727 -0.115106329 -0.059105627 0.469084859 0.108873509 0.876416564 -0.278934678
309509200 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.880132735 -0.040309701 -0.473013163 0.104434028 -0.016339598 0.993225932 -0.115044698 -0.062218033 0.474446356 0.108983450 0.873512030 -0.284099169
309542567 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.877503335 -0.040934134 -0.477820396 0.106557835 -0.016707798 0.993136227 -0.115763828 -0.063666219 0.479279459 0.109566472 0.870796442 -0.287968235
309575933 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.874932587 -0.041235302 -0.482485890 0.109189737 -0.017189724 0.993095100 -0.116045728 -0.065114870 0.483939558 0.109825991 0.868182421 -0.292159517
309609300 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.872594893 -0.041835159 -0.486649781 0.111891410 -0.017455684 0.993017972 -0.116664611 -0.066478178 0.488132656 0.110295743 0.865772128 -0.296800241
309642667 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.870423913 -0.042362280 -0.490476996 0.114751898 -0.017655376 0.992963910 -0.117093928 -0.068415958 0.491986305 0.110580906 0.863551557 -0.302172177
309676033 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.868541598 -0.042488880 -0.493791699 0.117357209 -0.018038228 0.992948353 -0.117167257 -0.070397371 0.495287955 0.110671766 0.861650527 -0.307614305
309709400 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.867060125 -0.042669602 -0.496372908 0.120216022 -0.018459057 0.992890000 -0.117595725 -0.072785828 0.497861445 0.111125141 0.860107660 -0.313736773
309742767 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.866088688 -0.043403506 -0.498002529 0.122453533 -0.018637195 0.992727280 -0.118933745 -0.074490817 0.499542832 0.112288542 0.858980954 -0.320628504
309776133 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.865360856 -0.043743186 -0.499236524 0.125043974 -0.018964697 0.992611408 -0.119845577 -0.076056429 0.500790298 0.113177545 0.858137488 -0.327066266
309809500 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.864906728 -0.043624546 -0.500033319 0.127037753 -0.019370638 0.992572725 -0.120100662 -0.077751779 0.501558721 0.113561831 0.857637763 -0.333684160
309842867 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.864720166 -0.043874834 -0.500333965 0.129334470 -0.019549016 0.992482126 -0.120818146 -0.078636717 0.501873374 0.114254922 0.857361615 -0.339359986
309876233 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.864809155 -0.044859517 -0.500092745 0.131949856 -0.019065719 0.992348671 -0.121986344 -0.079543457 0.501738608 0.115029529 0.857336879 -0.345644978
309909600 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.864958882 -0.045732468 -0.499754608 0.135265495 -0.018738804 0.992201388 -0.123228706 -0.079543122 0.501492739 0.115952566 0.857356429 -0.352907848
309942967 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.865252912 -0.046075005 -0.499213874 0.137627428 -0.018783100 0.992089391 -0.124120452 -0.079804843 0.500983596 0.116772369 0.857542753 -0.358817645
309976333 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.865726292 -0.046780419 -0.498326808 0.140424549 -0.018578010 0.991933227 -0.125392660 -0.079780661 0.500172853 0.117813639 0.857873559 -0.365907578
310009700 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.866338551 -0.047223259 -0.497219771 0.142905036 -0.018469006 0.991810381 -0.126376569 -0.079630830 0.499115646 0.118668057 0.858371377 -0.372282108
310043067 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.867112100 -0.048142657 -0.495781094 0.145746867 -0.017913677 0.991660655 -0.127625570 -0.079277595 0.497790813 0.119546935 0.859018505 -0.379002051
310076433 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.868119121 -0.048691329 -0.493961900 0.147833429 -0.017613675 0.991528034 -0.128693298 -0.078908849 0.496043295 0.120421596 0.859906793 -0.385541694
310109800 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.869378269 -0.049083445 -0.491703421 0.149461353 -0.017440626 0.991386771 -0.129800156 -0.078681040 0.493839294 0.121421054 0.861034095 -0.391950834
310143167 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.870869160 -0.049489144 -0.489017129 0.150919036 -0.017325647 0.991209030 -0.131166071 -0.078495760 0.491209477 0.122701019 0.862355888 -0.397968385
310176533 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.872643054 -0.050130539 -0.485778838 0.152483740 -0.016998386 0.990996718 -0.132802665 -0.078367440 0.488062710 0.124146774 0.863934219 -0.404034113
310209900 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.874584794 -0.050529797 -0.482232451 0.154256000 -0.016926475 0.990767121 -0.134513766 -0.078088113 0.484577030 0.125806183 0.865654588 -0.410133718
310243267 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.876766086 -0.051406279 -0.478161752 0.155479703 -0.016476333 0.990476072 -0.136695534 -0.077475491 0.480634779 0.127728358 0.867568851 -0.416093417
310276633 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.878964126 -0.051730625 -0.474073857 0.156369104 -0.016295806 0.990260482 -0.138270065 -0.077875636 0.476609409 0.129259840 0.869560421 -0.421790538
310310000 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.881164610 -0.052391429 -0.469897866 0.158341644 -0.015933618 0.989986777 -0.140258059 -0.077393435 0.472541004 0.131077617 0.871506572 -0.427604406
310343367 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.883479238 -0.053010881 -0.465461344 0.159226902 -0.015646443 0.989683807 -0.142412066 -0.076457552 0.468208939 0.133100927 0.873535633 -0.432891901
310376733 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.885817289 -0.053641621 -0.460923284 0.160340201 -0.015189376 0.989411891 -0.144337848 -0.076142363 0.463785470 0.134858102 0.875623405 -0.438094773
310410100 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.888150871 -0.054433405 -0.456316859 0.161588987 -0.014720289 0.989080846 -0.146636873 -0.075492546 0.459316224 0.136952788 0.877651751 -0.443043392
310443467 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.890383422 -0.055029280 -0.451872855 0.163150185 -0.014452326 0.988748491 -0.148887515 -0.074526466 0.454981804 0.139097601 0.879570007 -0.448171618
310476833 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.892586112 -0.055750020 -0.447417051 0.164995991 -0.014085147 0.988394022 -0.151257530 -0.074047945 0.450656950 0.141312301 0.881441534 -0.453008463
310510200 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.894740224 -0.056834452 -0.442955762 0.167782038 -0.013566600 0.987951994 -0.154165044 -0.073001057 0.446380883 0.143947065 0.883189321 -0.458082426
310543567 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.896918535 -0.057157692 -0.438486159 0.169308129 -0.013389797 0.987645686 -0.156130597 -0.072796601 0.441993028 0.145907670 0.885072410 -0.463078899
310576933 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.899071574 -0.057735413 -0.433977962 0.171331268 -0.012979353 0.987315476 -0.158239439 -0.073365528 0.437609196 0.147901341 0.886917949 -0.467917732
310610300 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.901260018 -0.058569729 -0.429301769 0.173732582 -0.012589230 0.986863136 -0.161067307 -0.073096514 0.433095753 0.150568098 0.888682902 -0.473122013
310643667 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.903436303 -0.058904551 -0.424656391 0.175884090 -0.012623640 0.986431837 -0.163685232 -0.072987367 0.428536385 0.153239891 0.890434802 -0.478237266
310677033 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.905687928 -0.059230026 -0.419787079 0.177477414 -0.012393922 0.986069798 -0.165869713 -0.073552826 0.423763841 0.155429006 0.892337382 -0.483350707
310710400 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.907836795 -0.059959110 -0.415014774 0.180778921 -0.011716512 0.985710561 -0.168039829 -0.074532453 0.419159949 0.157415256 0.894161820 -0.488501040
310743767 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.910057962 -0.060243253 -0.410079598 0.183003510 -0.011557028 0.985307992 -0.170395508 -0.075045629 0.414319873 0.159809083 0.895991147 -0.493310403
310777133 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.912262857 -0.061014563 -0.405035436 0.186047988 -0.010837990 0.984901488 -0.172776058 -0.075995106 0.409461886 0.162006959 0.897827804 -0.498587911
310810500 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.914368153 -0.061986543 -0.400110722 0.189918152 -0.009896113 0.984494388 -0.175136760 -0.076662609 0.404762864 0.164099008 0.899576843 -0.503910219
310843867 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.916438997 -0.062444899 -0.395272344 0.193107816 -0.009283510 0.984166741 -0.177001923 -0.078113470 0.400066763 0.165880978 0.901349068 -0.509387915
310877233 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.918424070 -0.063240312 -0.390509814 0.197583308 -0.008470051 0.983769834 -0.179234952 -0.079494934 0.395506650 0.167921335 0.902982235 -0.515165633
310910600 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.920403063 -0.064143844 -0.385673106 0.202006428 -0.007444195 0.983395875 -0.181320533 -0.080844478 0.390899926 0.169759005 0.904643118 -0.521350926
310943967 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.922320426 -0.064726412 -0.380966574 0.206782136 -0.006681100 0.983053684 -0.183196262 -0.082257295 0.386368215 0.171510920 0.906258047 -0.527740396
310977333 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.924230337 -0.065651573 -0.376149088 0.212224392 -0.005523140 0.982706308 -0.185088485 -0.084043648 0.381795466 0.173141927 0.907884419 -0.534609231
311010700 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.926216066 -0.066453293 -0.371089995 0.217445108 -0.004528342 0.982309401 -0.187210441 -0.086343467 0.376965940 0.175077736 0.909529805 -0.542115507
311044067 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.928198338 -0.067681506 -0.365878463 0.223651886 -0.003151126 0.981852353 -0.189620674 -0.087078935 0.372072458 0.177158520 0.911140442 -0.549962431
311077433 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.930172324 -0.068023682 -0.360766143 0.229434932 -0.002271302 0.981599092 -0.190939993 -0.089419263 0.367116153 0.178426504 0.912901819 -0.557824120
311110800 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.932240486 -0.069168136 -0.355166763 0.236034988 -0.000758263 0.981183887 -0.193074211 -0.090408553 0.361838460 0.180260912 0.914646864 -0.565607851
311144167 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.934392452 -0.069684349 -0.349363536 0.241615506 0.000344464 0.980858505 -0.194721580 -0.091157813 0.356245220 0.181826025 0.916530788 -0.573942644
311177533 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.936547995 -0.069909394 -0.343497574 0.247953960 0.001326483 0.980611086 -0.195959508 -0.092038047 0.350536942 0.183069825 0.918482065 -0.582354554
311210900 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.938818872 -0.070407048 -0.337137878 0.254541679 0.002780362 0.980399191 -0.197001785 -0.093128492 0.344399989 0.184011623 0.920613050 -0.591251028
311244267 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.941219509 -0.071606763 -0.330118626 0.261408430 0.004858018 0.980041802 -0.198732078 -0.093668373 0.337760627 0.185446784 0.922782362 -0.600339638
311277633 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.943640530 -0.073040113 -0.322812200 0.268115280 0.007165964 0.979625583 -0.200704545 -0.093540800 0.330894560 0.187079668 0.924937844 -0.610000653
311311000 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.945890248 -0.073155127 -0.316132903 0.274995138 0.008447010 0.979476154 -0.201382905 -0.093009238 0.324376851 0.187815741 0.927094877 -0.618876927
311344367 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.948223889 -0.073267952 -0.309036076 0.280257918 0.010031201 0.979450643 -0.201434463 -0.094604409 0.317444265 0.187904969 0.929473460 -0.627696107
311377733 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.950452030 -0.073764659 -0.301992863 0.286158851 0.011713448 0.979248285 -0.202325478 -0.094314557 0.310650468 0.188763276 0.931592584 -0.636671149
311411100 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.952679396 -0.074046955 -0.294820279 0.291001410 0.013235929 0.979062200 -0.203130454 -0.093537715 0.303688586 0.189615980 0.933712482 -0.646112429
311444467 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.954840362 -0.073838852 -0.287797987 0.295123006 0.014388267 0.978982508 -0.203435913 -0.093086854 0.296770692 0.190107912 0.935834467 -0.654897124
311477833 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.956956148 -0.073810153 -0.280690134 0.298572477 0.015708711 0.978876293 -0.203849196 -0.091982212 0.289807051 0.190665469 0.937901139 -0.663605042
311511200 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.958909094 -0.073471151 -0.274035364 0.302887300 0.016908780 0.978969991 -0.203302488 -0.091291349 0.283209264 0.190314993 0.939985514 -0.671861542
311544567 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.960915208 -0.073034637 -0.267035455 0.305233885 0.018035194 0.979039550 -0.202870086 -0.090740338 0.276254803 0.190124914 0.942091167 -0.679444737
311577933 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.962860346 -0.072711810 -0.260024965 0.307762635 0.019370908 0.979177117 -0.202081606 -0.090920256 0.269304216 0.189539433 0.944219291 -0.687383897
311611300 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.964723289 -0.072647713 -0.253043920 0.310300920 0.020861125 0.979244888 -0.201604083 -0.090013857 0.262438059 0.189213380 0.946215928 -0.695448574
311644667 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.966474175 -0.072109833 -0.246430144 0.312982566 0.021857465 0.979376078 -0.200860038 -0.089453733 0.255831778 0.188739702 0.948117852 -0.703693245
311678033 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.968180656 -0.071329243 -0.239871487 0.315451422 0.022735778 0.979626119 -0.199538723 -0.090076508 0.249217331 0.187735870 0.950076818 -0.711800049
311711400 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.969955564 -0.071214139 -0.232625648 0.317100588 0.024302205 0.979777157 -0.198610634 -0.090612638 0.242065176 0.186990172 0.952070951 -0.720312801
311744767 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.971625566 -0.070924461 -0.225640222 0.319752746 0.025612244 0.979922652 -0.197726145 -0.089705197 0.235133588 0.186336622 0.953934431 -0.728574924
311778133 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.973283887 -0.070123084 -0.218634889 0.321794502 0.026567144 0.980219960 -0.196119979 -0.090112218 0.228062809 0.185071915 0.955895245 -0.736742499
311811500 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.974936903 -0.069585592 -0.211319387 0.323677056 0.027854756 0.980532587 -0.194370747 -0.091601931 0.220730945 0.183612958 0.957895696 -0.745155898
311844867 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.976595521 -0.069111675 -0.203678071 0.325412651 0.029160401 0.980770290 -0.192974925 -0.093024940 0.213098213 0.182519123 0.959831178 -0.754124007
311878233 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.978225887 -0.068719827 -0.195835873 0.326756322 0.030567350 0.981006324 -0.191552296 -0.094584758 0.205279663 0.181395233 0.961746335 -0.763424584
311911600 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.979814947 -0.068927020 -0.187647790 0.328006537 0.032526776 0.981138051 -0.190552205 -0.095768335 0.197242588 0.180602327 0.963575721 -0.773234588
311944967 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.981304646 -0.068062313 -0.180024162 0.328447396 0.033387616 0.981400371 -0.189046592 -0.097650805 0.189542726 0.179501727 0.965325177 -0.782386835
312011700 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.984019220 -0.066852629 -0.165036008 0.330236824 0.035566866 0.981961370 -0.185706273 -0.102535450 0.174473941 0.176868737 0.968646646 -0.801399816
312045067 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.985290170 -0.067055240 -0.157184348 0.331143493 0.037411377 0.982126057 -0.184468970 -0.104100012 0.166744456 0.175874978 0.970187783 -0.810926836
312078433 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.986511946 -0.066422537 -0.149607107 0.331300563 0.038505107 0.982488394 -0.182301641 -0.108530280 0.159096181 0.174082100 0.971794128 -0.821033556
312111800 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.987675488 -0.066199258 -0.141826585 0.331072149 0.039902102 0.982707620 -0.180813685 -0.111372361 0.151343793 0.172926068 0.973237693 -0.830963007
312145167 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.988770187 -0.066454358 -0.133855835 0.331553583 0.041716520 0.982821703 -0.179780975 -0.112955349 0.143503651 0.172178060 0.974557042 -0.841682363
312178533 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.989837110 -0.066110969 -0.125903890 0.331052202 0.042937610 0.982986569 -0.178588212 -0.114563927 0.135568470 0.171367228 0.975835264 -0.852094252
312211900 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.990878463 -0.065578103 -0.117725685 0.330035525 0.044033892 0.983213663 -0.177064568 -0.116183646 0.127361059 0.170265540 0.977132976 -0.862277506
312278633 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.992744386 -0.066010535 -0.100504406 0.327745317 0.047590412 0.983286500 -0.175735101 -0.116441532 0.110424995 0.169676989 0.979293644 -0.882789347
312312000 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.993589461 -0.066247106 -0.091604158 0.325707106 0.049457088 0.983374178 -0.174726233 -0.116631901 0.101656273 0.169075668 0.980346560 -0.892507297
312345367 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.994361401 -0.066342220 -0.082729384 0.323188511 0.051096663 0.983346462 -0.174410120 -0.115774985 0.092922404 0.169199482 0.981191576 -0.902159014
312378733 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.995042205 -0.066394515 -0.074045502 0.321123021 0.052643880 0.983293295 -0.174249545 -0.114348298 0.084377661 0.169487610 0.981913626 -0.912162258
312412100 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.995613933 -0.066976570 -0.065322898 0.318750973 0.054698888 0.983161271 -0.174361438 -0.112374767 0.075901076 0.170023575 0.982512593 -0.921916724
312445467 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996080637 -0.067227937 -0.057478175 0.317307963 0.056290012 0.983075321 -0.174339861 -0.110280604 0.068225883 0.170421124 0.983006537 -0.931334752
312478833 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996468782 -0.067571491 -0.049840629 0.315618106 0.057968128 0.983067632 -0.173832446 -0.109043532 0.060742829 0.170329422 0.983513176 -0.939837206
312545567 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996968269 -0.069315284 -0.035350382 0.314066124 0.062181991 0.982864857 -0.173522487 -0.105838595 0.046772409 0.170798257 0.984195232 -0.955363982
312578933 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.997107208 -0.070450135 -0.028530471 0.313899569 0.064466320 0.982708693 -0.173573241 -0.103689727 0.040265400 0.171231866 0.984407604 -0.963730355
312612300 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.997229338 -0.070999384 -0.022197181 0.313870797 0.066098705 0.982622564 -0.173446819 -0.101718329 0.034126069 0.171499059 0.984593034 -0.971128799
312645667 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.997287095 -0.071806230 -0.016196592 0.314430778 0.067936778 0.982572377 -0.173020676 -0.100299084 0.028338285 0.171450943 0.984785020 -0.978330016
312679033 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.997283638 -0.072916023 -0.010419844 0.315631738 0.070025228 0.982450604 -0.172879487 -0.098055672 0.022842666 0.171680242 0.984887838 -0.984845557
312712400 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.997224033 -0.074311025 -0.004705389 0.317342441 0.072381146 0.982274473 -0.172909766 -0.096280689 0.017471086 0.172089189 0.984926403 -0.992249970
312745767 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.997132242 -0.075674556 0.000820772 0.318938381 0.074675784 0.982096016 -0.172947705 -0.094996659 0.012281665 0.172513023 0.984930694 -0.999476186
312812500 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996909559 -0.077861786 0.010432968 0.324937323 0.078491226 0.981783450 -0.173032805 -0.093071054 0.003229727 0.173316956 0.984860778 -1.013745687
312845867 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996714592 -0.079571702 0.015111177 0.328924485 0.080987111 0.981538177 -0.173274204 -0.092338295 -0.001044474 0.173928753 0.984757662 -1.021159174
312879233 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996513069 -0.081097923 0.019618591 0.333397250 0.083276160 0.981306016 -0.173503771 -0.091893403 -0.005181046 0.174532533 0.984637797 -1.029133014
312912600 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996304870 -0.082457803 0.024027303 0.337992655 0.085388854 0.981065631 -0.173836112 -0.091045184 -0.009238216 0.175245434 0.984481454 -1.037087884
312945967 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.996070623 -0.083987780 0.028095186 0.343686830 0.087608948 0.980874479 -0.173810199 -0.091655046 -0.012959917 0.175588638 0.984378338 -1.044999310
312979333 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.995813906 -0.085612535 0.032017611 0.350273173 0.089899555 0.980658352 -0.173859864 -0.091970538 -0.016513752 0.176010445 0.984249771 -1.053555188
313012700 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.995503247 -0.087631822 0.035970747 0.357148609 0.092598148 0.980293870 -0.174497828 -0.090711195 -0.019970341 0.177043974 0.984000325 -1.061743164
313079433 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.994937420 -0.090959743 0.042729396 0.373297310 0.097084567 0.979798555 -0.174840838 -0.090398274 -0.025962725 0.178104073 0.983669102 -1.078977520
313112800 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.994614422 -0.092794865 0.046166051 0.381723551 0.099512480 0.979511738 -0.175082847 -0.089482401 -0.028973402 0.178734019 0.983470738 -1.087453627
313146167 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.994313717 -0.094416864 0.049250986 0.390708714 0.101667836 0.979261696 -0.175243229 -0.088835057 -0.031683687 0.179253995 0.983292520 -1.095919010
313179533 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.993896365 -0.096884072 0.052758712 0.399868176 0.104732476 0.978918135 -0.175357893 -0.087236466 -0.034657072 0.179813117 0.983090103 -1.104234639
313212900 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.993500292 -0.099100336 0.056002490 0.409121225 0.107507646 0.978583992 -0.175543502 -0.085376763 -0.037406720 0.180423230 0.982877493 -1.112553390
313246267 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.993204832 -0.100486375 0.058708660 0.418360786 0.109358117 0.978399456 -0.175428808 -0.084241278 -0.039812319 0.180656999 0.982740045 -1.120491631
313279633 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.992889166 -0.101999812 0.061376773 0.427805427 0.111317404 0.978242636 -0.175070733 -0.083516541 -0.042184193 0.180658147 0.982640922 -1.127647949
313346367 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.992061853 -0.106166579 0.067394227 0.445514137 0.116490223 0.977736056 -0.174534425 -0.080255502 -0.047364041 0.180999696 0.982341945 -1.142586657
313379733 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.991662741 -0.107884496 0.070469089 0.453512451 0.118734807 0.977493227 -0.174381718 -0.078337505 -0.050069973 0.181294993 0.982153296 -1.150170997
313413100 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.991388381 -0.108525760 0.073288999 0.460772594 0.119887300 0.977330446 -0.174505711 -0.076005143 -0.052689210 0.181789353 0.981924891 -1.158226337
313446467 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.991075039 -0.109311312 0.076297723 0.467606044 0.121208653 0.977176785 -0.174453244 -0.073937978 -0.055486653 0.182144195 0.981705010 -1.166156226
313479833 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.990619421 -0.110958062 0.079758711 0.474376029 0.123461276 0.976910770 -0.174363598 -0.071721072 -0.058570098 0.182575077 0.981445789 -1.174459386
313513200 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.990172505 -0.112342887 0.083291881 0.480516238 0.125477433 0.976650715 -0.174381196 -0.068654050 -0.061756589 0.183118701 0.981149137 -1.183004429
313546567 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.989827931 -0.113005586 0.086431257 0.486888950 0.126704708 0.976506293 -0.174302593 -0.067375589 -0.064703502 0.183480829 0.980891585 -1.191320360
313613300 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.988945186 -0.115346938 0.093179770 0.499662732 0.130252182 0.976069808 -0.174132317 -0.063686413 -0.070864335 0.184344187 0.980303764 -1.208898578
313646667 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.988459945 -0.116609581 0.096691161 0.506144929 0.132137433 0.975849390 -0.173947304 -0.062244953 -0.074072085 0.184716463 0.979996502 -1.217641786
313680033 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.988017082 -0.117491372 0.100090228 0.512718848 0.133605868 0.975730240 -0.173493326 -0.061754347 -0.077277094 0.184787005 0.979735672 -1.226863594
313713400 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.987563848 -0.118110009 0.103767216 0.518689193 0.134914964 0.975524366 -0.173638180 -0.060329507 -0.080719039 0.185478538 0.979327381 -1.236867288
313746767 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.987033010 -0.118996739 0.107729606 0.524762689 0.136519402 0.975330830 -0.173470974 -0.059829060 -0.084429525 0.185928762 0.978929102 -1.246822833
313780133 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.986401439 -0.120181613 0.112109698 0.530664721 0.138509735 0.975059330 -0.173419476 -0.059006379 -0.088471778 0.186589509 0.978446245 -1.258217434
313813500 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.985843003 -0.120521255 0.116568401 0.535510951 0.139647990 0.974971712 -0.172998726 -0.059314780 -0.092800871 0.186828136 0.977999628 -1.269093504
313846867 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.985232770 -0.121064983 0.121076837 0.540980122 0.140994951 0.974857450 -0.172549531 -0.060076193 -0.097142950 0.187072679 0.977531075 -1.280287059
313880233 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.984595537 -0.121475510 0.125759080 0.546086741 0.142260700 0.974723399 -0.172267690 -0.060268856 -0.101654008 0.187504575 0.976989508 -1.291598254
313913600 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.983899355 -0.121927045 0.130674735 0.550774390 0.143620268 0.974562824 -0.172047913 -0.060330612 -0.106373444 0.188045368 0.976382911 -1.304011370
313946967 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.983338177 -0.121466361 0.135247782 0.555621417 0.143982768 0.974595249 -0.171560779 -0.061902538 -0.110972978 0.188175604 0.975845754 -1.315990335
313980333 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.982553363 -0.122138359 0.140253618 0.560540623 0.145558864 0.974434435 -0.171143770 -0.061990912 -0.115764737 0.188573048 0.975212157 -1.328550227
314013700 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.981700897 -0.122885726 0.145473063 0.564752855 0.147349611 0.974100709 -0.171510577 -0.060449997 -0.120629206 0.189807490 0.974382758 -1.341307995
314047067 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.981288731 -0.121079043 0.149707228 0.568971736 0.146342263 0.974299014 -0.171246439 -0.061612007 -0.125125244 0.189950690 0.973787665 -1.353429746
314080433 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.980634212 -0.120555021 0.154347181 0.573425937 0.146657199 0.974337697 -0.170756325 -0.061823666 -0.129800752 0.190085620 0.973149121 -1.365053305
314113800 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.979800701 -0.120869070 0.159314975 0.577789203 0.147821948 0.974305928 -0.169931293 -0.061812832 -0.134682089 0.190049052 0.972492695 -1.376315743
314147167 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.979101181 -0.120271996 0.163998693 0.582034585 0.148152247 0.974241316 -0.170014083 -0.060640746 -0.139326364 0.190757766 0.971699357 -1.388087496
314180533 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.978274286 -0.120084383 0.168994710 0.585730462 0.148978561 0.974074721 -0.170246392 -0.058376753 -0.144169539 0.191724256 0.970802248 -1.399741900
314213900 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.977639973 -0.118906699 0.173439533 0.589550700 0.148618758 0.974200964 -0.169837952 -0.057725391 -0.148770094 0.191816747 0.970089555 -1.410344264
314247267 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.977026641 -0.117842659 0.177572533 0.593727427 0.148278132 0.974358559 -0.169230476 -0.057729606 -0.153076753 0.191672817 0.969447792 -1.420295571
314280633 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.976238608 -0.117690220 0.181953743 0.597938800 0.148924977 0.974331498 -0.168817803 -0.056156679 -0.157415062 0.191903919 0.968707085 -1.430101872
314314000 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.975512862 -0.117311463 0.186044857 0.602367459 0.149293035 0.974342108 -0.168431297 -0.054473966 -0.161512420 0.192082092 0.967997015 -1.439945870
314347367 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.974990189 -0.115997307 0.189575255 0.606349966 0.148653150 0.974467874 -0.168269381 -0.053204794 -0.165216208 0.192241952 0.967339993 -1.448945088
314380733 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.974489450 -0.115080222 0.192683354 0.612068472 0.148202047 0.974687278 -0.167394280 -0.053188083 -0.168542251 0.191680029 0.966877580 -1.457338892
314414100 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.974016428 -0.114069194 0.195653155 0.617461597 0.147714734 0.974824429 -0.167025849 -0.052650804 -0.171674982 0.191586778 0.966344774 -1.466093030
314480833 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.973019421 -0.112725042 0.201311350 0.630637185 0.147350907 0.975013077 -0.166244537 -0.051408918 -0.177541271 0.191422582 0.965316772 -1.483657565
314514200 0.532139961 0.946026558 0.500000000 0.500000000 0.000000000 0.000000000 0.972666502 -0.111541182 0.203662574 0.637904932 0.146565586 0.975191951 -0.165888965 -0.051916880 -0.180106655 0.191204563 0.964884639 -1.492338334
BIN
View File
Binary file not shown.
BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 188 KiB

BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 71 KiB

BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 118 KiB

BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 110 KiB

BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 186 KiB

BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 11 KiB

BIN
View File
Binary file not shown.
-147
View File
@@ -1,147 +0,0 @@
# ComfyUI EasyAnimate
Easily use EasyAnimate inside ComfyUI!
[![Arxiv Page](https://img.shields.io/badge/Arxiv-Page-red)](https://arxiv.org/abs/2405.18991)
[![Project Page](https://img.shields.io/badge/Project-Website-green)](https://easyanimate.github.io/)
[![Modelscope Studio](https://img.shields.io/badge/Modelscope-Studio-blue)](https://modelscope.cn/studios/PAI/EasyAnimate/summary)
[![Hugging Face Spaces](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-yellow)](https://huggingface.co/spaces/alibaba-pai/EasyAnimate)
English | [简体中文](./README_zh-CN.md)
- [Installation](#installation)
- [Node types](#node-types)
- [Example workflows](#example-workflows)
## Installation
### Option 1: Install via ComfyUI Manager
![](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/ComfyUI_Manager.jpg)
### Option 2: Install manually
The EasyAnimate repository needs to be placed at `ComfyUI/custom_nodes/EasyAnimate/`.
```
cd ComfyUI/custom_nodes/
# Git clone the easyanimate itself
git clone https://github.com/aigc-apps/EasyAnimate.git
# Git clone the video outout node
git clone https://github.com/Kosinkadink/ComfyUI-VideoHelperSuite.git
git clone https://github.com/kijai/ComfyUI-KJNodes.git
cd EasyAnimate/
pip install -r comfyui/requirements.txt
```
### Download models into `ComfyUI/models/EasyAnimate/`
EasyAnimateV5.1:
7B:
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, and trajectory control. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera) | Official video camera control weights, supporting direction generation control by inputting camera motion trajectories. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
12B:
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV5.1-12b-zh-InP | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-12b-zh-Control | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, and trajectory control. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-12b-zh-Control-Camera | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control-Camera) | Official video camera control weights, supporting direction generation control by inputting camera motion trajectories. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
| EasyAnimateV5.1-12b-zh | EasyAnimateV5.1 | 39 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports for multilingual prediction. |
<details>
<summary>(Obsolete) EasyAnimateV5:</summary>
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV5-12b-zh-InP | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-InP) | Official image-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports bilingual prediction in Chinese and English. |
| EasyAnimateV5-12b-zh-Control | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-Control) | Official video control weights, supporting various control conditions such as Canny, Depth, Pose, MLSD, etc. Supports video prediction at multiple resolutions (512, 768, 1024) and is trained with 49 frames at 8 frames per second. Bilingual prediction in Chinese and English is supported. |
| EasyAnimateV5-12b-zh | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh) | Official text-to-video weights. Supports video prediction at multiple resolutions (512, 768, 1024), trained with 49 frames at 8 frames per second, and supports bilingual prediction in Chinese and English. |
</details>
<details>
<summary>(Obsolete) EasyAnimateV4:</summary>
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV4-XL-2-InP | EasyAnimateV4 | Before extraction: 8.9 GB \/ After extraction: 14.0 GB |[🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV4-XL-2-InP)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV4-XL-2-InP)| | Our official graph-generated video model is capable of predicting videos at multiple resolutions (512, 768, 1024, 1280) and has been trained on 144 frames at a rate of 24 frames per second. |
</details>
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
| Name | Type | Storage Space | Hugging Face | Model Scope | Description |
|--|--|--|--|--|--|
| EasyAnimateV3-XL-2-InP-512x512 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-512x512) | EasyAnimateV3 official weights for 512x512 text and image to video resolution. Training with 144 frames and fps 24 |
| EasyAnimateV3-XL-2-InP-768x768 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-768x768) | EasyAnimateV3 official weights for 768x768 text and image to video resolution. Training with 144 frames and fps 24 |
| EasyAnimateV3-XL-2-InP-960x960 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-960x960) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-960x960) | EasyAnimateV3 official weights for 960x960 text and image to video resolution. Training with 144 frames and fps 24 |
</details>
## Node types
- **LoadEasyAnimateModel**
- Loads the EasyAnimate model
- **EasyAnimate_TextBox**
- Write the prompt for EasyAnimate model
- **EasyAnimateI2VSampler**
- EasyAnimate Sampler for Image to Video
- **EasyAnimateT2VSampler**
- EasyAnimate Sampler for Text to Video
- **EasyAnimateV2VSampler**
- EasyAnimate Sampler for Video to Video
## Example workflows
### Text to Video Generation
Our user interface is shown as follows, this is the [json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_t2v.json):
![Workflow Diagram](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_t2v.jpg)
### Image to Video Generation
Our user interface is shown as follows, this is the [json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_i2v.json):
![Workflow Diagram](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_i2v.jpg)
You can run a demo using the following photo:
![Demo Image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png)
### Video to Video Generation
Our user interface is shown as follows, this is the [json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_v2v.json):
![Workflow Diagram](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_v2v.jpg)
You can run a demo using the following video:
[Demo Video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/play_guitar.mp4)
### Camera Control Video Generation
Our user interface is shown as follows, this is the [json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_camera.json):
![Workflow Diagram](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_camera.jpg)
You can run a demo using the following photo:
![Demo Image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png)
### Trajectory Control Video Generation
Our user interface is shown as follows, this is the [json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_trajectory.json):
![Workflow Diagram](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_trajectory.jpg)
You can run a demo using the following photo:
![Demo Image](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/dog.png)
### Control Video Generation
Our user interface is shown as follows, this is the [json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5/easyanimatev5.1_workflow_v2v_control.json):
![Workflow Diagram](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_v2v_control.jpg)
You can run a demo using the following video:
[Demo Video](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1.1/pose.mp4)
-155
View File
@@ -1,155 +0,0 @@
# ComfyUI EasyAnimate
在ComfyUI中使用EasyAnimate!
[![Arxiv Page](https://img.shields.io/badge/Arxiv-Page-red)](https://arxiv.org/abs/2405.18991)
[![Project Page](https://img.shields.io/badge/Project-Website-green)](https://easyanimate.github.io/)
[![Modelscope Studio](https://img.shields.io/badge/Modelscope-Studio-blue)](https://modelscope.cn/studios/PAI/EasyAnimate/summary)
[![Hugging Face Spaces](https://img.shields.io/badge/%F0%9F%A4%97%20Hugging%20Face-Spaces-yellow)](https://huggingface.co/spaces/alibaba-pai/EasyAnimate)
[English](./README.md) | 简体中文
- [安装](#安装)
- [节点类型](#节点类型)
- [示例工作流](#示例工作流)
## 安装
### 选项1:通过ComfyUI管理器安装
![](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/ComfyUI_Manager.jpg)
### 选项2:手动安装
EasyAnimate存储库需要放置在`ComfyUI/custom_nodes/EasyAnimate/`。
```
cd ComfyUI/custom_nodes/
# Git clone the easyanimate itself
git clone https://github.com/aigc-apps/EasyAnimate.git
# Git clone the video outout node
git clone https://github.com/Kosinkadink/ComfyUI-VideoHelperSuite.git
git clone https://github.com/kijai/ComfyUI-KJNodes.git
cd EasyAnimate/
pip install -r comfyui/requirements.txt
```
## 将模型下载到`ComfyUI/models/EasyAnimate/`
EasyAnimateV5.1:
7B:
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV5.1-7b-zh-InP | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-7b-zh-Control | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control)| 官方的视频控制权重,支持不同的控制条件,如Canny、Depth、Pose、MLSD等,同时支持使用轨迹控制。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-7b-zh-Control-Camera | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh-Control-Camera) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh-Control-Camera)| 官方的视频相机控制权重,支持通过输入相机运动轨迹控制生成方向。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
| EasyAnimateV5.1-7b-zh | EasyAnimateV5.1 | 30 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5.1-7b-zh)| 官方的文生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持多语言预测 |
12B:
|名称|类型|存储空间|拥抱面|型号范围|描述|
|--|--|--|--|--|--|
|EasyAnimateV5.1-12b-zh-InP | EasyAnimateV5.1 | 39 GB |[🤗链接](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-InP) | [😄链接](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-InP)|官方图像到视频权重。支持多种分辨率(5127681024)的视频预测,以每秒8帧的速度训练49帧,支持多语言预测|
|EasyAnimateV5.1-12b-zh-控件| EasyAnimateV5.1 | 39 GB |[🤗链接](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control) | [😄链接](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control)|官方视频控制权重,支持Canny、Depth、Pose、MLSD和轨迹控制等各种控制条件。支持多种分辨率(5127681024)的视频预测,以每秒8帧的速度训练49帧,支持多语言预测|
|EasyAnimateV5.1-12b-zh-控制摄像头| EasyAnimateV5.1 | 39 GB |[🤗链接](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh-Control-Camera) | [😄链接](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh-Control-Camera)|官方摄像机控制权重,支持通过输入摄像机运动轨迹进行方向生成控制。支持多种分辨率(5127681024)的视频预测,以每秒8帧的速度训练49帧,支持多语言预测|
|EasyAnimateV5.1-12b-zh| EasyAnimateV5.1 | 39 GB |[🤗链接](https://huggingface.co/alibaba-pai/EasyAnimateV5.1-12b-zh) | [😄链接](https://modelscope.cn/models/PAI/EasyAnimateV5.1-12b-zh)|官方文本到视频权重。支持多种分辨率(5127681024)的视频预测,以每秒8帧的速度训练49帧,支持多语言预测|
<details>
<summary>(Obsolete) EasyAnimateV5:</summary>
7B:
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV5-7b-zh-InP | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-7b-zh-InP)| 官方的7B图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-7b-zh | EasyAnimateV5 | 22 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-7b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh)| 官方的7B文生视频权重。可用于进行下游任务的fientune。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | 通过奖励反向传播技术,优化了EasyAnimateV5-12b生成的视频,以更好地匹配人类偏好|
12B:
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV5-12b-zh-InP | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-InP) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-12b-zh-Control | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh-Control) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh-Control)| 官方的视频控制权重,支持不同的控制条件,如Canny、Depth、Pose、MLSD等。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-12b-zh | EasyAnimateV5 | 34 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-12b-zh) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-12b-zh)| 官方的文生视频权重。可用于进行下游任务的fientune。支持多分辨率(512,768,1024)的视频预测,支持多分辨率(512,768,1024)的视频预测,以49帧、每秒8帧进行训练,支持中文与英文双语预测 |
| EasyAnimateV5-Reward-LoRAs | EasyAnimateV5 | - | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV5-Reward-LoRAs) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV5-Reward-LoRAs) | 通过奖励反向传播技术,优化了EasyAnimateV5-12b生成的视频,以更好地匹配人类偏好|
</details>
<details>
<summary>(Obsolete) EasyAnimateV4:</summary>
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV4-XL-2-InP | EasyAnimateV4 | 解压前 8.9 GB / 解压后 14.0 GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV4-XL-2-InP)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV4-XL-2-InP)| 官方的图生视频权重。支持多分辨率(512,768,1024,1280)的视频预测,以144帧、每秒24帧进行训练 |
</details>
<details>
<summary>(Obsolete) EasyAnimateV3:</summary>
| 名称 | 种类 | 存储空间 | Hugging Face | Model Scope | 描述 |
|--|--|--|--|--|--|
| EasyAnimateV3-XL-2-InP-512x512 | EasyAnimateV3 | 18.2GB| [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-512x512)| [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-512x512)| 官方的512x512分辨率的图生视频权重。以144帧、每秒24帧进行训练 |
| EasyAnimateV3-XL-2-InP-768x768 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-768x768) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-768x768)| 官方的768x768分辨率的图生视频权重。以144帧、每秒24帧进行训练 |
| EasyAnimateV3-XL-2-InP-960x960 | EasyAnimateV3 | 18.2GB | [🤗Link](https://huggingface.co/alibaba-pai/EasyAnimateV3-XL-2-InP-960x960) | [😄Link](https://modelscope.cn/models/PAI/EasyAnimateV3-XL-2-InP-960x960)| 官方的960x960(720P)分辨率的图生视频权重。以144帧、每秒24帧进行训练 |
</details>
## 节点类型
- **LoadEasyAnimateModel**
- 加载EasyAnimate模型
- **EasyAnimate_TextBox**
- 编写EasyAnimate模型的提示词
- **EasyAnimateI2VSampler**
- EasyAnimate图像到视频采样节点
- **EasyAnimateT2VSampler**
- EasyAnimate文本到视频采样节点
- **EasyAnimateV2VSampler**
- EasyAnimate视频到视频采样节点
## 示例工作流
### 文本到视频生成
我们的用户界面显示如下,这是[json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_t2v.json):
![工作流程图](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_t2v.jpg)
### 图像到视频生成
我们的用户界面显示如下,这是[json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_i2v.json):
![工作流程图](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_i2v.jpg)
您可以使用以下照片运行演示:
![演示图像](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png)
### 视频到视频生成
我们的用户界面显示如下,这是[json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_v2v.json):
![工作流程图](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_v2v.jpg)
您可以使用以下视频运行演示:
[演示视频](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/play_guitar.mp4)
### 镜头控制视频生成
我们的用户界面显示如下,这是[json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_camera.json):
![工作流程图](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_camera.jpg)
您可以使用以下照片运行演示:
![演示图像](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1/firework.png)
### 轨迹控制视频生成
我们的用户界面显示如下,这是[json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_trajectory.json):
![工作流程图](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_control_trajectory.jpg)
您可以使用以下照片运行演示:
![演示图像](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/dog.png)
### 控制视频生成
我们的用户界面显示如下,这是[json](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5/easyanimatev5.1_workflow_v2v_control.json):
![工作流程图](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/asset/v5.1/easyanimatev5.1_workflow_v2v_control.jpg)
您可以使用以下视频运行演示:
[演示视频](https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/cogvideox_fun/asset/v1.1/pose.mp4)
File diff suppressed because it is too large Load Diff
-26
View File
@@ -1,26 +0,0 @@
Pillow
einops
safetensors
timm
tomesd
torch>=2.1.2
torchdiffeq
torchsde
decord
datasets
numpy
scikit-image
opencv-python
omegaconf
SentencePiece
albumentations
imageio[ffmpeg]
imageio[pyav]
tensorboard
beautifulsoup4
ftfy
func_timeout
accelerate>=0.25.0
gradio>=3.41.2,<=3.48.0
diffusers>=0.30.1
transformers>=4.37.2
-80
View File
@@ -1,80 +0,0 @@
"""Modified from https://github.com/chaojie/ComfyUI-CameraCtrl-Wrapper/blob/main/camera_utils.py
"""
import copy
import numpy as np
CAMERA = {
# T
"base_T_norm": 1.5,
"base_angle": np.pi/3,
"Static": { "angle":[0., 0., 0.], "T":[0., 0., 0.]},
"Pan Up": { "angle":[0., 0., 0.], "T":[0., 1., 0.]},
"Pan Down": { "angle":[0., 0., 0.], "T":[0.,-1.,0.]},
"Pan Left": { "angle":[0., 0., 0.], "T":[1.,0.,0.]},
"Pan Right": { "angle":[0., 0., 0.], "T": [-1.,0.,0.]},
"Zoom In": { "angle":[0., 0., 0.], "T": [0.,0.,-2.]},
"Zoom Out": { "angle":[0., 0., 0.], "T": [0.,0.,2.]},
"ACW": { "angle": [0., 0., 1.], "T":[0., 0., 0.]},
"CW": { "angle": [0., 0., -1.], "T":[0., 0., 0.]},
}
def compute_R_form_rad_angle(angles):
theta_x, theta_y, theta_z = angles
Rx = np.array([[1, 0, 0],
[0, np.cos(theta_x), -np.sin(theta_x)],
[0, np.sin(theta_x), np.cos(theta_x)]])
Ry = np.array([[np.cos(theta_y), 0, np.sin(theta_y)],
[0, 1, 0],
[-np.sin(theta_y), 0, np.cos(theta_y)]])
Rz = np.array([[np.cos(theta_z), -np.sin(theta_z), 0],
[np.sin(theta_z), np.cos(theta_z), 0],
[0, 0, 1]])
# 计算相机外参的旋转矩阵
R = np.dot(Rz, np.dot(Ry, Rx))
return R
def get_camera_motion(angle, T, speed, n=16):
RT = []
for i in range(n):
_angle = (i/n)*speed*(CAMERA["base_angle"])*angle
R = compute_R_form_rad_angle(_angle)
# _T = (i/n)*speed*(T.reshape(3,1))
_T=(i/n)*speed*(CAMERA["base_T_norm"])*(T.reshape(3,1))
_RT = np.concatenate([R,_T], axis=1)
RT.append(_RT)
RT = np.stack(RT)
return RT
def create_relative(RT_list, K_1=4.7, dataset="syn"):
RT = copy.deepcopy(RT_list[0])
R_inv = RT[:,:3].T
T = RT[:,-1]
temp = []
for _RT in RT_list:
_RT[:,:3] = np.dot(_RT[:,:3], R_inv)
_RT[:,-1] = _RT[:,-1] - np.dot(_RT[:,:3], T)
temp.append(_RT)
RT_list = temp
return RT_list
def combine_camera_motion(RT_0, RT_1):
RT = copy.deepcopy(RT_0[-1])
R = RT[:,:3]
R_inv = RT[:,:3].T
T = RT[:,-1]
temp = []
for _RT in RT_1:
_RT[:,:3] = np.dot(_RT[:,:3], R)
_RT[:,-1] = _RT[:,-1] + np.dot(np.dot(_RT[:,:3], R_inv), T)
temp.append(_RT)
RT_1 = np.stack(temp)
return np.concatenate([RT_0, RT_1], axis=0)
-473
View File
@@ -1,473 +0,0 @@
{
"last_node_id": 81,
"last_link_id": 41,
"nodes": [
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": [
250,
160
],
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
38
],
"shape": 3,
"slot_index": 0
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion."
]
},
{
"id": 7,
"type": "LoadImage",
"pos": [
258.76883544921907,
468.15773315429715
],
"size": [
378.07147216796875,
314.0000114440918
],
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
39
],
"shape": 3,
"label": "图像",
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3,
"label": "遮罩"
}
],
"title": "Start Image(图片到视频的开始图片)",
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"firework.png",
"image"
]
},
{
"id": 79,
"type": "Note",
"pos": [
16,
460
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can upload image here\n(在此上传开始图像)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 80,
"type": "Note",
"pos": [
19.81457666015625,
-307.8177917480467
],
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 3,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": [
18,
-46
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 4,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 81,
"type": "Note",
"pos": [
789,
425
],
"size": {
"0": 248.36927795410156,
"1": 87.05973815917969
},
"flags": {},
"order": 5,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Pay attention to selecting a base length that is compatible with the model\n(注意选择和模型相兼容的base length)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 72,
"type": "EasyAnimateI2VSampler",
"pos": [
761,
93
],
"size": {
"0": 319.20001220703125,
"1": 282
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 35
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 37
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 38
},
{
"name": "start_img",
"type": "IMAGE",
"link": 39
},
{
"name": "end_img",
"type": "IMAGE",
"link": null
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
40
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateI2VSampler"
},
"widgets_values": [
72,
768,
43,
"fixed",
25,
7,
"Euler"
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": [
1134,
93
],
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 40,
"label": "图像",
"slot_index": 0
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null,
"label": "音频"
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理"
},
{
"name": "vae",
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3,
"label": "文件名",
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00049.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24
}
}
}
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": [
250,
-50
],
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 6,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
37
],
"shape": 3,
"slot_index": 0
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"fireworks display over night city. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic. "
]
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": [
239.81457666015626,
-307.8177917480467
],
"size": {
"0": 422.3550720214844,
"1": 154
},
"flags": {},
"order": 7,
"mode": 0,
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
35
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV3-XL-2-InP-768x768",
"model_cpu_offload",
"Inpaint",
"easyanimate_video_v3_slicevae_motion_module.yaml",
"bf16"
]
}
],
"links": [
[
35,
31,
0,
72,
0,
"EASYANIMATESMODEL"
],
[
37,
75,
0,
72,
1,
"STRING_PROMPT"
],
[
38,
73,
0,
72,
2,
"STRING_PROMPT"
],
[
39,
7,
0,
72,
3,
"IMAGE"
],
[
40,
72,
0,
17,
0,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Load EasyAnimate",
"bounding": [
220,
-388,
474,
240
],
"color": "#b06634",
"font_size": 24
},
{
"title": "Upload Your Start Image",
"bounding": [
218,
382,
452,
418
],
"color": "#a1309b",
"font_size": 24
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.7513148009015778,
"offset": [
270.9692117199942,
417.3951463953461
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
-383
View File
@@ -1,383 +0,0 @@
{
"last_node_id": 85,
"last_link_id": 48,
"nodes": [
{
"id": 80,
"type": "Note",
"pos": [
20,
-300
],
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 0,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": [
18,
-46
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": [
250,
160
],
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
46
],
"shape": 3,
"slot_index": 0
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion."
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": [
1148,
15
],
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 48,
"label": "图像",
"slot_index": 0
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null,
"label": "音频"
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理"
},
{
"name": "vae",
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3,
"label": "文件名",
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00050.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24
}
}
}
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": [
250,
-50
],
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
45
],
"shape": 3,
"slot_index": 0
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic."
]
},
{
"id": 85,
"type": "EasyAnimateT2VSampler",
"pos": [
769,
15
],
"size": {
"0": 315,
"1": 290
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 47,
"slot_index": 0
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 45
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 46
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
48
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EasyAnimateT2VSampler"
},
"widgets_values": [
72,
1008,
576,
false,
43,
"fixed",
25,
7,
"Euler"
]
},
{
"id": 81,
"type": "Note",
"pos": [
750,
-261
],
"size": {
"0": 376.2774963378906,
"1": 224.3558807373047
},
"flags": {},
"order": 4,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Pay attention to selecting a width and height compatible with the model;\nThe commonly used resolution for the 512x512 model is width=672 height=384;\nThe commonly used resolution for the 768x768 model is width=1008 height=576;\nThe commonly used resolution for the 1024x1024 model is width=1244 height=720;\n\n(注意选择和模型相兼容的高和宽;\n512x512模型的常用分辨率是width=672 height=384;\n768x768模型的常用分辨率是width=1008 height=576;\n1024x1024模型的常用分辨率是width=1244 height=720;)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": [
240,
-300
],
"size": {
"0": 422.3550720214844,
"1": 154
},
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
47
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV3-XL-2-InP-768x768",
"model_cpu_offload",
"Inpaint",
"easyanimate_video_v3_slicevae_motion_module.yaml",
"bf16"
]
}
],
"links": [
[
45,
75,
0,
85,
1,
"STRING_PROMPT"
],
[
46,
73,
0,
85,
2,
"STRING_PROMPT"
],
[
47,
31,
0,
85,
0,
"EASYANIMATESMODEL"
],
[
48,
85,
0,
17,
0,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Load EasyAnimate",
"bounding": [
220,
-380,
472,
240
],
"color": "#b06634",
"font_size": 24
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.9090909090909092,
"offset": [
33.575633594994244,
458.8503651453463
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
-450
View File
@@ -1,450 +0,0 @@
{
"last_node_id": 81,
"last_link_id": 41,
"nodes": [
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": [
250,
160
],
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
38
],
"shape": 3,
"slot_index": 0
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion."
]
},
{
"id": 7,
"type": "LoadImage",
"pos": [
258.76883544921907,
468.15773315429715
],
"size": [
378.07147216796875,
314.0000114440918
],
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
39
],
"shape": 3,
"label": "图像",
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3,
"label": "遮罩"
}
],
"title": "Start Image(图片到视频的开始图片)",
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"firework.png",
"image"
]
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": [
250,
-50
],
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
37
],
"shape": 3,
"slot_index": 0
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"fireworks display over night city. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic."
]
},
{
"id": 79,
"type": "Note",
"pos": [
16,
460
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 3,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can upload image here\n(在此上传开始图像)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 80,
"type": "Note",
"pos": [
19.024218139648436,
-329.9677812499998
],
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 4,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": [
18,
-46
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 5,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 72,
"type": "EasyAnimateI2VSampler",
"pos": [
761,
93
],
"size": {
"0": 319.20001220703125,
"1": 282
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 35
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 37
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 38
},
{
"name": "start_img",
"type": "IMAGE",
"link": 39
},
{
"name": "end_img",
"type": "IMAGE",
"link": null
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
40
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateI2VSampler"
},
"widgets_values": [
72,
768,
43,
"fixed",
25,
7,
"Euler"
]
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": [
244.9895975341797,
-333.2615217285155
],
"size": [
440.2642531237557,
166.77216219840386
],
"flags": {},
"order": 6,
"mode": 0,
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
35
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV4-XL-2-InP",
"model_cpu_offload",
"Inpaint",
"easyanimate_video_v4_slicevae_multi_text_encoder.yaml",
"bf16"
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": [
1134,
93
],
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 40,
"label": "图像",
"slot_index": 0
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null,
"label": "音频"
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理"
},
{
"name": "vae",
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3,
"label": "文件名",
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00055.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24
}
}
}
}
],
"links": [
[
35,
31,
0,
72,
0,
"EASYANIMATESMODEL"
],
[
37,
75,
0,
72,
1,
"STRING_PROMPT"
],
[
38,
73,
0,
72,
2,
"STRING_PROMPT"
],
[
39,
7,
0,
72,
3,
"IMAGE"
],
[
40,
72,
0,
17,
0,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Load EasyAnimate",
"bounding": [
219,
-410,
492,
259
],
"color": "#b06634",
"font_size": 24
},
{
"title": "Upload Your Start Image",
"bounding": [
218,
382,
452,
418
],
"color": "#a1309b",
"font_size": 24
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.7513148009015778,
"offset": [
237.60582500124437,
450.34259561409607
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
-408
View File
@@ -1,408 +0,0 @@
{
"last_node_id": 96,
"last_link_id": 77,
"nodes": [
{
"id": 80,
"type": "Note",
"pos": [
19.46723632812501,
-310.0837280273436
],
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 0,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": [
18,
-46
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": [
250,
160
],
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
52
],
"shape": 3,
"slot_index": 0
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion."
]
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": [
250,
-50
],
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
51
],
"shape": 3,
"slot_index": 0
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"1girl, black_hair, brown_eyes, earrings, freckles, grey_background, jewelry, lips, long_hair, looking_at_viewer, nose, piercing, realistic, red_lips, solo, upper_body"
]
},
{
"id": 96,
"type": "LoadEasyAnimateLora",
"pos": [
716.467236328125,
-316.0837280273436
],
"size": {
"0": 461.959228515625,
"1": 128.50164794921875
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 76
}
],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
77
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateLora"
},
"widgets_values": [
"easyanimate/easyanimatev4_minimalism_lora.safetensors",
0.7000000000000001
]
},
{
"id": 88,
"type": "EasyAnimateT2VSampler",
"pos": [
801,
35
],
"size": {
"0": 315,
"1": 290
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 77
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 51
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 52,
"slot_index": 2
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
53
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EasyAnimateT2VSampler"
},
"widgets_values": [
72,
672,
384,
false,
43,
"fixed",
25,
7,
"Euler"
]
},
{
"id": 95,
"type": "LoadEasyAnimateModel",
"pos": [
243.467236328125,
-315.0837280273436
],
"size": {
"0": 436.8020324707031,
"1": 154
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
76
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV4-XL-2-InP",
"model_cpu_offload",
"Inpaint",
"easyanimate_video_v4_slicevae_multi_text_encoder.yaml",
"bf16"
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": [
1163,
34
],
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 53,
"label": "图像",
"slot_index": 0
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null,
"label": "音频"
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理"
},
{
"name": "vae",
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3,
"label": "文件名",
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00054.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24
}
}
}
}
],
"links": [
[
51,
75,
0,
88,
1,
"STRING_PROMPT"
],
[
52,
73,
0,
88,
2,
"STRING_PROMPT"
],
[
53,
88,
0,
17,
0,
"IMAGE"
],
[
76,
95,
0,
96,
0,
"EASYANIMATESMODEL"
],
[
77,
96,
0,
88,
0,
"EASYANIMATESMODEL"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Load EasyAnimate",
"bounding": [
219,
-390,
985,
240
],
"color": "#b06634",
"font_size": 24
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.9090909090909091,
"offset": [
38.375070223172365,
449.4001463595484
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
-360
View File
@@ -1,360 +0,0 @@
{
"last_node_id": 87,
"last_link_id": 49,
"nodes": [
{
"id": 80,
"type": "Note",
"pos": [
18.171542358398433,
-312.8636376953127
],
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 0,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": [
18,
-46
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": [
250,
160
],
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
46
],
"shape": 3,
"slot_index": 0
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion."
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": [
1148,
15
],
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 48,
"label": "图像",
"slot_index": 0
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null,
"label": "音频"
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理"
},
{
"name": "vae",
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3,
"label": "文件名",
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00052.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24
}
}
}
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": [
250,
-50
],
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
45
],
"shape": 3,
"slot_index": 0
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"A young woman with beautiful and clear eyes and blonde hair standing and white dress in a forest wearing a crown. She seems to be lost in thought, and the camera focuses on her face. The video is of high quality, and the view is very clear. High quality, masterpiece, best quality, highres, ultra-detailed, fantastic."
]
},
{
"id": 85,
"type": "EasyAnimateT2VSampler",
"pos": [
769,
15
],
"size": {
"0": 315,
"1": 290
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 49,
"slot_index": 0
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 45
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 46
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
48
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EasyAnimateT2VSampler"
},
"widgets_values": [
72,
1008,
576,
false,
43,
"fixed",
25,
7,
"Euler"
]
},
{
"id": 87,
"type": "LoadEasyAnimateModel",
"pos": [
252,
-308
],
"size": [
441.4525528088707,
154
],
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
49
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV4-XL-2-InP",
"model_cpu_offload",
"Inpaint",
"easyanimate_video_v4_slicevae_multi_text_encoder.yaml",
"bf16"
]
}
],
"links": [
[
45,
75,
0,
85,
1,
"STRING_PROMPT"
],
[
46,
73,
0,
85,
2,
"STRING_PROMPT"
],
[
48,
85,
0,
17,
0,
"IMAGE"
],
[
49,
87,
0,
85,
0,
"EASYANIMATESMODEL"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Load EasyAnimate",
"bounding": [
218,
-393,
503,
254
],
"color": "#b06634",
"font_size": 24
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.8264462809917354,
"offset": [
159.35166594112934,
529.544431725907
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
-495
View File
@@ -1,495 +0,0 @@
{
"last_node_id": 86,
"last_link_id": 47,
"nodes": [
{
"id": 80,
"type": "Note",
"pos": [
18.27766899414062,
-307.4300560058589
],
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 0,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": [
18,
-46
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": [
250,
160
],
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
45
],
"shape": 3,
"slot_index": 0
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"The video is not of a high quality, it has a low resolution, and the audio quality is not clear. Strange motion trajectory, a poor composition and deformed video, low resolution, duplicate and ugly, strange body structure, long and strange neck, bad teeth, bad eyes, bad limbs, bad hands, rotating camera, blurry camera, shaking camera. Deformation, low-resolution, blurry, ugly, distortion."
]
},
{
"id": 79,
"type": "Note",
"pos": [
15.739953613281248,
462.38664912015946
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 3,
"mode": 0,
"properties": {
"text": ""
},
"widgets_values": [
"You can upload video here\n(在此上传视频)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": [
250,
-50
],
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
44
],
"shape": 3,
"slot_index": 0
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"一只穿着小外套的猫咪正在花园秋千上安静地弹吉他。晚霞的余光洒在它柔软的毛皮上,和煦的微风轻轻拂过,周围斑驳的光影随着音乐的旋律轻轻摇曳。"
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": [
1134,
93
],
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 47,
"label": "图像",
"slot_index": 0
},
{
"name": "audio",
"type": "VHS_AUDIO",
"link": null,
"label": "音频"
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理"
},
{
"name": "vae",
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"shape": 3,
"label": "文件名",
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00053.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24
}
}
}
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": [
238.27766899414075,
-307.4300560058589
],
"size": [
482.82215786889094,
154
],
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
43
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV4-XL-2-InP",
"model_cpu_offload",
"Inpaint",
"easyanimate_video_v4_slicevae_multi_text_encoder.yaml",
"bf16"
]
},
{
"id": 86,
"type": "EasyAnimateV2VSampler",
"pos": [
762,
41
],
"size": {
"0": 319.20001220703125,
"1": 306
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 43,
"slot_index": 0
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 44
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 45
},
{
"name": "validation_video",
"type": "IMAGE",
"link": 46,
"slot_index": 3
},
{
"name": "control_video",
"type": "IMAGE",
"link": null
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
47
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EasyAnimateV2VSampler"
},
"widgets_values": [
48,
768,
43,
"fixed",
25,
7,
0.7000000000000001,
"Euler"
]
},
{
"id": 85,
"type": "VHS_LoadVideo",
"pos": [
336,
470
],
"size": [
235.1999969482422,
398.971426827567
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null
},
{
"name": "vae",
"type": "VAE",
"link": null
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
46
],
"shape": 3,
"slot_index": 0
},
{
"name": "frame_count",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "audio",
"type": "AUDIO",
"links": null,
"shape": 3
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "1.mp4",
"force_rate": 0,
"force_size": "Disabled",
"custom_width": 512,
"custom_height": 512,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1,
"choose video to upload": "image",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"frame_load_cap": 0,
"skip_first_frames": 0,
"force_rate": 0,
"filename": "1.mp4",
"type": "input",
"format": "video/mp4",
"select_every_nth": 1
}
}
}
}
],
"links": [
[
43,
31,
0,
86,
0,
"EASYANIMATESMODEL"
],
[
44,
75,
0,
86,
1,
"STRING_PROMPT"
],
[
45,
73,
0,
86,
2,
"STRING_PROMPT"
],
[
46,
85,
0,
86,
3,
"IMAGE"
],
[
47,
86,
0,
17,
0,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Load EasyAnimate",
"bounding": [
218,
-387,
542,
248
],
"color": "#b06634",
"font_size": 24
},
{
"title": "Upload Your Video",
"bounding": [
218,
385,
456,
498
],
"color": "#a1309b",
"font_size": 24
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.683013455365071,
"offset": [
284.93704305884313,
442.10948247114595
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
@@ -1,680 +0,0 @@
{
"last_node_id": 133,
"last_link_id": 283,
"nodes": [
{
"id": 105,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 234,
"1": 813
},
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
273
],
"slot_index": 0
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 123,
"type": "Note",
"pos": {
"0": -5,
"1": 616
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 125,
"type": "Note",
"pos": {
"0": -117,
"1": 843
},
"size": {
"0": 326.1556091308594,
"1": 145.20904541015625
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 127,
"type": "Note",
"pos": {
"0": 1125.8018798828125,
"1": 1088.0283203125
},
"size": {
"0": 538.7950439453125,
"1": 127.34957885742188
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"CameraCombine is used to combine multiple camera movements, while CameraBasic produces a single camera movement. The nodes come from https://github.com/chaojie/ComfyUI-CameraCtrl-Wrapper/. Since ComfyUI-CameraCtrl-Wrapper requires a specific version of diffusers, the code has been copied into the current repository.\n(CameraCombine用于组合多个镜头运动,CameraBasic产出单个镜头运动;节点来自于https://github.com/chaojie/ComfyUI-CameraCtrl-Wrapper/,由于ComfyUI-CameraCtrl-Wrapper有具体diffusers版本要求,故复制代码到当前库中。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 133,
"type": "Note",
"pos": {
"0": -201,
"1": 247
},
"size": {
"0": 427.074951171875,
"1": 143.9142608642578
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Due to the large size of models from EasyAnimateV5 and above, when using the 12B model, if your graphics card has 24GB or less of VRAM, please set GPU_memory_mode to model_cpu_offload_and_qfloat8. This will load the model in float8 to reduce VRAM consumption, otherwise you may receive an out-of-memory error. \n(由于EasyAnimateV5以上的模型较大,当使用12B模型时,如果使用的显卡显存为24G及以下,请将GPU_memory_mode设置为model_cpu_offload_and_qfloat8,使得模型加载在float8上减少显存消耗,否则会提示显存不足。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 100,
"type": "LoadImage",
"pos": {
"0": 238,
"1": 1165
},
"size": {
"0": 378.07147216796875,
"1": 314
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
274
],
"slot_index": 0,
"shape": 3,
"label": "图像"
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3,
"label": "遮罩"
}
],
"title": "Start Image(图片到视频的开始图片)",
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"5.png",
"image"
]
},
{
"id": 99,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 234,
"1": 240
},
"size": {
"0": 409.7983703613281,
"1": 158.7380828857422
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
271
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5.1-12b-zh-Control-Camera",
"model_cpu_offload_and_qfloat8",
"Control",
"easyanimate_video_v5.1_magvit_qwen.yaml",
"bf16"
]
},
{
"id": 129,
"type": "CameraTrajectoryFromChaoJie",
"pos": {
"0": 1156.0380859375,
"1": 881.3924560546875
},
"size": {
"0": 367.79998779296875,
"1": 150
},
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "camera_pose",
"type": "CameraPose",
"link": 283
}
],
"outputs": [
{
"name": "camera_trajectory",
"type": "STRING",
"links": [
275
],
"slot_index": 0
},
{
"name": "video_length",
"type": "INT",
"links": null
}
],
"properties": {
"Node name for S&R": "CameraTrajectoryFromChaoJie"
},
"widgets_values": [
0.532139961,
0.946026558,
0.5,
0.5
]
},
{
"id": 128,
"type": "CameraBasicFromChaoJie",
"pos": {
"0": 779.039306640625,
"1": 1120.39013671875
},
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "CameraPose",
"type": "CameraPose",
"links": [],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CameraBasicFromChaoJie"
},
"widgets_values": [
"Pan Up",
1,
49
]
},
{
"id": 130,
"type": "CameraCombineFromChaoJie",
"pos": {
"0": 779.491943359375,
"1": 881.4488525390625
},
"size": {
"0": 315,
"1": 178
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "CameraPose",
"type": "CameraPose",
"links": [
283
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CameraCombineFromChaoJie"
},
"widgets_values": [
"Pan Up",
"Pan Left",
"Static",
"Static",
1,
49
]
},
{
"id": 104,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 233,
"1": 539
},
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 9,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
272
],
"slot_index": 0
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"Fireworks light up the evening sky over a sprawling cityscape with gothic-style buildings featuring pointed towers and clock faces. The city is lit by both artificial lights from the buildings and the colorful bursts of the fireworks. The scene is viewed from an elevated angle, showcasing a vibrant urban environment set against a backdrop of a dramatic, partially cloudy sky at dusk."
]
},
{
"id": 106,
"type": "VHS_VideoCombine",
"pos": {
"0": 1416,
"1": 170
},
"size": [
390,
546
],
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 276,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00049.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 132,
"type": "Note",
"pos": {
"0": 819,
"1": 658
},
"size": {
"0": 517.6458129882812,
"1": 93.61251831054688
},
"flags": {},
"order": 10,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Please set the video_length of the Camera Trajectory below to be the same as the video_length of the Sampler above.\n(请将下方Camera Trajectory的Video Length设置的与上方Sampler的video_legnth一样。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 131,
"type": "EasyAnimateV5_V2VSampler",
"pos": {
"0": 822,
"1": 211
},
"size": {
"0": 504,
"1": 394
},
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 271
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 272
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 273
},
{
"name": "validation_video",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "control_video",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "ref_image",
"type": "IMAGE",
"link": 274,
"shape": 7
},
{
"name": "camera_conditions",
"type": "STRING",
"link": 275,
"widget": {
"name": "camera_conditions"
},
"shape": 7
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
276
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_V2VSampler"
},
"widgets_values": [
49,
512,
43,
"fixed",
43,
6,
1,
"Flow",
0.08,
true,
""
]
}
],
"links": [
[
271,
99,
0,
131,
0,
"EASYANIMATESMODEL"
],
[
272,
104,
0,
131,
1,
"STRING_PROMPT"
],
[
273,
105,
0,
131,
2,
"STRING_PROMPT"
],
[
274,
100,
0,
131,
5,
"IMAGE"
],
[
275,
129,
0,
131,
6,
"STRING"
],
[
276,
131,
0,
106,
0,
"IMAGE"
],
[
283,
130,
0,
129,
0,
"CameraPose"
]
],
"groups": [
{
"title": "Load EasyAnimate",
"bounding": [
191,
151,
475,
287
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Prompts",
"bounding": [
191,
456,
475,
587
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
},
{
"title": "First Image of Trajectory",
"bounding": [
191,
1068,
475,
456
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
},
{
"title": "Generate Camera Control Video",
"bounding": [
750,
781,
932,
470
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 1.1,
"offset": [
-465.8996857769304,
51.92597569190605
]
},
"node_versions": {
"EasyAnimate": "de24d49f07f6d9b12b4e98de98ec959a8b44b989",
"comfy-core": "v0.2.7-3-g8afb97c",
"ComfyUI-VideoHelperSuite": "70faa9bcef65932ab72e7404d6373fb300013a2e"
}
},
"version": 0.4
}
File diff suppressed because it is too large Load Diff
@@ -1,497 +0,0 @@
{
"last_node_id": 85,
"last_link_id": 49,
"nodes": [
{
"id": 79,
"type": "Note",
"pos": {
"0": 16,
"1": 460
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can upload image here\n(在此上传开始图像)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1134,
"1": 93
},
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 42,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00050.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
45
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 84,
"type": "Note",
"pos": {
"0": -98,
"1": 198
},
"size": {
"0": 326.1556091308594,
"1": 145.20904541015625
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
44
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"Fireworks light up the evening sky over a sprawling cityscape with gothic-style buildings featuring pointed towers and clock faces. The city is lit by both artificial lights from the buildings and the colorful bursts of the fireworks. The scene is viewed from an elevated angle, showcasing a vibrant urban environment set against a backdrop of a dramatic, partially cloudy sky at dusk."
]
},
{
"id": 82,
"type": "EasyAnimateV5_I2VSampler",
"pos": {
"0": 767,
"1": 93
},
"size": {
"0": 336,
"1": 282
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 48
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 44
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 45
},
{
"name": "start_img",
"type": "IMAGE",
"link": 49,
"shape": 7
},
{
"name": "end_img",
"type": "IMAGE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
42
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_I2VSampler"
},
"widgets_values": [
49,
512,
43,
"fixed",
50,
6,
"Flow",
0.08,
true
]
},
{
"id": 83,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 258,
"1": -324
},
"size": {
"0": 427.9729919433594,
"1": 154
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
48
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5.1-12b-zh-InP",
"model_cpu_offload_and_qfloat8",
"Inpaint",
"easyanimate_video_v5.1_magvit_qwen.yaml",
"bf16"
]
},
{
"id": 7,
"type": "LoadImage",
"pos": {
"0": 259,
"1": 468
},
"size": {
"0": 378.07147216796875,
"1": 314
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
49
],
"slot_index": 0,
"shape": 3,
"label": "图像"
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3,
"label": "遮罩"
}
],
"title": "Start Image(图片到视频的开始图片)",
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"5.png",
"image"
]
},
{
"id": 85,
"type": "Note",
"pos": {
"0": -179,
"1": -318
},
"size": [
427.074951171875,
143.9142608642578
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Due to the large size of models from EasyAnimateV5 and above, when using the 12B model, if your graphics card has 24GB or less of VRAM, please set GPU_memory_mode to model_cpu_offload_and_qfloat8. This will load the model in float8 to reduce VRAM consumption, otherwise you may receive an out-of-memory error. \n(由于EasyAnimateV5以上的模型较大,当使用12B模型时,如果使用的显卡显存为24G及以下,请将GPU_memory_mode设置为model_cpu_offload_and_qfloat8,使得模型加载在float8上减少显存消耗,否则会提示显存不足。)"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
42,
82,
0,
17,
0,
"IMAGE"
],
[
44,
75,
0,
82,
1,
"STRING_PROMPT"
],
[
45,
73,
0,
82,
2,
"STRING_PROMPT"
],
[
48,
83,
0,
82,
0,
"EASYANIMATESMODEL"
],
[
49,
7,
0,
82,
3,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
},
{
"title": "Load EasyAnimate",
"bounding": [
219,
-410,
492,
259
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Upload Your Start Image",
"bounding": [
218,
382,
452,
418
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.6830134553650716,
"offset": [
353.6981370759636,
518.5082328158873
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
@@ -1,402 +0,0 @@
{
"last_node_id": 90,
"last_link_id": 53,
"nodes": [
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 88,
"type": "EasyAnimateV5_T2VSampler",
"pos": {
"0": 786,
"1": 15
},
"size": {
"0": 327.6000061035156,
"1": 290
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 51,
"slot_index": 0
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 52,
"slot_index": 1
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 53,
"slot_index": 2
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
50
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_T2VSampler"
},
"widgets_values": [
49,
672,
384,
false,
43,
"fixed",
50,
6,
"Flow",
0.08,
true
]
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
53
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1148,
"1": 15
},
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 50,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00053.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 89,
"type": "Note",
"pos": {
"0": -97,
"1": 193
},
"size": {
"0": 326.1556091308594,
"1": 145.20904541015625
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 87,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 252,
"1": -308
},
"size": {
"0": 441.4525451660156,
"1": 154
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
51
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5.1-12b-zh-InP",
"model_cpu_offload_and_qfloat8",
"Inpaint",
"easyanimate_video_v5.1_magvit_qwen.yaml",
"bf16"
]
},
{
"id": 90,
"type": "Note",
"pos": {
"0": -180,
"1": -301
},
"size": [
427.074951171875,
143.9142608642578
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Due to the large size of models from EasyAnimateV5 and above, when using the 12B model, if your graphics card has 24GB or less of VRAM, please set GPU_memory_mode to model_cpu_offload_and_qfloat8. This will load the model in float8 to reduce VRAM consumption, otherwise you may receive an out-of-memory error. \n(由于EasyAnimateV5以上的模型较大,当使用12B模型时,如果使用的显卡显存为24G及以下,请将GPU_memory_mode设置为model_cpu_offload_and_qfloat8,使得模型加载在float8上减少显存消耗,否则会提示显存不足。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
52
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"一只棕褐色的狗正摇晃着脑袋,坐在一个舒适的房间里的浅色沙发上。沙发看起来柔软而宽敞,为这只活泼的狗狗提供了一个完美的休息地点。在狗的后面,靠墙摆放着一个架子,架子上挂着一幅精美的镶框画,画中描绘着一些美丽的风景或场景。画框周围装饰着粉红色的花朵,这些花朵不仅增添了房间的色彩,还带来了一丝自然和生机。房间里的灯光柔和而温暖,从天花板上的吊灯和角落里的台灯散发出来,营造出一种温馨舒适的氛围。整个空间给人一种宁静和谐的感觉,仿佛时间在这里变得缓慢而美好。"
]
}
],
"links": [
[
50,
88,
0,
17,
0,
"IMAGE"
],
[
51,
87,
0,
88,
0,
"EASYANIMATESMODEL"
],
[
52,
75,
0,
88,
1,
"STRING_PROMPT"
],
[
53,
73,
0,
88,
2,
"STRING_PROMPT"
]
],
"groups": [
{
"title": "Load EasyAnimate",
"bounding": [
218,
-393,
503,
254
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.6830134553650716,
"offset": [
351.74219098221363,
574.4414281283872
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
@@ -1,555 +0,0 @@
{
"last_node_id": 90,
"last_link_id": 58,
"nodes": [
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 79,
"type": "Note",
"pos": {
"0": 15.739953994750977,
"1": 462.38665771484375
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can upload video here\n(在此上传视频)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
55
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
54
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"一只穿着小外套的猫咪正安静地坐在花园的秋千上弹吉他。它的小外套精致而合身,增添了几分俏皮与可爱。晚霞的余光洒在它柔软的毛皮上,给它的毛发镀上了一层温暖的金色光辉。和煦的微风轻轻拂过,带来阵阵花香和草木的气息,令人心旷神怡。周围斑驳的光影随着音乐的旋律轻轻摇曳,仿佛整个花园都在为这只小猫咪的演奏伴舞。阳光透过树叶间的缝隙,投下一片片光影交错的图案,与悠扬的吉他声交织在一起,营造出一种梦幻而宁静的氛围。猫咪专注而投入地弹奏着,每一个音符都似乎充满了魔力,让这个傍晚变得更加美好。"
]
},
{
"id": 88,
"type": "Note",
"pos": {
"0": -97,
"1": 195
},
"size": {
"0": 326.1556091308594,
"1": 145.20904541015625
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1314,
"1": -57
},
"size": [
390.9534912109375,
535.9734235491071
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 57,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00055.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 89,
"type": "EasyAnimateV5_V2VSampler",
"pos": {
"0": 774,
"1": -57
},
"size": {
"0": 504,
"1": 350
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 53
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 54
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 55
},
{
"name": "validation_video",
"type": "IMAGE",
"link": 58,
"shape": 7
},
{
"name": "control_video",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "ref_image",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "camera_conditions",
"type": "STRING",
"link": null,
"widget": {
"name": "camera_conditions"
},
"shape": 7
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
57
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_V2VSampler"
},
"widgets_values": [
49,
512,
43,
"fixed",
50,
6,
0.7000000000000001,
"Flow",
0.08,
true,
""
]
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 238.2776641845703,
"1": -307.4300537109375
},
"size": {
"0": 482.8221435546875,
"1": 154
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
53
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5.1-12b-zh-InP",
"model_cpu_offload_and_qfloat8",
"Inpaint",
"easyanimate_video_v5.1_magvit_qwen.yaml",
"bf16"
]
},
{
"id": 85,
"type": "VHS_LoadVideo",
"pos": {
"0": 335,
"1": 476
},
"size": [
252.056640625,
408.6037946428571
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
58
],
"slot_index": 0,
"shape": 3
},
{
"name": "frame_count",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "audio",
"type": "AUDIO",
"links": null,
"shape": 3
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "1.mp4",
"force_rate": 8,
"force_size": "Disabled",
"custom_width": 512,
"custom_height": 512,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1,
"choose video to upload": "image",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"frame_load_cap": 0,
"skip_first_frames": 0,
"force_rate": 8,
"filename": "1.mp4",
"type": "input",
"format": "video/mp4",
"select_every_nth": 1
}
}
}
},
{
"id": 90,
"type": "Note",
"pos": {
"0": -186,
"1": -295
},
"size": [
427.074951171875,
143.9142608642578
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Due to the large size of models from EasyAnimateV5 and above, when using the 12B model, if your graphics card has 24GB or less of VRAM, please set GPU_memory_mode to model_cpu_offload_and_qfloat8. This will load the model in float8 to reduce VRAM consumption, otherwise you may receive an out-of-memory error. \n(由于EasyAnimateV5以上的模型较大,当使用12B模型时,如果使用的显卡显存为24G及以下,请将GPU_memory_mode设置为model_cpu_offload_and_qfloat8,使得模型加载在float8上减少显存消耗,否则会提示显存不足。)"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
53,
31,
0,
89,
0,
"EASYANIMATESMODEL"
],
[
54,
75,
0,
89,
1,
"STRING_PROMPT"
],
[
55,
73,
0,
89,
2,
"STRING_PROMPT"
],
[
57,
89,
0,
17,
0,
"IMAGE"
],
[
58,
85,
0,
89,
3,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
},
{
"title": "Load EasyAnimate",
"bounding": [
218,
-387,
542,
248
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Upload Your Video",
"bounding": [
218,
385,
479,
529
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.6209213230591561,
"offset": [
447.5231554509637,
537.020461913544
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
@@ -1,563 +0,0 @@
{
"last_node_id": 89,
"last_link_id": 53,
"nodes": [
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 79,
"type": "Note",
"pos": {
"0": 15.739953994750977,
"1": 462.38665771484375
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can upload video here\n(在此上传视频)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
51
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
50
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"在这个阳光明媚的户外花园里,美女身穿一袭及膝的白色无袖连衣裙,裙摆在她轻盈的舞姿中轻柔地摆动,宛如一只翩翩起舞的蝴蝶。阳光透过树叶间洒下斑驳的光影,映衬出她柔和的脸庞和清澈的眼眸,显得格外优雅。仿佛每一个动作都在诉说着青春与活力,她在草地上旋转,裙摆随之飞扬,仿佛整个花园都因她的舞动而欢愉。周围五彩缤纷的花朵在微风中摇曳,玫瑰、菊花、百合,各自释放出阵阵香气,营造出一种轻松而愉快的氛围。"
]
},
{
"id": 88,
"type": "Note",
"pos": {
"0": -99,
"1": 197
},
"size": {
"0": 326.1556091308594,
"1": 145.20904541015625
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1173,
"1": 15
},
"size": [
390.9534912109375,
546.5720947265625
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 48,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00054.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 87,
"type": "EasyAnimateV5_V2VSampler",
"pos": {
"0": 816,
"1": 13
},
"size": {
"0": 336,
"1": 394
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 49
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 50,
"slot_index": 1
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 51,
"slot_index": 2
},
{
"name": "validation_video",
"type": "IMAGE",
"link": null,
"slot_index": 3,
"shape": 7
},
{
"name": "control_video",
"type": "IMAGE",
"link": 53,
"shape": 7
},
{
"name": "ref_image",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "camera_conditions",
"type": "STRING",
"link": null,
"widget": {
"name": "camera_conditions"
},
"shape": 7
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
48
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_V2VSampler"
},
"widgets_values": [
49,
512,
43,
"fixed",
50,
6,
1,
"Flow",
0.08,
true,
""
]
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 238.2776641845703,
"1": -307.4300537109375
},
"size": {
"0": 482.8221435546875,
"1": 154
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
49
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5.1-12b-zh-Control",
"model_cpu_offload_and_qfloat8",
"Control",
"easyanimate_video_v5.1_magvit_qwen.yaml",
"bf16"
]
},
{
"id": 85,
"type": "VHS_LoadVideo",
"pos": {
"0": 335,
"1": 476
},
"size": [
252.056640625,
262
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
53
],
"slot_index": 0,
"shape": 3
},
{
"name": "frame_count",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "audio",
"type": "AUDIO",
"links": null,
"shape": 3
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "demo_pose.mp4",
"force_rate": 0,
"force_size": "Disabled",
"custom_width": 512,
"custom_height": 512,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1,
"choose video to upload": "image",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"frame_load_cap": 0,
"skip_first_frames": 0,
"force_rate": 0,
"filename": "demo_pose.mp4",
"type": "input",
"format": "video/mp4",
"select_every_nth": 1
}
}
}
},
{
"id": 89,
"type": "Note",
"pos": {
"0": -192,
"1": -293
},
"size": {
"0": 427.074951171875,
"1": 143.9142608642578
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Due to the large size of models from EasyAnimateV5 and above, when using the 12B model, if your graphics card has 24GB or less of VRAM, please set GPU_memory_mode to model_cpu_offload_and_qfloat8. This will load the model in float8 to reduce VRAM consumption, otherwise you may receive an out-of-memory error. \n(由于EasyAnimateV5以上的模型较大,当使用12B模型时,如果使用的显卡显存为24G及以下,请将GPU_memory_mode设置为model_cpu_offload_and_qfloat8,使得模型加载在float8上减少显存消耗,否则会提示显存不足。)"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
48,
87,
0,
17,
0,
"IMAGE"
],
[
49,
31,
0,
87,
0,
"EASYANIMATESMODEL"
],
[
50,
75,
0,
87,
1,
"STRING_PROMPT"
],
[
51,
73,
0,
87,
2,
"STRING_PROMPT"
],
[
53,
85,
0,
87,
4,
"IMAGE"
]
],
"groups": [
{
"title": "Upload Your Video",
"bounding": [
218,
385,
487,
789
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
},
{
"title": "Load EasyAnimate",
"bounding": [
218,
-387,
542,
248
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.8264462809917354,
"offset": [
-156.13347668602108,
275.2525393282698
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
},
"node_versions": {
"EasyAnimate": "de24d49f07f6d9b12b4e98de98ec959a8b44b989",
"ComfyUI-VideoHelperSuite": "70faa9bcef65932ab72e7404d6373fb300013a2e"
}
},
"version": 0.4
}
-495
View File
@@ -1,495 +0,0 @@
{
"last_node_id": 84,
"last_link_id": 48,
"nodes": [
{
"id": 79,
"type": "Note",
"pos": {
"0": 16,
"1": 460
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can upload image here\n(在此上传开始图像)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 80,
"type": "Note",
"pos": {
"0": 19.02421760559082,
"1": -329.9677734375
},
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1134,
"1": 93
},
"size": [
390.9534912109375,
310
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 42,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00004.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 82,
"type": "EasyAnimateV5_I2VSampler",
"pos": {
"0": 767,
"1": 93
},
"size": {
"0": 336,
"1": 282
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 48
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 44
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 45
},
{
"name": "start_img",
"type": "IMAGE",
"link": 47,
"shape": 7
},
{
"name": "end_img",
"type": "IMAGE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
42
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_I2VSampler"
},
"widgets_values": [
49,
512,
43,
"fixed",
50,
6,
"DDIM"
]
},
{
"id": 83,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 258,
"1": -324
},
"size": {
"0": 427.9729919433594,
"1": 154
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
48
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5-12b-zh-InP",
"model_cpu_offload_and_qfloat8",
"Inpaint",
"easyanimate_video_v5_magvit_multi_text_encoder.yaml",
"bf16"
]
},
{
"id": 7,
"type": "LoadImage",
"pos": {
"0": 259,
"1": 468
},
"size": {
"0": 378.07147216796875,
"1": 314
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
47
],
"slot_index": 0,
"shape": 3,
"label": "图像"
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3,
"label": "遮罩"
}
],
"title": "Start Image(图片到视频的开始图片)",
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"firework.png",
"image"
]
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
44
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"夜城烟花汇演"
]
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
45
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 84,
"type": "Note",
"pos": {
"0": -98,
"1": 198
},
"size": [
326.1556114026207,
145.20905264909447
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
42,
82,
0,
17,
0,
"IMAGE"
],
[
44,
75,
0,
82,
1,
"STRING_PROMPT"
],
[
45,
73,
0,
82,
2,
"STRING_PROMPT"
],
[
47,
7,
0,
82,
3,
"IMAGE"
],
[
48,
83,
0,
82,
0,
"EASYANIMATESMODEL"
]
],
"groups": [
{
"title": "Upload Your Start Image",
"bounding": [
218,
382,
452,
418
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
},
{
"title": "Load EasyAnimate",
"bounding": [
219,
-410,
492,
259
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.7513148009015777,
"offset": [
206.91128312862938,
440.0942364134056
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
-400
View File
@@ -1,400 +0,0 @@
{
"last_node_id": 89,
"last_link_id": 53,
"nodes": [
{
"id": 80,
"type": "Note",
"pos": {
"0": 18.17154312133789,
"1": -312.8636474609375
},
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 87,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 252,
"1": -308
},
"size": {
"0": 441.4525451660156,
"1": 154
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
51
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5-12b-zh-InP",
"model_cpu_offload_and_qfloat8",
"Inpaint",
"easyanimate_video_v5_magvit_multi_text_encoder.yaml",
"bf16"
]
},
{
"id": 88,
"type": "EasyAnimateV5_T2VSampler",
"pos": {
"0": 786,
"1": 15
},
"size": {
"0": 327.6000061035156,
"1": 290
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 51,
"slot_index": 0
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 52,
"slot_index": 1
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 53,
"slot_index": 2
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
50
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_T2VSampler"
},
"widgets_values": [
49,
672,
384,
false,
43,
"fixed",
50,
6,
"DDIM"
]
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
52
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"一位年轻女子,有着美丽清澈的眼睛和金发,站在森林里,穿着白色的衣服,戴着皇冠。她似乎陷入了沉思,相机聚焦在她的脸上。质量高、杰作、最佳品质、高分辨率、超精细、梦幻般。"
]
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
53
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1148,
"1": 15
},
"size": [
390.9534912109375,
310
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 50,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00003.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 89,
"type": "Note",
"pos": {
"0": -97,
"1": 193
},
"size": [
326.1556114026207,
145.20905264909447
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
50,
88,
0,
17,
0,
"IMAGE"
],
[
51,
87,
0,
88,
0,
"EASYANIMATESMODEL"
],
[
52,
75,
0,
88,
1,
"STRING_PROMPT"
],
[
53,
73,
0,
88,
2,
"STRING_PROMPT"
]
],
"groups": [
{
"title": "Load EasyAnimate",
"bounding": [
218,
-393,
503,
254
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.7513148009015777,
"offset": [
206.91128312862938,
440.0942364134056
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
-559
View File
@@ -1,559 +0,0 @@
{
"last_node_id": 88,
"last_link_id": 52,
"nodes": [
{
"id": 80,
"type": "Note",
"pos": {
"0": 18.27766990661621,
"1": -307.4300537109375
},
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 79,
"type": "Note",
"pos": {
"0": 15.739953994750977,
"1": 462.38665771484375
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can upload video here\n(在此上传视频)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 238.2776641845703,
"1": -307.4300537109375
},
"size": {
"0": 482.8221435546875,
"1": 154
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
49
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5-12b-zh-InP",
"model_cpu_offload_and_qfloat8",
"Inpaint",
"easyanimate_video_v5_magvit_multi_text_encoder.yaml",
"bf16"
]
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
51
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
50
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"一只穿着小外套的猫咪正在花园秋千上安静地弹吉他。晚霞的余光洒在它柔软的毛皮上,和煦的微风轻轻拂过,周围斑驳的光影随着音乐的旋律轻轻摇曳。"
]
},
{
"id": 87,
"type": "EasyAnimateV5_V2VSampler",
"pos": {
"0": 816,
"1": 13
},
"size": {
"0": 336,
"1": 306
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 49
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 50,
"slot_index": 1
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 51,
"slot_index": 2
},
{
"name": "validation_video",
"type": "IMAGE",
"link": 52,
"slot_index": 3,
"shape": 7
},
{
"name": "control_video",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "ref_image",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "camera_conditions",
"type": "STRING",
"link": null,
"widget": {
"name": "camera_conditions"
},
"shape": 7
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
48
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_V2VSampler"
},
"widgets_values": [
49,
512,
43,
"fixed",
35,
7,
0.7,
"DDIM",
0.10,
true,
""
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1173,
"1": 15
},
"size": [
390.9534912109375,
310
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 48,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00005.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 85,
"type": "VHS_LoadVideo",
"pos": {
"0": 335,
"1": 476
},
"size": [
252.056640625,
262
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
52
],
"slot_index": 0,
"shape": 3
},
{
"name": "frame_count",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "audio",
"type": "AUDIO",
"links": null,
"shape": 3
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "1.mp4",
"force_rate": 8,
"force_size": "Disabled",
"custom_width": 512,
"custom_height": 512,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1,
"choose video to upload": "image",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"frame_load_cap": 0,
"skip_first_frames": 0,
"force_rate": 8,
"filename": "1.mp4",
"type": "input",
"format": "video/mp4",
"select_every_nth": 1
}
}
}
},
{
"id": 88,
"type": "Note",
"pos": {
"0": -97,
"1": 195
},
"size": [
326.1556091308594,
145.20904541015625
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
48,
87,
0,
17,
0,
"IMAGE"
],
[
49,
31,
0,
87,
0,
"EASYANIMATESMODEL"
],
[
50,
75,
0,
87,
1,
"STRING_PROMPT"
],
[
51,
73,
0,
87,
2,
"STRING_PROMPT"
],
[
52,
85,
0,
87,
3,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
},
{
"title": "Load EasyAnimate",
"bounding": [
218,
-387,
542,
248
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Upload Your Video",
"bounding": [
218,
385,
479,
529
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.7513148009015777,
"offset": [
206.91128312862938,
440.0942364134056
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
@@ -1,559 +0,0 @@
{
"last_node_id": 88,
"last_link_id": 53,
"nodes": [
{
"id": 80,
"type": "Note",
"pos": {
"0": 18.27766990661621,
"1": -307.4300537109375
},
"size": {
"0": 210,
"1": 66.98204040527344
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Load model here\n(在此选择要使用的模型)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 78,
"type": "Note",
"pos": {
"0": 18,
"1": -46
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can write prompt here\n(你可以在此填写提示词)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 79,
"type": "Note",
"pos": {
"0": 15.739953994750977,
"1": 462.38665771484375
},
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"You can upload video here\n(在此上传视频)"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 73,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": 160
},
"size": {
"0": 383.7149963378906,
"1": 183.83506774902344
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
51
],
"slot_index": 0,
"shape": 3
}
],
"title": "Negtive Prompt(反向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"扭曲的身体,肢体残缺,文本字幕,漫画,静止,丑陋,错误,乱码。"
]
},
{
"id": 17,
"type": "VHS_VideoCombine",
"pos": {
"0": 1173,
"1": 15
},
"size": [
390.9534912109375,
310
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 48,
"slot_index": 0,
"label": "图像",
"shape": 7
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"label": "音频",
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"label": "批次管理",
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null,
"slot_index": 0,
"shape": 3,
"label": "文件名"
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 8,
"loop_count": 0,
"filename_prefix": "EasyAnimate",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 22,
"save_metadata": true,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "EasyAnimate_00006.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 8
}
}
}
},
{
"id": 31,
"type": "LoadEasyAnimateModel",
"pos": {
"0": 238.2776641845703,
"1": -307.4300537109375
},
"size": {
"0": 482.8221435546875,
"1": 154
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"links": [
49
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadEasyAnimateModel"
},
"widgets_values": [
"EasyAnimateV5-12b-zh-Control",
"model_cpu_offload_and_qfloat8",
"Control",
"easyanimate_video_v5_magvit_multi_text_encoder.yaml",
"bf16"
]
},
{
"id": 85,
"type": "VHS_LoadVideo",
"pos": {
"0": 335,
"1": 476
},
"size": [
252.056640625,
262
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
53
],
"slot_index": 0,
"shape": 3
},
{
"name": "frame_count",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "audio",
"type": "AUDIO",
"links": null,
"shape": 3
},
{
"name": "video_info",
"type": "VHS_VIDEOINFO",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VHS_LoadVideo"
},
"widgets_values": {
"video": "pose.mp4",
"force_rate": 0,
"force_size": "Disabled",
"custom_width": 512,
"custom_height": 512,
"frame_load_cap": 0,
"skip_first_frames": 0,
"select_every_nth": 1,
"choose video to upload": "image",
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"frame_load_cap": 0,
"skip_first_frames": 0,
"force_rate": 0,
"filename": "pose.mp4",
"type": "input",
"format": "video/mp4",
"select_every_nth": 1
}
}
}
},
{
"id": 75,
"type": "EasyAnimate_TextBox",
"pos": {
"0": 250,
"1": -50
},
"size": {
"0": 383.54010009765625,
"1": 156.71620178222656
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "prompt",
"type": "STRING_PROMPT",
"links": [
50
],
"slot_index": 0,
"shape": 3
}
],
"title": "Positive Prompt(正向提示词)",
"properties": {
"Node name for S&R": "EasyAnimate_TextBox"
},
"widgets_values": [
"一个穿着及膝白色无袖连衣裙和白色高跟凉鞋的美女在一个光线充足、木地板的房间里跳舞。房间的背景是一扇紧闭的门、一个展示透明玻璃瓶酒精饮料的架子和一个部分可见的深色沙发。"
]
},
{
"id": 87,
"type": "EasyAnimateV5_V2VSampler",
"pos": {
"0": 816,
"1": 13
},
"size": {
"0": 336,
"1": 306
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "easyanimate_model",
"type": "EASYANIMATESMODEL",
"link": 49
},
{
"name": "prompt",
"type": "STRING_PROMPT",
"link": 50,
"slot_index": 1
},
{
"name": "negative_prompt",
"type": "STRING_PROMPT",
"link": 51,
"slot_index": 2
},
{
"name": "validation_video",
"type": "IMAGE",
"link": null,
"slot_index": 3,
"shape": 7
},
{
"name": "control_video",
"type": "IMAGE",
"link": 53,
"shape": 7
},
{
"name": "ref_image",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "camera_conditions",
"type": "STRING",
"link": null,
"widget": {
"name": "camera_conditions"
},
"shape": 7
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
48
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "EasyAnimateV5_V2VSampler"
},
"widgets_values": [
49,
512,
43,
"fixed",
35,
6,
1,
"DDIM",
0.10,
true,
""
]
},
{
"id": 88,
"type": "Note",
"pos": {
"0": -99,
"1": 197
},
"size": [
326.1556114026207,
145.20905264909447
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {
"text": ""
},
"widgets_values": [
"Using longer neg prompt such as \"Blurring, mutation, deformation, distortion, dark and solid, comics.\" can increase stability. Adding words such as \"quiet, solid\" to the neg prompt can increase dynamism.\n(使用更长的neg prompt如\"模糊,突变,变形,失真,画面暗,画面固定,连环画,漫画,线稿,没有主体。\",可以增加稳定性。在neg prompt中添加\"安静,固定\"等词语可以增加动态性。)"
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
[
48,
87,
0,
17,
0,
"IMAGE"
],
[
49,
31,
0,
87,
0,
"EASYANIMATESMODEL"
],
[
50,
75,
0,
87,
1,
"STRING_PROMPT"
],
[
51,
73,
0,
87,
2,
"STRING_PROMPT"
],
[
53,
85,
0,
87,
4,
"IMAGE"
]
],
"groups": [
{
"title": "Prompts",
"bounding": [
218,
-127,
450,
483
],
"color": "#3f789e",
"font_size": 24,
"flags": {}
},
{
"title": "Load EasyAnimate",
"bounding": [
218,
-387,
542,
248
],
"color": "#b06634",
"font_size": 24,
"flags": {}
},
{
"title": "Upload Your Video",
"bounding": [
218,
385,
487,
789
],
"color": "#a1309b",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.7513148009015777,
"offset": [
206.91128312862938,
440.0942364134056
]
},
"workspace_info": {
"id": "776b62b4-bd17-4ed3-9923-b7aad000b1ea"
}
},
"version": 0.4
}
+8
View File
@@ -0,0 +1,8 @@
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
vae_kwargs:
enable_magvit: true
+8
View File
@@ -0,0 +1,8 @@
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
vae_kwargs:
enable_magvit: false
@@ -0,0 +1,9 @@
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
vae_kwargs:
enable_magvit: true
slice_compression_vae: true
@@ -0,0 +1,27 @@
transformer_additional_kwargs:
patch_3d: false
fake_3d: false
casual_3d: true
casual_3d_upsampler_index: [16, 20]
time_patch_size: 4
basic_block_type: "motionmodule"
time_position_encoding_before_transformer: false
motion_module_type: "VanillaGrid"
motion_module_kwargs:
num_attention_heads: 8
num_transformer_block: 1
attention_block_types: [ "Temporal_Self", "Temporal_Self" ]
temporal_position_encoding: true
temporal_position_encoding_max_len: 4096
temporal_attention_dim_div: 1
block_size: 2
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
vae_kwargs:
enable_magvit: false
@@ -0,0 +1,14 @@
transformer_additional_kwargs:
patch_3d: false
fake_3d: false
basic_block_type: "selfattentiontemporal"
time_position_encoding_before_transformer: true
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
vae_kwargs:
enable_magvit: false
@@ -1,5 +1,4 @@
transformer_additional_kwargs:
transformer_type: "Transformer3DModel"
patch_3d: false
fake_3d: false
basic_block_type: "motionmodule"
@@ -16,14 +15,12 @@ transformer_additional_kwargs:
temporal_attention_dim_div: 1
block_size: 1
vae_kwargs:
vae_type: "AutoencoderKLMagvit"
mini_batch_encoder: 9
mini_batch_decoder: 3
slice_mag_vae: true
slice_compression_vae: false
cache_compression_vae: false
cache_mag_vae: false
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
text_encoder_kwargs:
enable_multi_text_encoder: false
vae_kwargs:
enable_magvit: true
mini_batch_encoder: 9
@@ -1,5 +1,4 @@
transformer_additional_kwargs:
transformer_type: "Transformer3DModel"
patch_3d: false
fake_3d: false
basic_block_type: "motionmodule"
@@ -15,8 +14,11 @@ transformer_additional_kwargs:
temporal_attention_dim_div: 1
block_size: 2
vae_kwargs:
vae_type: "AutoencoderKL"
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
text_encoder_kwargs:
enable_multi_text_encoder: false
vae_kwargs:
enable_magvit: false
@@ -0,0 +1,27 @@
transformer_additional_kwargs:
patch_3d: false
fake_3d: false
basic_block_type: "motionmodule"
time_position_encoding_before_transformer: false
motion_module_type: "Vanilla"
enable_uvit: true
motion_module_kwargs:
num_attention_heads: 8
num_transformer_block: 1
attention_block_types: [ "Temporal_Self", "Temporal_Self" ]
temporal_position_encoding: true
temporal_position_encoding_max_len: 4096
temporal_attention_dim_div: 1
block_size: 1
noise_scheduler_kwargs:
beta_start: 0.0001
beta_end: 0.02
beta_schedule: "linear"
steps_offset: 1
vae_kwargs:
enable_magvit: true
slice_compression_vae: true
mini_batch_encoder: 8
@@ -1,39 +0,0 @@
transformer_additional_kwargs:
transformer_type: "Transformer3DModel"
patch_3d: false
fake_3d: false
basic_block_type: "global_motionmodule"
time_position_encoding_before_transformer: false
motion_module_type: "Vanilla"
enable_uvit: true
motion_module_kwargs_even:
num_attention_heads: 16
num_transformer_block: 1
attention_block_types: [ "Temporal_Self", "Temporal_Self" ]
temporal_position_encoding: true
temporal_position_encoding_max_len: 4096
temporal_attention_dim_div: 1
block_size: 1
remove_time_embedding_in_photo: false
motion_module_kwargs_odd:
num_attention_heads: 16
num_transformer_block: 1
attention_block_types: [ "Temporal_Self", "Global_Self" ]
temporal_position_encoding: true
temporal_position_encoding_max_len: 4096
temporal_attention_dim_div: 1
block_size: 1
remove_time_embedding_in_photo: false
vae_kwargs:
vae_type: "AutoencoderKLMagvit"
mini_batch_encoder: 8
mini_batch_decoder: 2
slice_mag_vae: false
slice_compression_vae: true
cache_compression_vae: false
cache_mag_vae: false
text_encoder_kwargs:
enable_multi_text_encoder: false
@@ -1,20 +0,0 @@
transformer_additional_kwargs:
transformer_type: "HunyuanTransformer3DModel"
basic_block_type: "basic"
after_norm: false
time_position_encoding_type: "2d_rope"
time_position_encoding: true
resize_inpaint_mask_directly: false
enable_clip_in_inpaint: true
vae_kwargs:
vae_type: "AutoencoderKLMagvit"
mini_batch_encoder: 8
mini_batch_decoder: 2
slice_mag_vae: false
slice_compression_vae: false
cache_compression_vae: true
cache_mag_vae: false
text_encoder_kwargs:
enable_multi_text_encoder: true
@@ -1,21 +0,0 @@
transformer_additional_kwargs:
transformer_type: "EasyAnimateTransformer3DModel"
after_norm: false
time_position_encoding_type: "3d_rope"
resize_inpaint_mask_directly: true
enable_text_attention_mask: true
enable_clip_in_inpaint: false
add_ref_latent_in_control_model: true
vae_kwargs:
vae_type: "AutoencoderKLMagvit"
mini_batch_encoder: 4
mini_batch_decoder: 1
slice_mag_vae: false
slice_compression_vae: false
cache_compression_vae: false
cache_mag_vae: true
text_encoder_kwargs:
enable_multi_text_encoder: false
replace_t5_to_llm: true
@@ -1,19 +0,0 @@
transformer_additional_kwargs:
transformer_type: "EasyAnimateTransformer3DModel"
after_norm: false
time_position_encoding_type: "3d_rope"
resize_inpaint_mask_directly: true
enable_text_attention_mask: false
enable_clip_in_inpaint: false
vae_kwargs:
vae_type: "AutoencoderKLMagvit"
mini_batch_encoder: 4
mini_batch_decoder: 1
slice_mag_vae: false
slice_compression_vae: false
cache_compression_vae: false
cache_mag_vae: true
text_encoder_kwargs:
enable_multi_text_encoder: true
-16
View File
@@ -1,16 +0,0 @@
{
"bf16": {
"enabled": true
},
"train_micro_batch_size_per_gpu": 1,
"train_batch_size": "auto",
"gradient_accumulation_steps": "auto",
"dump_state": true,
"zero_optimization": {
"stage": 2,
"overlap_comm": true,
"contiguous_gradients": true,
"sub_group_size": 1e9,
"reduce_bucket_size": 5e8
}
}
+9 -89
View File
@@ -1,16 +1,10 @@
import base64
import gc
import hashlib
import io
import os
import tempfile
from io import BytesIO
import gradio as gr
import base64
import torch
from fastapi import FastAPI
from PIL import Image
import gradio as gr
from fastapi import FastAPI
from io import BytesIO
# Function to encode a file to Base64
def encode_file_to_base64(file_path):
@@ -55,34 +49,6 @@ def update_diffusion_transformer_api(_: gr.Blocks, app: FastAPI, controller):
return {"message": comment}
def save_base64_video(base64_string):
video_data = base64.b64decode(base64_string)
md5_hash = hashlib.md5(video_data).hexdigest()
filename = f"{md5_hash}.mp4"
temp_dir = tempfile.gettempdir()
file_path = os.path.join(temp_dir, filename)
with open(file_path, 'wb') as video_file:
video_file.write(video_data)
return file_path
def save_base64_image(base64_string):
video_data = base64.b64decode(base64_string)
md5_hash = hashlib.md5(video_data).hexdigest()
filename = f"{md5_hash}.jpg"
temp_dir = tempfile.gettempdir()
file_path = os.path.join(temp_dir, filename)
with open(file_path, 'wb') as video_file:
video_file.write(video_data)
return file_path
def infer_forward_api(_: gr.Blocks, app: FastAPI, controller):
@app.post("/easyanimate/infer_forward")
def _infer_forward_api(
@@ -93,46 +59,16 @@ def infer_forward_api(_: gr.Blocks, app: FastAPI, controller):
lora_model_path = datas.get('lora_model_path', 'none')
lora_alpha_slider = datas.get('lora_alpha_slider', 0.55)
prompt_textbox = datas.get('prompt_textbox', None)
negative_prompt_textbox = datas.get('negative_prompt_textbox', 'Blurring, mutation, deformation, distortion, dark and solid, comics, text subtitles, line art.')
negative_prompt_textbox = datas.get('negative_prompt_textbox', '')
sampler_dropdown = datas.get('sampler_dropdown', 'Euler')
sample_step_slider = datas.get('sample_step_slider', 30)
resize_method = datas.get('resize_method', "Generate by")
width_slider = datas.get('width_slider', 672)
height_slider = datas.get('height_slider', 384)
base_resolution = datas.get('base_resolution', 512)
is_image = datas.get('is_image', False)
generation_method = datas.get('generation_method', False)
length_slider = datas.get('length_slider', 49)
overlap_video_length = datas.get('overlap_video_length', 4)
partial_video_length = datas.get('partial_video_length', 72)
length_slider = datas.get('length_slider', 144)
cfg_scale_slider = datas.get('cfg_scale_slider', 6)
start_image = datas.get('start_image', None)
end_image = datas.get('end_image', None)
validation_video = datas.get('validation_video', None)
validation_video_mask = datas.get('validation_video_mask', None)
control_video = datas.get('control_video', None)
denoise_strength = datas.get('denoise_strength', 0.70)
seed_textbox = datas.get("seed_textbox", 43)
generation_method = "Image Generation" if is_image else generation_method
if start_image is not None:
start_image = base64.b64decode(start_image)
start_image = [Image.open(BytesIO(start_image))]
if end_image is not None:
end_image = base64.b64decode(end_image)
end_image = [Image.open(BytesIO(end_image))]
if validation_video is not None:
validation_video = save_base64_video(validation_video)
if validation_video_mask is not None:
validation_video_mask = save_base64_image(validation_video_mask)
if control_video is not None:
control_video = save_base64_video(control_video)
try:
save_sample_path, comment = controller.generate(
"",
@@ -144,33 +80,17 @@ def infer_forward_api(_: gr.Blocks, app: FastAPI, controller):
negative_prompt_textbox,
sampler_dropdown,
sample_step_slider,
resize_method,
width_slider,
height_slider,
base_resolution,
generation_method,
is_image,
length_slider,
overlap_video_length,
partial_video_length,
cfg_scale_slider,
start_image,
end_image,
validation_video,
validation_video_mask,
control_video,
denoise_strength,
seed_textbox,
is_api = True,
)
except Exception as e:
gc.collect()
torch.cuda.empty_cache()
torch.cuda.ipc_collect()
save_sample_path = ""
comment = f"Error. error information is {str(e)}"
return {"message": comment}
if save_sample_path != "":
return {"message": comment, "save_sample_path": save_sample_path, "base64_encoding": encode_file_to_base64(save_sample_path)}
else:
return {"message": comment, "save_sample_path": save_sample_path}
return {"message": comment, "save_sample_path": save_sample_path, "base64_encoding": encode_file_to_base64(save_sample_path)}
+8 -9
View File
@@ -7,6 +7,7 @@ from io import BytesIO
import cv2
import requests
import base64
def post_diffusion_transformer(diffusion_transformer_path, url='http://127.0.0.1:7860'):
@@ -25,7 +26,7 @@ def post_update_edition(edition, url='http://0.0.0.0:7860'):
data = r.content.decode('utf-8')
return data
def post_infer(generation_method, length_slider, url='http://127.0.0.1:7860'):
def post_infer(is_image, length_slider, url='http://127.0.0.1:7860'):
datas = json.dumps({
"base_model_path": "none",
"motion_module_path": "none",
@@ -37,7 +38,7 @@ def post_infer(generation_method, length_slider, url='http://127.0.0.1:7860'):
"sample_step_slider": 30,
"width_slider": 672,
"height_slider": 384,
"generation_method": "Video Generation",
"is_image": is_image,
"length_slider": length_slider,
"cfg_scale_slider": 6,
"seed_textbox": 43,
@@ -54,31 +55,29 @@ if __name__ == '__main__':
# -------------------------- #
# Step 1: update edition
# -------------------------- #
edition = "v5.1"
edition = "v2"
outputs = post_update_edition(edition)
print('Output update edition: ', outputs)
# -------------------------- #
# Step 2: update edition
# -------------------------- #
diffusion_transformer_path = "models/Diffusion_Transformer/EasyAnimateV5.1-12b-zh-InP"
diffusion_transformer_path = "/your-path/EasyAnimate/models/Diffusion_Transformer/EasyAnimateV2-XL-2-512x512"
outputs = post_diffusion_transformer(diffusion_transformer_path)
print('Output update edition: ', outputs)
# -------------------------- #
# Step 3: infer
# -------------------------- #
# "Video Generation" and "Image Generation"
generation_method = "Video Generation"
length_slider = 21
outputs = post_infer(generation_method, length_slider)
is_image = False
length_slider = 27
outputs = post_infer(is_image, length_slider)
# Get decoded data
outputs = json.loads(outputs)
base64_encoding = outputs["base64_encoding"]
decoded_data = base64.b64decode(base64_encoding)
is_image = True if generation_method == "Image Generation" else False
if is_image or length_slider == 1:
file_path = "1.png"
else:
+3 -3
View File
@@ -254,7 +254,7 @@ class AspectRatioBatchSampler(BatchSampler):
width = int(width)
ratio = height / width # self.dataset[idx]
except Exception as e:
print(e, self.dataset[idx], "This item is error, please check it.")
print(e)
continue
# find the closest aspect ratio
closest_ratio = min(self.aspect_ratios.keys(), key=lambda r: abs(float(r) - ratio))
@@ -330,7 +330,7 @@ class AspectRatioBatchImageVideoSampler(BatchSampler):
width = int(width)
ratio = height / width # self.dataset[idx]
except Exception as e:
print(e, self.dataset[idx], "This item is error, please check it.")
print(e)
continue
# find the closest aspect ratio
closest_ratio = min(self.aspect_ratios.keys(), key=lambda r: abs(float(r) - ratio))
@@ -365,7 +365,7 @@ class AspectRatioBatchImageVideoSampler(BatchSampler):
width = int(width)
ratio = height / width # self.dataset[idx]
except Exception as e:
print(e, self.dataset[idx], "This item is error, please check it.")
print(e)
continue
# find the closest aspect ratio
closest_ratio = min(self.aspect_ratios.keys(), key=lambda r: abs(float(r) - ratio))
+20 -558
View File
@@ -1,255 +1,26 @@
import csv
import gc
import io
import json
import math
import os
import random
from contextlib import contextmanager
from threading import Thread
import albumentations
import cv2
import gc
import numpy as np
import torch
import torch.nn.functional as F
import torchvision.transforms as transforms
from func_timeout import func_timeout, FunctionTimedOut
from decord import VideoReader
from einops import rearrange
from func_timeout import FunctionTimedOut, func_timeout
from packaging import version as pver
from PIL import Image
from torch.utils.data import BatchSampler, Sampler
from torch.utils.data.dataset import Dataset
from contextlib import contextmanager
VIDEO_READER_TIMEOUT = 20
def get_random_mask(shape):
f, c, h, w = shape
if f != 1:
mask_index = np.random.choice([0, 1, 2, 3, 4, 5, 6, 7, 8, 9], p=[0.05, 0.2, 0.2, 0.2, 0.05, 0.05, 0.05, 0.1, 0.05, 0.05])
else:
mask_index = np.random.choice([0, 1], p = [0.2, 0.8])
mask = torch.zeros((f, 1, h, w), dtype=torch.uint8)
if mask_index == 0:
center_x = torch.randint(0, w, (1,)).item()
center_y = torch.randint(0, h, (1,)).item()
block_size_x = torch.randint(w // 4, w // 4 * 3, (1,)).item() # 方块的宽度范围
block_size_y = torch.randint(h // 4, h // 4 * 3, (1,)).item() # 方块的高度范围
start_x = max(center_x - block_size_x // 2, 0)
end_x = min(center_x + block_size_x // 2, w)
start_y = max(center_y - block_size_y // 2, 0)
end_y = min(center_y + block_size_y // 2, h)
mask[:, :, start_y:end_y, start_x:end_x] = 1
elif mask_index == 1:
mask[:, :, :, :] = 1
elif mask_index == 2:
mask_frame_index = np.random.randint(1, 5)
mask[mask_frame_index:, :, :, :] = 1
elif mask_index == 3:
mask_frame_index = np.random.randint(1, 5)
mask[mask_frame_index:-mask_frame_index, :, :, :] = 1
elif mask_index == 4:
center_x = torch.randint(0, w, (1,)).item()
center_y = torch.randint(0, h, (1,)).item()
block_size_x = torch.randint(w // 4, w // 4 * 3, (1,)).item() # 方块的宽度范围
block_size_y = torch.randint(h // 4, h // 4 * 3, (1,)).item() # 方块的高度范围
start_x = max(center_x - block_size_x // 2, 0)
end_x = min(center_x + block_size_x // 2, w)
start_y = max(center_y - block_size_y // 2, 0)
end_y = min(center_y + block_size_y // 2, h)
mask_frame_before = np.random.randint(0, f // 2)
mask_frame_after = np.random.randint(f // 2, f)
mask[mask_frame_before:mask_frame_after, :, start_y:end_y, start_x:end_x] = 1
elif mask_index == 5:
mask = torch.randint(0, 2, (f, 1, h, w), dtype=torch.uint8)
elif mask_index == 6:
num_frames_to_mask = random.randint(1, max(f // 2, 1))
frames_to_mask = random.sample(range(f), num_frames_to_mask)
for i in frames_to_mask:
block_height = random.randint(1, h // 4)
block_width = random.randint(1, w // 4)
top_left_y = random.randint(0, h - block_height)
top_left_x = random.randint(0, w - block_width)
mask[i, 0, top_left_y:top_left_y + block_height, top_left_x:top_left_x + block_width] = 1
elif mask_index == 7:
center_x = torch.randint(0, w, (1,)).item()
center_y = torch.randint(0, h, (1,)).item()
a = torch.randint(min(w, h) // 8, min(w, h) // 4, (1,)).item() # 长半轴
b = torch.randint(min(h, w) // 8, min(h, w) // 4, (1,)).item() # 短半轴
for i in range(h):
for j in range(w):
if ((i - center_y) ** 2) / (b ** 2) + ((j - center_x) ** 2) / (a ** 2) < 1:
mask[:, :, i, j] = 1
elif mask_index == 8:
center_x = torch.randint(0, w, (1,)).item()
center_y = torch.randint(0, h, (1,)).item()
radius = torch.randint(min(h, w) // 8, min(h, w) // 4, (1,)).item()
for i in range(h):
for j in range(w):
if (i - center_y) ** 2 + (j - center_x) ** 2 < radius ** 2:
mask[:, :, i, j] = 1
elif mask_index == 9:
for idx in range(f):
if np.random.rand() > 0.5:
mask[idx, :, :, :] = 1
else:
raise ValueError(f"The mask_index {mask_index} is not define")
return mask
class Camera(object):
"""Copied from https://github.com/hehao13/CameraCtrl/blob/main/inference.py
"""
def __init__(self, entry):
fx, fy, cx, cy = entry[1:5]
self.fx = fx
self.fy = fy
self.cx = cx
self.cy = cy
w2c_mat = np.array(entry[7:]).reshape(3, 4)
w2c_mat_4x4 = np.eye(4)
w2c_mat_4x4[:3, :] = w2c_mat
self.w2c_mat = w2c_mat_4x4
self.c2w_mat = np.linalg.inv(w2c_mat_4x4)
def custom_meshgrid(*args):
"""Copied from https://github.com/hehao13/CameraCtrl/blob/main/inference.py
"""
# ref: https://pytorch.org/docs/stable/generated/torch.meshgrid.html?highlight=meshgrid#torch.meshgrid
if pver.parse(torch.__version__) < pver.parse('1.10'):
return torch.meshgrid(*args)
else:
return torch.meshgrid(*args, indexing='ij')
def get_relative_pose(cam_params):
"""Copied from https://github.com/hehao13/CameraCtrl/blob/main/inference.py
"""
abs_w2cs = [cam_param.w2c_mat for cam_param in cam_params]
abs_c2ws = [cam_param.c2w_mat for cam_param in cam_params]
cam_to_origin = 0
target_cam_c2w = np.array([
[1, 0, 0, 0],
[0, 1, 0, -cam_to_origin],
[0, 0, 1, 0],
[0, 0, 0, 1]
])
abs2rel = target_cam_c2w @ abs_w2cs[0]
ret_poses = [target_cam_c2w, ] + [abs2rel @ abs_c2w for abs_c2w in abs_c2ws[1:]]
ret_poses = np.array(ret_poses, dtype=np.float32)
return ret_poses
def ray_condition(K, c2w, H, W, device):
"""Copied from https://github.com/hehao13/CameraCtrl/blob/main/inference.py
"""
# c2w: B, V, 4, 4
# K: B, V, 4
B = K.shape[0]
j, i = custom_meshgrid(
torch.linspace(0, H - 1, H, device=device, dtype=c2w.dtype),
torch.linspace(0, W - 1, W, device=device, dtype=c2w.dtype),
)
i = i.reshape([1, 1, H * W]).expand([B, 1, H * W]) + 0.5 # [B, HxW]
j = j.reshape([1, 1, H * W]).expand([B, 1, H * W]) + 0.5 # [B, HxW]
fx, fy, cx, cy = K.chunk(4, dim=-1) # B,V, 1
zs = torch.ones_like(i) # [B, HxW]
xs = (i - cx) / fx * zs
ys = (j - cy) / fy * zs
zs = zs.expand_as(ys)
directions = torch.stack((xs, ys, zs), dim=-1) # B, V, HW, 3
directions = directions / directions.norm(dim=-1, keepdim=True) # B, V, HW, 3
rays_d = directions @ c2w[..., :3, :3].transpose(-1, -2) # B, V, 3, HW
rays_o = c2w[..., :3, 3] # B, V, 3
rays_o = rays_o[:, :, None].expand_as(rays_d) # B, V, 3, HW
# c2w @ dirctions
rays_dxo = torch.cross(rays_o, rays_d)
plucker = torch.cat([rays_dxo, rays_d], dim=-1)
plucker = plucker.reshape(B, c2w.shape[1], H, W, 6) # B, V, H, W, 6
# plucker = plucker.permute(0, 1, 4, 2, 3)
return plucker
def process_pose_file(pose_file_path, width=672, height=384, original_pose_width=1280, original_pose_height=720, device='cpu', return_poses=False):
"""Modified from https://github.com/hehao13/CameraCtrl/blob/main/inference.py
"""
with open(pose_file_path, 'r') as f:
poses = f.readlines()
poses = [pose.strip().split(' ') for pose in poses[1:]]
cam_params = [[float(x) for x in pose] for pose in poses]
if return_poses:
return cam_params
else:
cam_params = [Camera(cam_param) for cam_param in cam_params]
sample_wh_ratio = width / height
pose_wh_ratio = original_pose_width / original_pose_height # Assuming placeholder ratios, change as needed
if pose_wh_ratio > sample_wh_ratio:
resized_ori_w = height * pose_wh_ratio
for cam_param in cam_params:
cam_param.fx = resized_ori_w * cam_param.fx / width
else:
resized_ori_h = width / pose_wh_ratio
for cam_param in cam_params:
cam_param.fy = resized_ori_h * cam_param.fy / height
intrinsic = np.asarray([[cam_param.fx * width,
cam_param.fy * height,
cam_param.cx * width,
cam_param.cy * height]
for cam_param in cam_params], dtype=np.float32)
K = torch.as_tensor(intrinsic)[None] # [1, 1, 4]
c2ws = get_relative_pose(cam_params) # Assuming this function is defined elsewhere
c2ws = torch.as_tensor(c2ws)[None] # [1, n_frame, 4, 4]
plucker_embedding = ray_condition(K, c2ws, height, width, device=device)[0].permute(0, 3, 1, 2).contiguous() # V, 6, H, W
plucker_embedding = plucker_embedding[None]
plucker_embedding = rearrange(plucker_embedding, "b f c h w -> b f h w c")[0]
return plucker_embedding
def process_pose_params(cam_params, width=672, height=384, original_pose_width=1280, original_pose_height=720, device='cpu'):
"""Modified from https://github.com/hehao13/CameraCtrl/blob/main/inference.py
"""
cam_params = [Camera(cam_param) for cam_param in cam_params]
sample_wh_ratio = width / height
pose_wh_ratio = original_pose_width / original_pose_height # Assuming placeholder ratios, change as needed
if pose_wh_ratio > sample_wh_ratio:
resized_ori_w = height * pose_wh_ratio
for cam_param in cam_params:
cam_param.fx = resized_ori_w * cam_param.fx / width
else:
resized_ori_h = width / pose_wh_ratio
for cam_param in cam_params:
cam_param.fy = resized_ori_h * cam_param.fy / height
intrinsic = np.asarray([[cam_param.fx * width,
cam_param.fy * height,
cam_param.cx * width,
cam_param.cy * height]
for cam_param in cam_params], dtype=np.float32)
K = torch.as_tensor(intrinsic)[None] # [1, 1, 4]
c2ws = get_relative_pose(cam_params) # Assuming this function is defined elsewhere
c2ws = torch.as_tensor(c2ws)[None] # [1, n_frame, 4, 4]
plucker_embedding = ray_condition(K, c2ws, height, width, device=device)[0].permute(0, 3, 1, 2).contiguous() # V, 6, H, W
plucker_embedding = plucker_embedding[None]
plucker_embedding = rearrange(plucker_embedding, "b f c h w -> b f h w c")[0]
return plucker_embedding
class ImageVideoSampler(BatchSampler):
"""A sampler wrapper for grouping images with similar aspect ratio into a same batch.
@@ -310,35 +81,18 @@ def get_video_reader_batch(video_reader, batch_index):
frames = video_reader.get_batch(batch_index).asnumpy()
return frames
def resize_frame(frame, target_short_side):
h, w, _ = frame.shape
if h < w:
if target_short_side > h:
return frame
new_h = target_short_side
new_w = int(target_short_side * w / h)
else:
if target_short_side > w:
return frame
new_w = target_short_side
new_h = int(target_short_side * h / w)
resized_frame = cv2.resize(frame, (new_w, new_h))
return resized_frame
class ImageVideoDataset(Dataset):
def __init__(
self,
ann_path, data_root=None,
video_sample_size=512, video_sample_stride=4, video_sample_n_frames=16,
image_sample_size=512,
video_repeat=0,
text_drop_ratio=0.1,
enable_bucket=False,
video_length_drop_start=0.1,
video_length_drop_end=0.9,
enable_inpaint=False,
):
self,
ann_path, data_root=None,
video_sample_size=512, video_sample_stride=4, video_sample_n_frames=16,
image_sample_size=512,
video_repeat=0,
text_drop_ratio=0.001,
enable_bucket=False,
video_length_drop_start=0.1,
video_length_drop_end=0.9,
):
# Loading annotations from files
print(f"loading annotations from {ann_path} ...")
if ann_path.endswith('.csv'):
@@ -366,19 +120,17 @@ class ImageVideoDataset(Dataset):
# TODO: enable bucket training
self.enable_bucket = enable_bucket
self.text_drop_ratio = text_drop_ratio
self.enable_inpaint = enable_inpaint
self.video_length_drop_start = video_length_drop_start
self.video_length_drop_end = video_length_drop_end
# Video params
self.video_sample_stride = video_sample_stride
self.video_sample_n_frames = video_sample_n_frames
self.video_sample_size = tuple(video_sample_size) if not isinstance(video_sample_size, int) else (video_sample_size, video_sample_size)
video_sample_size = tuple(video_sample_size) if not isinstance(video_sample_size, int) else (video_sample_size, video_sample_size)
self.video_transforms = transforms.Compose(
[
transforms.Resize(min(self.video_sample_size)),
transforms.CenterCrop(self.video_sample_size),
transforms.Resize(video_sample_size[0]),
transforms.CenterCrop(video_sample_size),
transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5], inplace=True),
]
)
@@ -391,9 +143,7 @@ class ImageVideoDataset(Dataset):
transforms.ToTensor(),
transforms.Normalize([0.5, 0.5, 0.5],[0.5, 0.5, 0.5])
])
self.larger_side_of_image_and_video = max(min(self.image_sample_size), min(self.video_sample_size))
def get_batch(self, idx):
data_info = self.dataset[idx % len(self.dataset)]
@@ -408,14 +158,14 @@ class ImageVideoDataset(Dataset):
with VideoReader_contextmanager(video_dir, num_threads=2) as video_reader:
min_sample_n_frames = min(
self.video_sample_n_frames,
int(len(video_reader) * (self.video_length_drop_end - self.video_length_drop_start) // self.video_sample_stride)
int(len(video_reader) * (self.video_length_drop_end - self.video_length_drop_start))
)
if min_sample_n_frames == 0:
raise ValueError(f"No Frames in video.")
video_length = int(self.video_length_drop_end * len(video_reader))
clip_length = min(video_length, (min_sample_n_frames - 1) * self.video_sample_stride + 1)
start_idx = random.randint(int(self.video_length_drop_start * video_length), video_length - clip_length) if video_length != clip_length else 0
start_idx = random.randint(int(self.video_length_drop_start * video_length), video_length - clip_length)
batch_index = np.linspace(start_idx, start_idx + clip_length - 1, min_sample_n_frames, dtype=int)
try:
@@ -423,12 +173,6 @@ class ImageVideoDataset(Dataset):
pixel_values = func_timeout(
VIDEO_READER_TIMEOUT, get_video_reader_batch, args=sample_args
)
resized_frames = []
for i in range(len(pixel_values)):
frame = pixel_values[i]
resized_frame = resize_frame(frame, self.larger_side_of_image_and_video)
resized_frames.append(resized_frame)
pixel_values = np.array(resized_frames)
except FunctionTimedOut:
raise ValueError(f"Read {idx} timeout.")
except Exception as e:
@@ -486,288 +230,6 @@ class ImageVideoDataset(Dataset):
except Exception as e:
print(e, self.dataset[idx % len(self.dataset)])
idx = random.randint(0, self.length-1)
if self.enable_inpaint and not self.enable_bucket:
mask = get_random_mask(pixel_values.size())
mask_pixel_values = pixel_values * (1 - mask) + torch.ones_like(pixel_values) * -1 * mask
sample["mask_pixel_values"] = mask_pixel_values
sample["mask"] = mask
clip_pixel_values = sample["pixel_values"][0].permute(1, 2, 0).contiguous()
clip_pixel_values = (clip_pixel_values * 0.5 + 0.5) * 255
sample["clip_pixel_values"] = clip_pixel_values
ref_pixel_values = sample["pixel_values"][0].unsqueeze(0)
if (mask == 1).all():
ref_pixel_values = torch.ones_like(ref_pixel_values) * -1
sample["ref_pixel_values"] = ref_pixel_values
return sample
class ImageVideoControlDataset(Dataset):
def __init__(
self,
ann_path, data_root=None,
video_sample_size=512, video_sample_stride=4, video_sample_n_frames=16,
image_sample_size=512,
video_repeat=0,
text_drop_ratio=0.1,
enable_bucket=False,
video_length_drop_start=0.1,
video_length_drop_end=0.9,
enable_inpaint=False,
enable_camera_info=False,
):
# Loading annotations from files
print(f"loading annotations from {ann_path} ...")
if ann_path.endswith('.csv'):
with open(ann_path, 'r') as csvfile:
dataset = list(csv.DictReader(csvfile))
elif ann_path.endswith('.json'):
dataset = json.load(open(ann_path))
self.data_root = data_root
# It's used to balance num of images and videos.
self.dataset = []
for data in dataset:
if data.get('type', 'image') != 'video':
self.dataset.append(data)
if video_repeat > 0:
for _ in range(video_repeat):
for data in dataset:
if data.get('type', 'image') == 'video':
self.dataset.append(data)
del dataset
self.length = len(self.dataset)
print(f"data scale: {self.length}")
# TODO: enable bucket training
self.enable_bucket = enable_bucket
self.text_drop_ratio = text_drop_ratio
self.enable_inpaint = enable_inpaint
self.enable_camera_info = enable_camera_info
self.video_length_drop_start = video_length_drop_start
self.video_length_drop_end = video_length_drop_end
# Video params
self.video_sample_stride = video_sample_stride
self.video_sample_n_frames = video_sample_n_frames
self.video_sample_size = tuple(video_sample_size) if not isinstance(video_sample_size, int) else (video_sample_size, video_sample_size)
self.video_transforms = transforms.Compose(
[
transforms.Resize(min(self.video_sample_size)),
transforms.CenterCrop(self.video_sample_size),
transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5], inplace=True),
]
)
if self.enable_camera_info:
self.video_transforms_camera = transforms.Compose(
[
transforms.Resize(min(self.video_sample_size)),
transforms.CenterCrop(self.video_sample_size)
]
)
# Image params
self.image_sample_size = tuple(image_sample_size) if not isinstance(image_sample_size, int) else (image_sample_size, image_sample_size)
self.image_transforms = transforms.Compose([
transforms.Resize(min(self.image_sample_size)),
transforms.CenterCrop(self.image_sample_size),
transforms.ToTensor(),
transforms.Normalize([0.5, 0.5, 0.5],[0.5, 0.5, 0.5])
])
self.larger_side_of_image_and_video = max(min(self.image_sample_size), min(self.video_sample_size))
def get_batch(self, idx):
data_info = self.dataset[idx % len(self.dataset)]
video_id, text = data_info['file_path'], data_info['text']
if data_info.get('type', 'image')=='video':
if self.data_root is None:
video_dir = video_id
else:
video_dir = os.path.join(self.data_root, video_id)
with VideoReader_contextmanager(video_dir, num_threads=2) as video_reader:
min_sample_n_frames = min(
self.video_sample_n_frames,
int(len(video_reader) * (self.video_length_drop_end - self.video_length_drop_start) // self.video_sample_stride)
)
if min_sample_n_frames == 0:
raise ValueError(f"No Frames in video.")
video_length = int(self.video_length_drop_end * len(video_reader))
clip_length = min(video_length, (min_sample_n_frames - 1) * self.video_sample_stride + 1)
start_idx = random.randint(int(self.video_length_drop_start * video_length), video_length - clip_length) if video_length != clip_length else 0
batch_index = np.linspace(start_idx, start_idx + clip_length - 1, min_sample_n_frames, dtype=int)
try:
sample_args = (video_reader, batch_index)
pixel_values = func_timeout(
VIDEO_READER_TIMEOUT, get_video_reader_batch, args=sample_args
)
resized_frames = []
for i in range(len(pixel_values)):
frame = pixel_values[i]
resized_frame = resize_frame(frame, self.larger_side_of_image_and_video)
resized_frames.append(resized_frame)
pixel_values = np.array(resized_frames)
except FunctionTimedOut:
raise ValueError(f"Read {idx} timeout.")
except Exception as e:
raise ValueError(f"Failed to extract frames from video. Error is {e}.")
if not self.enable_bucket:
pixel_values = torch.from_numpy(pixel_values).permute(0, 3, 1, 2).contiguous()
pixel_values = pixel_values / 255.
del video_reader
else:
pixel_values = pixel_values
if not self.enable_bucket:
pixel_values = self.video_transforms(pixel_values)
# Random use no text generation
if random.random() < self.text_drop_ratio:
text = ''
control_video_id = data_info['control_file_path']
if self.data_root is None:
control_video_id = control_video_id
else:
control_video_id = os.path.join(self.data_root, control_video_id)
if self.enable_camera_info:
if control_video_id.lower().endswith('.txt'):
if not self.enable_bucket:
control_pixel_values = torch.zeros_like(pixel_values)
control_camera_values = process_pose_file(control_video_id, width=self.video_sample_size[1], height=self.video_sample_size[0])
control_camera_values = torch.from_numpy(control_camera_values).permute(0, 3, 1, 2).contiguous()
control_camera_values = F.interpolate(control_camera_values, size=(len(video_reader), control_camera_values.size(3)), mode='bilinear', align_corners=True)
control_camera_values = self.video_transforms_camera(control_camera_values)
else:
control_pixel_values = np.zeros_like(pixel_values)
control_camera_values = process_pose_file(control_video_id, width=self.video_sample_size[1], height=self.video_sample_size[0], return_poses=True)
control_camera_values = torch.from_numpy(np.array(control_camera_values)).unsqueeze(0).unsqueeze(0)
control_camera_values = F.interpolate(control_camera_values, size=(len(video_reader), control_camera_values.size(3)), mode='bilinear', align_corners=True)[0][0]
control_camera_values = np.array([control_camera_values[index] for index in batch_index])
else:
if not self.enable_bucket:
control_pixel_values = torch.zeros_like(pixel_values)
control_camera_values = None
else:
control_pixel_values = np.zeros_like(pixel_values)
control_camera_values = None
else:
with VideoReader_contextmanager(control_video_id, num_threads=2) as control_video_reader:
try:
sample_args = (control_video_reader, batch_index)
control_pixel_values = func_timeout(
VIDEO_READER_TIMEOUT, get_video_reader_batch, args=sample_args
)
resized_frames = []
for i in range(len(control_pixel_values)):
frame = control_pixel_values[i]
resized_frame = resize_frame(frame, self.larger_side_of_image_and_video)
resized_frames.append(resized_frame)
control_pixel_values = np.array(resized_frames)
except FunctionTimedOut:
raise ValueError(f"Read {idx} timeout.")
except Exception as e:
raise ValueError(f"Failed to extract frames from video. Error is {e}.")
if not self.enable_bucket:
control_pixel_values = torch.from_numpy(control_pixel_values).permute(0, 3, 1, 2).contiguous()
control_pixel_values = control_pixel_values / 255.
del control_video_reader
else:
control_pixel_values = control_pixel_values
if not self.enable_bucket:
control_pixel_values = self.video_transforms(control_pixel_values)
control_camera_values = None
return pixel_values, control_pixel_values, control_camera_values, text, "video"
else:
image_path, text = data_info['file_path'], data_info['text']
if self.data_root is not None:
image_path = os.path.join(self.data_root, image_path)
image = Image.open(image_path).convert('RGB')
if not self.enable_bucket:
image = self.image_transforms(image).unsqueeze(0)
else:
image = np.expand_dims(np.array(image), 0)
if random.random() < self.text_drop_ratio:
text = ''
control_image_id = data_info['control_file_path']
if self.data_root is None:
control_image_id = control_image_id
else:
control_image_id = os.path.join(self.data_root, control_image_id)
control_image = Image.open(control_image_id).convert('RGB')
if not self.enable_bucket:
control_image = self.image_transforms(control_image).unsqueeze(0)
else:
control_image = np.expand_dims(np.array(control_image), 0)
return image, control_image, None, text, 'image'
def __len__(self):
return self.length
def __getitem__(self, idx):
data_info = self.dataset[idx % len(self.dataset)]
data_type = data_info.get('type', 'image')
while True:
sample = {}
try:
data_info_local = self.dataset[idx % len(self.dataset)]
data_type_local = data_info_local.get('type', 'image')
if data_type_local != data_type:
raise ValueError("data_type_local != data_type")
pixel_values, control_pixel_values, control_camera_values, name, data_type = self.get_batch(idx)
sample["pixel_values"] = pixel_values
sample["control_pixel_values"] = control_pixel_values
sample["text"] = name
sample["data_type"] = data_type
sample["idx"] = idx
if self.enable_camera_info:
sample["control_camera_values"] = control_camera_values
if len(sample) > 0:
break
except Exception as e:
print(e, self.dataset[idx % len(self.dataset)])
idx = random.randint(0, self.length-1)
if self.enable_inpaint and not self.enable_bucket:
mask = get_random_mask(pixel_values.size())
mask_pixel_values = pixel_values * (1 - mask) + torch.ones_like(pixel_values) * -1 * mask
sample["mask_pixel_values"] = mask_pixel_values
sample["mask"] = mask
clip_pixel_values = sample["pixel_values"][0].permute(1, 2, 0).contiguous()
clip_pixel_values = (clip_pixel_values * 0.5 + 0.5) * 255
sample["clip_pixel_values"] = clip_pixel_values
ref_pixel_values = sample["pixel_values"][0].unsqueeze(0)
if (mask == 1).all():
ref_pixel_values = torch.ones_like(ref_pixel_values) * -1
sample["ref_pixel_values"] = ref_pixel_values
return sample
if __name__ == "__main__":
@@ -776,4 +238,4 @@ if __name__ == "__main__":
)
dataloader = torch.utils.data.DataLoader(dataset, batch_size=4, num_workers=16)
for idx, batch in enumerate(dataloader):
print(batch["pixel_values"].shape, len(batch["text"]))
print(batch["pixel_values"].shape, len(batch["text"]))
-15
View File
@@ -1,15 +0,0 @@
from .autoencoder_magvit import (AutoencoderKL, AutoencoderKLCogVideoX,
AutoencoderKLMagvit)
from .transformer3d import (EasyAnimateTransformer3DModel,
HunyuanTransformer3DModel, Transformer3DModel)
name_to_transformer3d = {
"Transformer3DModel": Transformer3DModel,
"HunyuanTransformer3DModel": HunyuanTransformer3DModel,
"EasyAnimateTransformer3DModel": EasyAnimateTransformer3DModel,
}
name_to_autoencoder_magvit = {
"AutoencoderKL": AutoencoderKL,
"AutoencoderKLMagvit": AutoencoderKLMagvit,
"AutoencoderKLCogVideoX": AutoencoderKLCogVideoX,
}
File diff suppressed because it is too large Load Diff
+123 -543
View File
@@ -15,20 +15,9 @@ from typing import Dict, Optional, Tuple, Union
import torch
import torch.nn as nn
import torch.nn.functional as F
from diffusers.configuration_utils import ConfigMixin, register_to_config
from diffusers.loaders.single_file_model import FromOriginalModelMixin
from diffusers.models.autoencoders.vae import (DecoderOutput,
DiagonalGaussianDistribution)
from diffusers.models.modeling_outputs import AutoencoderKLOutput
from diffusers.models.modeling_utils import ModelMixin
from diffusers.utils import logging
from diffusers.utils.accelerate_utils import apply_forward_hook
try:
from diffusers.loaders import FromOriginalVAEMixin
except:
from diffusers.loaders import FromOriginalModelMixin as FromOriginalVAEMixin
from diffusers.loaders import FromOriginalVAEMixin
from diffusers.models.attention_processor import (
ADDED_KV_ATTENTION_PROCESSORS, CROSS_ATTENTION_PROCESSORS, Attention,
AttentionProcessor, AttnAddedKVProcessor, AttnProcessor)
@@ -38,17 +27,10 @@ from diffusers.models.modeling_outputs import AutoencoderKLOutput
from diffusers.models.modeling_utils import ModelMixin
from diffusers.utils.accelerate_utils import apply_forward_hook
from torch import nn
from diffusers import AutoencoderKL
from ..vae.ldm.models.cogvideox_enc_dec import (CogVideoXCausalConv3d,
CogVideoXDecoder3D,
CogVideoXEncoder3D,
CogVideoXSafeConv3d)
from ..vae.ldm.models.omnigen_enc_dec import CausalConv3d
from ..vae.ldm.models.omnigen_enc_dec import Decoder as omnigen_Mag_Decoder
from ..vae.ldm.models.omnigen_enc_dec import Encoder as omnigen_Mag_Encoder
logger = logging.get_logger(__name__) # pylint: disable=invalid-name
def str_eval(item):
if type(item) == str:
@@ -97,7 +79,6 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
out_channels: int = 3,
ch = 128,
ch_mult = [ 1,2,4,4 ],
block_out_channels = [128, 256, 512, 512],
use_gc_blocks = None,
down_block_types: tuple = None,
up_block_types: tuple = None,
@@ -111,20 +92,9 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
latent_channels: int = 4,
norm_num_groups: int = 32,
scaling_factor: float = 0.1825,
force_upcast: float = True,
slice_mag_vae=True,
slice_compression_vae=False,
cache_compression_vae=False,
cache_mag_vae=False,
use_tiling=False,
use_tiling_encoder=False,
use_tiling_decoder=False,
mini_batch_encoder=9,
mini_batch_decoder=3,
upcast_vae=False,
spatial_group_norm=False,
tile_sample_min_size=384,
tile_overlap_factor=0.25,
):
super().__init__()
down_block_types = str_eval(down_block_types)
@@ -133,9 +103,8 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
in_channels=in_channels,
out_channels=latent_channels,
down_block_types=down_block_types,
ch=ch,
ch_mult=ch_mult,
block_out_channels=block_out_channels,
ch = ch,
ch_mult = ch_mult,
use_gc_blocks=use_gc_blocks,
mid_block_type=mid_block_type,
mid_block_use_attention=mid_block_use_attention,
@@ -146,21 +115,16 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
act_fn=act_fn,
num_attention_heads=num_attention_heads,
double_z=True,
slice_mag_vae=slice_mag_vae,
slice_compression_vae=slice_compression_vae,
cache_compression_vae=cache_compression_vae,
cache_mag_vae=cache_mag_vae,
mini_batch_encoder=mini_batch_encoder,
spatial_group_norm=spatial_group_norm,
)
self.decoder = omnigen_Mag_Decoder(
in_channels=latent_channels,
out_channels=out_channels,
up_block_types=up_block_types,
ch=ch,
ch_mult=ch_mult,
block_out_channels=block_out_channels,
ch = ch,
ch_mult = ch_mult,
use_gc_blocks=use_gc_blocks,
mid_block_type=mid_block_type,
mid_block_use_attention=mid_block_use_attention,
@@ -170,30 +134,20 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
norm_num_groups=norm_num_groups,
act_fn=act_fn,
num_attention_heads=num_attention_heads,
slice_mag_vae=slice_mag_vae,
slice_compression_vae=slice_compression_vae,
cache_compression_vae=cache_compression_vae,
cache_mag_vae=cache_mag_vae,
mini_batch_decoder=mini_batch_decoder,
spatial_group_norm=spatial_group_norm,
)
self.quant_conv = nn.Conv3d(2 * latent_channels, 2 * latent_channels, kernel_size=1)
self.post_quant_conv = nn.Conv3d(latent_channels, latent_channels, kernel_size=1)
self.slice_mag_vae = slice_mag_vae
self.slice_compression_vae = slice_compression_vae
self.cache_compression_vae = cache_compression_vae
self.cache_mag_vae = cache_mag_vae
self.mini_batch_encoder = mini_batch_encoder
self.mini_batch_decoder = mini_batch_decoder
self.use_slicing = False
self.use_tiling = use_tiling
self.use_tiling_encoder = use_tiling_encoder
self.use_tiling_decoder = use_tiling_decoder
self.upcast_vae = upcast_vae
self.tile_sample_min_size = tile_sample_min_size
self.tile_overlap_factor = tile_overlap_factor
self.use_tiling = False
self.tile_sample_min_size = 256
self.tile_overlap_factor = 0.25
self.tile_latent_min_size = int(self.tile_sample_min_size / (2 ** (len(ch_mult) - 1)))
self.scaling_factor = scaling_factor
@@ -201,10 +155,81 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
if isinstance(module, (omnigen_Mag_Encoder, omnigen_Mag_Decoder)):
module.gradient_checkpointing = value
def _clear_conv_cache(self):
for name, module in self.named_modules():
if isinstance(module, CausalConv3d):
module._clear_conv_cache()
@property
# Copied from diffusers.models.unets.unet_2d_condition.UNet2DConditionModel.attn_processors
def attn_processors(self) -> Dict[str, AttentionProcessor]:
r"""
Returns:
`dict` of attention processors: A dictionary containing all attention processors used in the model with
indexed by its weight name.
"""
# set recursively
processors = {}
def fn_recursive_add_processors(name: str, module: torch.nn.Module, processors: Dict[str, AttentionProcessor]):
if hasattr(module, "get_processor"):
processors[f"{name}.processor"] = module.get_processor(return_deprecated_lora=True)
for sub_name, child in module.named_children():
fn_recursive_add_processors(f"{name}.{sub_name}", child, processors)
return processors
for name, module in self.named_children():
fn_recursive_add_processors(name, module, processors)
return processors
# Copied from diffusers.models.unets.unet_2d_condition.UNet2DConditionModel.set_attn_processor
def set_attn_processor(self, processor: Union[AttentionProcessor, Dict[str, AttentionProcessor]]):
r"""
Sets the attention processor to use to compute attention.
Parameters:
processor (`dict` of `AttentionProcessor` or only `AttentionProcessor`):
The instantiated processor class or a dictionary of processor classes that will be set as the processor
for **all** `Attention` layers.
If `processor` is a dict, the key needs to define the path to the corresponding cross attention
processor. This is strongly recommended when setting trainable attention processors.
"""
count = len(self.attn_processors.keys())
if isinstance(processor, dict) and len(processor) != count:
raise ValueError(
f"A dict of processors was passed, but the number of processors {len(processor)} does not match the"
f" number of attention layers: {count}. Please make sure to pass {count} processor classes."
)
def fn_recursive_attn_processor(name: str, module: torch.nn.Module, processor):
if hasattr(module, "set_processor"):
if not isinstance(processor, dict):
module.set_processor(processor)
else:
module.set_processor(processor.pop(f"{name}.processor"))
for sub_name, child in module.named_children():
fn_recursive_attn_processor(f"{name}.{sub_name}", child, processor)
for name, module in self.named_children():
fn_recursive_attn_processor(name, module, processor)
# Copied from diffusers.models.unets.unet_2d_condition.UNet2DConditionModel.set_default_attn_processor
def set_default_attn_processor(self):
"""
Disables custom attention processors and sets the default attention implementation.
"""
if all(proc.__class__ in ADDED_KV_ATTENTION_PROCESSORS for proc in self.attn_processors.values()):
processor = AttnAddedKVProcessor()
elif all(proc.__class__ in CROSS_ATTENTION_PROCESSORS for proc in self.attn_processors.values()):
processor = AttnProcessor()
else:
raise ValueError(
f"Cannot call `set_default_attn_processor` when attention processors are of type {next(iter(self.attn_processors.values()))}"
)
self.set_attn_processor(processor)
@apply_forward_hook
def encode(
@@ -222,16 +247,8 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
The latent representations of the encoded images. If `return_dict` is True, a
[`~models.autoencoder_kl.AutoencoderKLOutput`] is returned, otherwise a plain `tuple` is returned.
"""
if self.upcast_vae:
x = x.float()
self.encoder = self.encoder.float()
self.quant_conv = self.quant_conv.float()
if self.use_tiling and (x.shape[-1] > self.tile_sample_min_size or x.shape[-2] > self.tile_sample_min_size):
x = self.tiled_encode(x, return_dict=return_dict)
return x
if self.use_tiling_encoder and (x.shape[-1] > self.tile_sample_min_size or x.shape[-2] > self.tile_sample_min_size):
x = self.tiled_encode(x, return_dict=return_dict)
return x
return self.tiled_encode(x, return_dict=return_dict)
if self.use_slicing and x.shape[0] > 1:
encoded_slices = [self.encoder(x_slice) for x_slice in x.split(1)]
@@ -242,22 +259,14 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
moments = self.quant_conv(h)
posterior = DiagonalGaussianDistribution(moments)
self._clear_conv_cache()
if not return_dict:
return (posterior,)
return AutoencoderKLOutput(latent_dist=posterior)
def _decode(self, z: torch.FloatTensor, return_dict: bool = True) -> Union[DecoderOutput, torch.FloatTensor]:
if self.upcast_vae:
z = z.float()
self.decoder = self.decoder.float()
self.post_quant_conv = self.post_quant_conv.float()
if self.use_tiling and (z.shape[-1] > self.tile_latent_min_size or z.shape[-2] > self.tile_latent_min_size):
return self.tiled_decode(z, return_dict=return_dict)
if self.use_tiling_decoder and (z.shape[-1] > self.tile_latent_min_size or z.shape[-2] > self.tile_latent_min_size):
return self.tiled_decode(z, return_dict=return_dict)
z = self.post_quant_conv(z)
dec = self.decoder(z)
@@ -290,7 +299,6 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
else:
decoded = self._decode(z).sample
self._clear_conv_cache()
if not return_dict:
return (decoded,)
@@ -394,34 +402,6 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
result_rows.append(torch.cat(result_row, dim=4))
dec = torch.cat(result_rows, dim=3)
# Handle the lower right corner tile separately
lower_right_original = z[
:,
:,
:,
-self.tile_latent_min_size:,
-self.tile_latent_min_size:
]
quantized_lower_right = self.decoder(self.post_quant_conv(lower_right_original))
# Combine
H, W = quantized_lower_right.size(-2), quantized_lower_right.size(-1)
x_weights = torch.linspace(0, 1, W).unsqueeze(0).repeat(H, 1)
y_weights = torch.linspace(0, 1, H).unsqueeze(1).repeat(1, W)
weights = torch.min(x_weights, y_weights)
if len(dec.size()) == 4:
weights = weights.unsqueeze(0).unsqueeze(0)
elif len(dec.size()) == 5:
weights = weights.unsqueeze(0).unsqueeze(0).unsqueeze(0)
weights = weights.to(dec.device)
quantized_area = dec[:, :, :, -H:, -W:]
combined = weights * quantized_lower_right + (1 - weights) * quantized_area
dec[:, :, :, -H:, -W:] = combined
if not return_dict:
return (dec,)
@@ -455,6 +435,44 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
return DecoderOutput(sample=dec)
# Copied from diffusers.models.unets.unet_2d_condition.UNet2DConditionModel.fuse_qkv_projections
def fuse_qkv_projections(self):
"""
Enables fused QKV projections. For self-attention modules, all projection matrices (i.e., query,
key, value) are fused. For cross-attention modules, key and value projection matrices are fused.
<Tip warning={true}>
This API is 🧪 experimental.
</Tip>
"""
self.original_attn_processors = None
for _, attn_processor in self.attn_processors.items():
if "Added" in str(attn_processor.__class__.__name__):
raise ValueError("`fuse_qkv_projections()` is not supported for models having added KV projections.")
self.original_attn_processors = self.attn_processors
for module in self.modules():
if isinstance(module, Attention):
module.fuse_projections(fuse=True)
# Copied from diffusers.models.unets.unet_2d_condition.UNet2DConditionModel.unfuse_qkv_projections
def unfuse_qkv_projections(self):
"""Disables the fused QKV projection if enabled.
<Tip warning={true}>
This API is 🧪 experimental.
</Tip>
"""
if self.original_attn_processors is not None:
self.set_attn_processor(self.original_attn_processors)
@classmethod
def from_pretrained(cls, pretrained_model_path, subfolder=None, **vae_additional_kwargs):
import json
@@ -483,441 +501,3 @@ class AutoencoderKLMagvit(ModelMixin, ConfigMixin, FromOriginalVAEMixin):
print(f"### missing keys: {len(m)}; \n### unexpected keys: {len(u)};")
print(m, u)
return model
# Modified from https://github.com/huggingface/diffusers/blob/main/src/diffusers/models/autoencoders/autoencoder_kl_cogvideox.py
# Copyright 2024 The CogVideoX team, Tsinghua University & ZhipuAI and The HuggingFace Team.
# All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
class AutoencoderKLCogVideoX(ModelMixin, ConfigMixin, FromOriginalModelMixin):
r"""
A VAE model with KL loss for encoding images into latents and decoding latent representations into images. Used in
[CogVideoX](https://github.com/THUDM/CogVideo).
This model inherits from [`ModelMixin`]. Check the superclass documentation for it's generic methods implemented
for all models (such as downloading or saving).
Parameters:
in_channels (int, *optional*, defaults to 3): Number of channels in the input image.
out_channels (int, *optional*, defaults to 3): Number of channels in the output.
down_block_types (`Tuple[str]`, *optional*, defaults to `("DownEncoderBlock2D",)`):
Tuple of downsample block types.
up_block_types (`Tuple[str]`, *optional*, defaults to `("UpDecoderBlock2D",)`):
Tuple of upsample block types.
block_out_channels (`Tuple[int]`, *optional*, defaults to `(64,)`):
Tuple of block output channels.
act_fn (`str`, *optional*, defaults to `"silu"`): The activation function to use.
sample_size (`int`, *optional*, defaults to `32`): Sample input size.
scaling_factor (`float`, *optional*, defaults to `1.15258426`):
The component-wise standard deviation of the trained latent space computed using the first batch of the
training set. This is used to scale the latent space to have unit variance when training the diffusion
model. The latents are scaled with the formula `z = z * scaling_factor` before being passed to the
diffusion model. When decoding, the latents are scaled back to the original scale with the formula: `z = 1
/ scaling_factor * z`. For more details, refer to sections 4.3.2 and D.1 of the [High-Resolution Image
Synthesis with Latent Diffusion Models](https://arxiv.org/abs/2112.10752) paper.
force_upcast (`bool`, *optional*, default to `True`):
If enabled it will force the VAE to run in float32 for high image resolution pipelines, such as SD-XL. VAE
can be fine-tuned / trained to a lower range without loosing too much precision in which case
`force_upcast` can be set to `False` - see: https://huggingface.co/madebyollin/sdxl-vae-fp16-fix
"""
_supports_gradient_checkpointing = True
_no_split_modules = ["CogVideoXResnetBlock3D"]
@register_to_config
def __init__(
self,
in_channels: int = 3,
out_channels: int = 3,
down_block_types: Tuple[str] = (
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
),
up_block_types: Tuple[str] = (
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
),
block_out_channels: Tuple[int] = (128, 256, 256, 512),
latent_channels: int = 16,
layers_per_block: int = 3,
act_fn: str = "silu",
norm_eps: float = 1e-6,
norm_num_groups: int = 32,
temporal_compression_ratio: float = 4,
sample_height: int = 480,
sample_width: int = 720,
scaling_factor: float = 1.15258426,
shift_factor: Optional[float] = None,
latents_mean: Optional[Tuple[float]] = None,
latents_std: Optional[Tuple[float]] = None,
force_upcast: float = True,
use_quant_conv: bool = False,
use_post_quant_conv: bool = False,
slice_mag_vae=False,
slice_compression_vae=False,
cache_compression_vae=False,
cache_mag_vae=True,
use_tiling=False,
mini_batch_encoder=4,
mini_batch_decoder=1,
):
super().__init__()
self.encoder = CogVideoXEncoder3D(
in_channels=in_channels,
out_channels=latent_channels,
down_block_types=down_block_types,
block_out_channels=block_out_channels,
layers_per_block=layers_per_block,
act_fn=act_fn,
norm_eps=norm_eps,
norm_num_groups=norm_num_groups,
temporal_compression_ratio=temporal_compression_ratio,
)
self.decoder = CogVideoXDecoder3D(
in_channels=latent_channels,
out_channels=out_channels,
up_block_types=up_block_types,
block_out_channels=block_out_channels,
layers_per_block=layers_per_block,
act_fn=act_fn,
norm_eps=norm_eps,
norm_num_groups=norm_num_groups,
temporal_compression_ratio=temporal_compression_ratio,
)
self.quant_conv = CogVideoXSafeConv3d(2 * out_channels, 2 * out_channels, 1) if use_quant_conv else None
self.post_quant_conv = CogVideoXSafeConv3d(out_channels, out_channels, 1) if use_post_quant_conv else None
self.use_slicing = False
self.use_tiling = use_tiling
# Can be increased to decode more latent frames at once, but comes at a reasonable memory cost and it is not
# recommended because the temporal parts of the VAE, here, are tricky to understand.
# If you decode X latent frames together, the number of output frames is:
# (X + (2 conv cache) + (2 time upscale_1) + (4 time upscale_2) - (2 causal conv downscale)) => X + 6 frames
#
# Example with num_latent_frames_batch_size = 2:
# - 12 latent frames: (0, 1), (2, 3), (4, 5), (6, 7), (8, 9), (10, 11) are processed together
# => (12 // 2 frame slices) * ((2 num_latent_frames_batch_size) + (2 conv cache) + (2 time upscale_1) + (4 time upscale_2) - (2 causal conv downscale))
# => 6 * 8 = 48 frames
# - 13 latent frames: (0, 1, 2) (special case), (3, 4), (5, 6), (7, 8), (9, 10), (11, 12) are processed together
# => (1 frame slice) * ((3 num_latent_frames_batch_size) + (2 conv cache) + (2 time upscale_1) + (4 time upscale_2) - (2 causal conv downscale)) +
# ((13 - 3) // 2) * ((2 num_latent_frames_batch_size) + (2 conv cache) + (2 time upscale_1) + (4 time upscale_2) - (2 causal conv downscale))
# => 1 * 9 + 5 * 8 = 49 frames
# It has been implemented this way so as to not have "magic values" in the code base that would be hard to explain. Note that
# setting it to anything other than 2 would give poor results because the VAE hasn't been trained to be adaptive with different
# number of temporal frames.
self.num_latent_frames_batch_size = 2
# We make the minimum height and width of sample for tiling half that of the generally supported
self.tile_sample_min_height = sample_height // 2
self.tile_sample_min_width = sample_width // 2
self.tile_latent_min_height = int(
self.tile_sample_min_height / (2 ** (len(self.config.block_out_channels) - 1))
)
self.tile_latent_min_width = int(self.tile_sample_min_width / (2 ** (len(self.config.block_out_channels) - 1)))
# These are experimental overlap factors that were chosen based on experimentation and seem to work best for
# 720x480 (WxH) resolution. The above resolution is the strongly recommended generation resolution in CogVideoX
# and so the tiling implementation has only been tested on those specific resolutions.
self.tile_overlap_factor_height = 1 / 6
self.tile_overlap_factor_width = 1 / 5
def _set_gradient_checkpointing(self, module, value=False):
if isinstance(module, (CogVideoXEncoder3D, CogVideoXDecoder3D)):
module.gradient_checkpointing = value
def _clear_fake_context_parallel_cache(self):
for name, module in self.named_modules():
if isinstance(module, CogVideoXCausalConv3d):
logger.debug(f"Clearing fake Context Parallel cache for layer: {name}")
module._clear_fake_context_parallel_cache()
def enable_tiling(
self,
tile_sample_min_height: Optional[int] = None,
tile_sample_min_width: Optional[int] = None,
tile_overlap_factor_height: Optional[float] = None,
tile_overlap_factor_width: Optional[float] = None,
) -> None:
r"""
Enable tiled VAE decoding. When this option is enabled, the VAE will split the input tensor into tiles to
compute decoding and encoding in several steps. This is useful for saving a large amount of memory and to allow
processing larger images.
Args:
tile_sample_min_height (`int`, *optional*):
The minimum height required for a sample to be separated into tiles across the height dimension.
tile_sample_min_width (`int`, *optional*):
The minimum width required for a sample to be separated into tiles across the width dimension.
tile_overlap_factor_height (`int`, *optional*):
The minimum amount of overlap between two consecutive vertical tiles. This is to ensure that there are
no tiling artifacts produced across the height dimension. Must be between 0 and 1. Setting a higher
value might cause more tiles to be processed leading to slow down of the decoding process.
tile_overlap_factor_width (`int`, *optional*):
The minimum amount of overlap between two consecutive horizontal tiles. This is to ensure that there
are no tiling artifacts produced across the width dimension. Must be between 0 and 1. Setting a higher
value might cause more tiles to be processed leading to slow down of the decoding process.
"""
self.use_tiling = True
self.tile_sample_min_height = tile_sample_min_height or self.tile_sample_min_height
self.tile_sample_min_width = tile_sample_min_width or self.tile_sample_min_width
self.tile_latent_min_height = int(
self.tile_sample_min_height / (2 ** (len(self.config.block_out_channels) - 1))
)
self.tile_latent_min_width = int(self.tile_sample_min_width / (2 ** (len(self.config.block_out_channels) - 1)))
self.tile_overlap_factor_height = tile_overlap_factor_height or self.tile_overlap_factor_height
self.tile_overlap_factor_width = tile_overlap_factor_width or self.tile_overlap_factor_width
def disable_tiling(self) -> None:
r"""
Disable tiled VAE decoding. If `enable_tiling` was previously enabled, this method will go back to computing
decoding in one step.
"""
self.use_tiling = False
def enable_slicing(self) -> None:
r"""
Enable sliced VAE decoding. When this option is enabled, the VAE will split the input tensor in slices to
compute decoding in several steps. This is useful to save some memory and allow larger batch sizes.
"""
self.use_slicing = True
def disable_slicing(self) -> None:
r"""
Disable sliced VAE decoding. If `enable_slicing` was previously enabled, this method will go back to computing
decoding in one step.
"""
self.use_slicing = False
@apply_forward_hook
def encode(
self, x: torch.Tensor, return_dict: bool = True
) -> Union[AutoencoderKLOutput, Tuple[DiagonalGaussianDistribution]]:
"""
Encode a batch of images into latents.
Args:
x (`torch.Tensor`): Input batch of images.
return_dict (`bool`, *optional*, defaults to `True`):
Whether to return a [`~models.autoencoder_kl.AutoencoderKLOutput`] instead of a plain tuple.
Returns:
The latent representations of the encoded images. If `return_dict` is True, a
[`~models.autoencoder_kl.AutoencoderKLOutput`] is returned, otherwise a plain `tuple` is returned.
"""
batch_size, num_channels, num_frames, height, width = x.shape
if num_frames == 1:
h = self.encoder(x)
if self.quant_conv is not None:
h = self.quant_conv(h)
posterior = DiagonalGaussianDistribution(h)
else:
frame_batch_size = 4
h = []
for i in range(num_frames // frame_batch_size):
remaining_frames = num_frames % frame_batch_size
start_frame = frame_batch_size * i + (0 if i == 0 else remaining_frames)
end_frame = frame_batch_size * (i + 1) + remaining_frames
z_intermediate = x[:, :, start_frame:end_frame]
z_intermediate = self.encoder(z_intermediate)
if self.quant_conv is not None:
z_intermediate = self.quant_conv(z_intermediate)
h.append(z_intermediate)
self._clear_fake_context_parallel_cache()
h = torch.cat(h, dim=2)
posterior = DiagonalGaussianDistribution(h)
self._clear_fake_context_parallel_cache()
if not return_dict:
return (posterior,)
return AutoencoderKLOutput(latent_dist=posterior)
def _decode(self, z: torch.Tensor, return_dict: bool = True) -> Union[DecoderOutput, torch.Tensor]:
batch_size, num_channels, num_frames, height, width = z.shape
if self.use_tiling and (width > self.tile_latent_min_width or height > self.tile_latent_min_height):
return self.tiled_decode(z, return_dict=return_dict)
if num_frames == 1:
dec = []
z_intermediate = z
if self.post_quant_conv is not None:
z_intermediate = self.post_quant_conv(z_intermediate)
z_intermediate = self.decoder(z_intermediate)
dec.append(z_intermediate)
else:
frame_batch_size = self.num_latent_frames_batch_size
dec = []
for i in range(num_frames // frame_batch_size):
remaining_frames = num_frames % frame_batch_size
start_frame = frame_batch_size * i + (0 if i == 0 else remaining_frames)
end_frame = frame_batch_size * (i + 1) + remaining_frames
z_intermediate = z[:, :, start_frame:end_frame]
if self.post_quant_conv is not None:
z_intermediate = self.post_quant_conv(z_intermediate)
z_intermediate = self.decoder(z_intermediate)
dec.append(z_intermediate)
self._clear_fake_context_parallel_cache()
dec = torch.cat(dec, dim=2)
if not return_dict:
return (dec,)
return DecoderOutput(sample=dec)
@apply_forward_hook
def decode(self, z: torch.Tensor, return_dict: bool = True) -> Union[DecoderOutput, torch.Tensor]:
"""
Decode a batch of images.
Args:
z (`torch.Tensor`): Input batch of latent vectors.
return_dict (`bool`, *optional*, defaults to `True`):
Whether to return a [`~models.vae.DecoderOutput`] instead of a plain tuple.
Returns:
[`~models.vae.DecoderOutput`] or `tuple`:
If return_dict is True, a [`~models.vae.DecoderOutput`] is returned, otherwise a plain `tuple` is
returned.
"""
if self.use_slicing and z.shape[0] > 1:
decoded_slices = [self._decode(z_slice).sample for z_slice in z.split(1)]
decoded = torch.cat(decoded_slices)
else:
decoded = self._decode(z).sample
if not return_dict:
return (decoded,)
return DecoderOutput(sample=decoded)
def blend_v(self, a: torch.Tensor, b: torch.Tensor, blend_extent: int) -> torch.Tensor:
blend_extent = min(a.shape[3], b.shape[3], blend_extent)
for y in range(blend_extent):
b[:, :, :, y, :] = a[:, :, :, -blend_extent + y, :] * (1 - y / blend_extent) + b[:, :, :, y, :] * (
y / blend_extent
)
return b
def blend_h(self, a: torch.Tensor, b: torch.Tensor, blend_extent: int) -> torch.Tensor:
blend_extent = min(a.shape[4], b.shape[4], blend_extent)
for x in range(blend_extent):
b[:, :, :, :, x] = a[:, :, :, :, -blend_extent + x] * (1 - x / blend_extent) + b[:, :, :, :, x] * (
x / blend_extent
)
return b
def tiled_decode(self, z: torch.Tensor, return_dict: bool = True) -> Union[DecoderOutput, torch.Tensor]:
r"""
Decode a batch of images using a tiled decoder.
Args:
z (`torch.Tensor`): Input batch of latent vectors.
return_dict (`bool`, *optional*, defaults to `True`):
Whether or not to return a [`~models.vae.DecoderOutput`] instead of a plain tuple.
Returns:
[`~models.vae.DecoderOutput`] or `tuple`:
If return_dict is True, a [`~models.vae.DecoderOutput`] is returned, otherwise a plain `tuple` is
returned.
"""
# Rough memory assessment:
# - In CogVideoX-2B, there are a total of 24 CausalConv3d layers.
# - The biggest intermediate dimensions are: [1, 128, 9, 480, 720].
# - Assume fp16 (2 bytes per value).
# Memory required: 1 * 128 * 9 * 480 * 720 * 24 * 2 / 1024**3 = 17.8 GB
#
# Memory assessment when using tiling:
# - Assume everything as above but now HxW is 240x360 by tiling in half
# Memory required: 1 * 128 * 9 * 240 * 360 * 24 * 2 / 1024**3 = 4.5 GB
batch_size, num_channels, num_frames, height, width = z.shape
overlap_height = int(self.tile_latent_min_height * (1 - self.tile_overlap_factor_height))
overlap_width = int(self.tile_latent_min_width * (1 - self.tile_overlap_factor_width))
blend_extent_height = int(self.tile_sample_min_height * self.tile_overlap_factor_height)
blend_extent_width = int(self.tile_sample_min_width * self.tile_overlap_factor_width)
row_limit_height = self.tile_sample_min_height - blend_extent_height
row_limit_width = self.tile_sample_min_width - blend_extent_width
frame_batch_size = self.num_latent_frames_batch_size
# Split z into overlapping tiles and decode them separately.
# The tiles have an overlap to avoid seams between tiles.
rows = []
for i in range(0, height, overlap_height):
row = []
for j in range(0, width, overlap_width):
time = []
for k in range(num_frames // frame_batch_size):
remaining_frames = num_frames % frame_batch_size
start_frame = frame_batch_size * k + (0 if k == 0 else remaining_frames)
end_frame = frame_batch_size * (k + 1) + remaining_frames
tile = z[
:,
:,
start_frame:end_frame,
i : i + self.tile_latent_min_height,
j : j + self.tile_latent_min_width,
]
if self.post_quant_conv is not None:
tile = self.post_quant_conv(tile)
tile = self.decoder(tile)
time.append(tile)
self._clear_fake_context_parallel_cache()
row.append(torch.cat(time, dim=2))
rows.append(row)
result_rows = []
for i, row in enumerate(rows):
result_row = []
for j, tile in enumerate(row):
# blend the above tile and the left tile
# to the current tile and add the current tile to the result row
if i > 0:
tile = self.blend_v(rows[i - 1][j], tile, blend_extent_height)
if j > 0:
tile = self.blend_h(row[j - 1], tile, blend_extent_width)
result_row.append(tile[:, :, :, :row_limit_height, :row_limit_width])
result_rows.append(torch.cat(result_row, dim=4))
dec = torch.cat(result_rows, dim=3)
if not return_dict:
return (dec,)
return DecoderOutput(sample=dec)
def forward(
self,
sample: torch.Tensor,
sample_posterior: bool = False,
return_dict: bool = True,
generator: Optional[torch.Generator] = None,
) -> Union[torch.Tensor, torch.Tensor]:
x = sample
posterior = self.encode(x).latent_dist
if sample_posterior:
z = posterior.sample(generator=generator)
else:
z = posterior.mode()
dec = self.decode(z)
if not return_dict:
return (dec,)
return dec
-108
View File
@@ -1,108 +0,0 @@
import math
from typing import Optional
import numpy as np
import torch
import torch.nn.functional as F
from diffusers.models.embeddings import (PixArtAlphaTextProjection,
TimestepEmbedding, Timesteps,
get_timestep_embedding)
from einops import rearrange
from torch import nn
class HunyuanDiTAttentionPool(nn.Module):
def __init__(self, spacial_dim: int, embed_dim: int, num_heads: int, output_dim: int = None):
super().__init__()
self.positional_embedding = nn.Parameter(torch.randn(spacial_dim + 1, embed_dim) / embed_dim**0.5)
self.k_proj = nn.Linear(embed_dim, embed_dim)
self.q_proj = nn.Linear(embed_dim, embed_dim)
self.v_proj = nn.Linear(embed_dim, embed_dim)
self.c_proj = nn.Linear(embed_dim, output_dim or embed_dim)
self.num_heads = num_heads
def forward(self, x):
x = torch.cat([x.mean(dim=1, keepdim=True), x], dim=1)
x = x + self.positional_embedding[None, :, :].to(x.dtype)
query = self.q_proj(x[:, :1])
key = self.k_proj(x)
value = self.v_proj(x)
batch_size, _, _ = query.size()
query = query.reshape(batch_size, -1, self.num_heads, query.size(-1) // self.num_heads).transpose(1, 2) # (1, H, N, E/H)
key = key.reshape(batch_size, -1, self.num_heads, key.size(-1) // self.num_heads).transpose(1, 2) # (L+1, H, N, E/H)
value = value.reshape(batch_size, -1, self.num_heads, value.size(-1) // self.num_heads).transpose(1, 2) # (L+1, H, N, E/H)
x = F.scaled_dot_product_attention(query=query, key=key, value=value, attn_mask=None, dropout_p=0.0, is_causal=False)
x = x.transpose(1, 2).reshape(batch_size, 1, -1)
x = x.to(query.dtype)
x = self.c_proj(x)
return x.squeeze(1)
class HunyuanCombinedTimestepTextSizeStyleEmbedding(nn.Module):
def __init__(self, embedding_dim, pooled_projection_dim=1024, seq_len=256, cross_attention_dim=2048):
super().__init__()
self.time_proj = Timesteps(num_channels=256, flip_sin_to_cos=True, downscale_freq_shift=0)
self.timestep_embedder = TimestepEmbedding(in_channels=256, time_embed_dim=embedding_dim)
self.pooler = HunyuanDiTAttentionPool(
seq_len, cross_attention_dim, num_heads=8, output_dim=pooled_projection_dim
)
# Here we use a default learned embedder layer for future extension.
self.style_embedder = nn.Embedding(1, embedding_dim)
extra_in_dim = 256 * 6 + embedding_dim + pooled_projection_dim
self.extra_embedder = PixArtAlphaTextProjection(
in_features=extra_in_dim,
hidden_size=embedding_dim * 4,
out_features=embedding_dim,
act_fn="silu_fp32",
)
def forward(self, timestep, encoder_hidden_states, image_meta_size, style, hidden_dtype=None):
timesteps_proj = self.time_proj(timestep)
timesteps_emb = self.timestep_embedder(timesteps_proj.to(dtype=hidden_dtype)) # (N, 256)
# extra condition1: text
pooled_projections = self.pooler(encoder_hidden_states) # (N, 1024)
# extra condition2: image meta size embdding
image_meta_size = get_timestep_embedding(image_meta_size.view(-1), 256, True, 0)
image_meta_size = image_meta_size.to(dtype=hidden_dtype)
image_meta_size = image_meta_size.view(-1, 6 * 256) # (N, 1536)
# extra condition3: style embedding
style_embedding = self.style_embedder(style) # (N, embedding_dim)
# Concatenate all extra vectors
extra_cond = torch.cat([pooled_projections, image_meta_size, style_embedding], dim=1)
conditioning = timesteps_emb + self.extra_embedder(extra_cond) # [B, D]
return conditioning
class TimePositionalEncoding(nn.Module):
def __init__(
self,
d_model,
dropout = 0.,
max_len = 24
):
super().__init__()
self.dropout = nn.Dropout(p=dropout)
position = torch.arange(max_len).unsqueeze(1)
div_term = torch.exp(torch.arange(0, d_model, 2) * (-math.log(10000.0) / d_model))
pe = torch.zeros(1, max_len, d_model)
pe[0, :, 0::2] = torch.sin(position * div_term)
pe[0, :, 1::2] = torch.cos(position * div_term)
self.register_buffer('pe', pe)
def forward(self, x):
b, c, f, h, w = x.size()
x = rearrange(x, "b c f h w -> (b h w) f c")
x = x + self.pe[:, :x.size(1)]
x = rearrange(x, "(b h w) f c -> b c f h w", b=b, h=h, w=w)
return self.dropout(x)
+277 -146
View File
@@ -1,33 +1,248 @@
"""Modified from https://github.com/guoyww/AnimateDiff/blob/main/animatediff/models/motion_module.py
"""
import math
from typing import Any, Callable, List, Optional, Tuple, Union
import diffusers
import pkg_resources
import torch
installed_version = diffusers.__version__
if pkg_resources.parse_version(installed_version) >= pkg_resources.parse_version("0.28.2"):
from diffusers.models.attention_processor import (Attention,
AttnProcessor2_0,
HunyuanAttnProcessor2_0)
else:
from diffusers.models.attention_processor import Attention, AttnProcessor2_0
import torch.nn.functional as F
from diffusers.models.attention import FeedForward
from diffusers.utils.import_utils import is_xformers_available
from einops import rearrange, repeat
from torch import nn
from .norm import FP32LayerNorm
if is_xformers_available():
import xformers
import xformers.ops
else:
xformers = None
class CrossAttention(nn.Module):
r"""
A cross attention layer.
Parameters:
query_dim (`int`): The number of channels in the query.
cross_attention_dim (`int`, *optional*):
The number of channels in the encoder_hidden_states. If not given, defaults to `query_dim`.
heads (`int`, *optional*, defaults to 8): The number of heads to use for multi-head attention.
dim_head (`int`, *optional*, defaults to 64): The number of channels in each head.
dropout (`float`, *optional*, defaults to 0.0): The dropout probability to use.
bias (`bool`, *optional*, defaults to False):
Set to `True` for the query, key, and value linear layers to contain a bias parameter.
"""
def __init__(
self,
query_dim: int,
cross_attention_dim: Optional[int] = None,
heads: int = 8,
dim_head: int = 64,
dropout: float = 0.0,
bias=False,
upcast_attention: bool = False,
upcast_softmax: bool = False,
added_kv_proj_dim: Optional[int] = None,
norm_num_groups: Optional[int] = None,
):
super().__init__()
inner_dim = dim_head * heads
cross_attention_dim = cross_attention_dim if cross_attention_dim is not None else query_dim
self.upcast_attention = upcast_attention
self.upcast_softmax = upcast_softmax
self.scale = dim_head**-0.5
self.heads = heads
# for slice_size > 0 the attention score computation
# is split across the batch axis to save memory
# You can set slice_size with `set_attention_slice`
self.sliceable_head_dim = heads
self._slice_size = None
self._use_memory_efficient_attention_xformers = False
self.added_kv_proj_dim = added_kv_proj_dim
if norm_num_groups is not None:
self.group_norm = nn.GroupNorm(num_channels=inner_dim, num_groups=norm_num_groups, eps=1e-5, affine=True)
else:
self.group_norm = None
self.to_q = nn.Linear(query_dim, inner_dim, bias=bias)
self.to_k = nn.Linear(cross_attention_dim, inner_dim, bias=bias)
self.to_v = nn.Linear(cross_attention_dim, inner_dim, bias=bias)
if self.added_kv_proj_dim is not None:
self.add_k_proj = nn.Linear(added_kv_proj_dim, cross_attention_dim)
self.add_v_proj = nn.Linear(added_kv_proj_dim, cross_attention_dim)
self.to_out = nn.ModuleList([])
self.to_out.append(nn.Linear(inner_dim, query_dim))
self.to_out.append(nn.Dropout(dropout))
def set_use_memory_efficient_attention_xformers(
self, valid: bool, attention_op: Optional[Callable] = None
) -> None:
self._use_memory_efficient_attention_xformers = valid
def reshape_heads_to_batch_dim(self, tensor):
batch_size, seq_len, dim = tensor.shape
head_size = self.heads
tensor = tensor.reshape(batch_size, seq_len, head_size, dim // head_size)
tensor = tensor.permute(0, 2, 1, 3).reshape(batch_size * head_size, seq_len, dim // head_size)
return tensor
def reshape_batch_dim_to_heads(self, tensor):
batch_size, seq_len, dim = tensor.shape
head_size = self.heads
tensor = tensor.reshape(batch_size // head_size, head_size, seq_len, dim)
tensor = tensor.permute(0, 2, 1, 3).reshape(batch_size // head_size, seq_len, dim * head_size)
return tensor
def set_attention_slice(self, slice_size):
if slice_size is not None and slice_size > self.sliceable_head_dim:
raise ValueError(f"slice_size {slice_size} has to be smaller or equal to {self.sliceable_head_dim}.")
self._slice_size = slice_size
def forward(self, hidden_states, encoder_hidden_states=None, attention_mask=None):
batch_size, sequence_length, _ = hidden_states.shape
encoder_hidden_states = encoder_hidden_states
if self.group_norm is not None:
hidden_states = self.group_norm(hidden_states.transpose(1, 2)).transpose(1, 2)
query = self.to_q(hidden_states)
dim = query.shape[-1]
query = self.reshape_heads_to_batch_dim(query)
if self.added_kv_proj_dim is not None:
key = self.to_k(hidden_states)
value = self.to_v(hidden_states)
encoder_hidden_states_key_proj = self.add_k_proj(encoder_hidden_states)
encoder_hidden_states_value_proj = self.add_v_proj(encoder_hidden_states)
key = self.reshape_heads_to_batch_dim(key)
value = self.reshape_heads_to_batch_dim(value)
encoder_hidden_states_key_proj = self.reshape_heads_to_batch_dim(encoder_hidden_states_key_proj)
encoder_hidden_states_value_proj = self.reshape_heads_to_batch_dim(encoder_hidden_states_value_proj)
key = torch.concat([encoder_hidden_states_key_proj, key], dim=1)
value = torch.concat([encoder_hidden_states_value_proj, value], dim=1)
else:
encoder_hidden_states = encoder_hidden_states if encoder_hidden_states is not None else hidden_states
key = self.to_k(encoder_hidden_states)
value = self.to_v(encoder_hidden_states)
key = self.reshape_heads_to_batch_dim(key)
value = self.reshape_heads_to_batch_dim(value)
if attention_mask is not None:
if attention_mask.shape[-1] != query.shape[1]:
target_length = query.shape[1]
attention_mask = F.pad(attention_mask, (0, target_length), value=0.0)
attention_mask = attention_mask.repeat_interleave(self.heads, dim=0)
# attention, what we cannot get enough of
if self._use_memory_efficient_attention_xformers:
hidden_states = self._memory_efficient_attention_xformers(query, key, value, attention_mask)
# Some versions of xformers return output in fp32, cast it back to the dtype of the input
hidden_states = hidden_states.to(query.dtype)
else:
if self._slice_size is None or query.shape[0] // self._slice_size == 1:
hidden_states = self._attention(query, key, value, attention_mask)
else:
hidden_states = self._sliced_attention(query, key, value, sequence_length, dim, attention_mask)
# linear proj
hidden_states = self.to_out[0](hidden_states)
# dropout
hidden_states = self.to_out[1](hidden_states)
return hidden_states
def _attention(self, query, key, value, attention_mask=None):
if self.upcast_attention:
query = query.float()
key = key.float()
attention_scores = torch.baddbmm(
torch.empty(query.shape[0], query.shape[1], key.shape[1], dtype=query.dtype, device=query.device),
query,
key.transpose(-1, -2),
beta=0,
alpha=self.scale,
)
if attention_mask is not None:
attention_scores = attention_scores + attention_mask
if self.upcast_softmax:
attention_scores = attention_scores.float()
attention_probs = attention_scores.softmax(dim=-1)
# cast back to the original dtype
attention_probs = attention_probs.to(value.dtype)
# compute attention output
hidden_states = torch.bmm(attention_probs, value)
# reshape hidden_states
hidden_states = self.reshape_batch_dim_to_heads(hidden_states)
return hidden_states
def _sliced_attention(self, query, key, value, sequence_length, dim, attention_mask):
batch_size_attention = query.shape[0]
hidden_states = torch.zeros(
(batch_size_attention, sequence_length, dim // self.heads), device=query.device, dtype=query.dtype
)
slice_size = self._slice_size if self._slice_size is not None else hidden_states.shape[0]
for i in range(hidden_states.shape[0] // slice_size):
start_idx = i * slice_size
end_idx = (i + 1) * slice_size
query_slice = query[start_idx:end_idx]
key_slice = key[start_idx:end_idx]
if self.upcast_attention:
query_slice = query_slice.float()
key_slice = key_slice.float()
attn_slice = torch.baddbmm(
torch.empty(slice_size, query.shape[1], key.shape[1], dtype=query_slice.dtype, device=query.device),
query_slice,
key_slice.transpose(-1, -2),
beta=0,
alpha=self.scale,
)
if attention_mask is not None:
attn_slice = attn_slice + attention_mask[start_idx:end_idx]
if self.upcast_softmax:
attn_slice = attn_slice.float()
attn_slice = attn_slice.softmax(dim=-1)
# cast back to the original dtype
attn_slice = attn_slice.to(value.dtype)
attn_slice = torch.bmm(attn_slice, value[start_idx:end_idx])
hidden_states[start_idx:end_idx] = attn_slice
# reshape hidden_states
hidden_states = self.reshape_batch_dim_to_heads(hidden_states)
return hidden_states
def _memory_efficient_attention_xformers(self, query, key, value, attention_mask):
# TODO attention_mask
query = query.contiguous()
key = key.contiguous()
value = value.contiguous()
hidden_states = xformers.ops.memory_efficient_attention(query, key, value, attn_bias=attention_mask)
hidden_states = self.reshape_batch_dim_to_heads(hidden_states)
return hidden_states
def zero_module(module):
# Zero out the parameters of a module and return it.
for p in module.parameters():
@@ -60,11 +275,6 @@ class VanillaTemporalModule(nn.Module):
zero_initialize = True,
block_size = 1,
grid = False,
remove_time_embedding_in_photo = False,
global_num_attention_heads = 16,
global_attention = False,
qk_norm = False,
):
super().__init__()
@@ -79,87 +289,17 @@ class VanillaTemporalModule(nn.Module):
temporal_position_encoding_max_len=temporal_position_encoding_max_len,
grid=grid,
block_size=block_size,
remove_time_embedding_in_photo=remove_time_embedding_in_photo,
qk_norm=qk_norm,
)
self.global_transformer = GlobalTransformer3DModel(
in_channels=in_channels,
num_attention_heads=global_num_attention_heads,
attention_head_dim=in_channels // global_num_attention_heads // temporal_attention_dim_div,
qk_norm=qk_norm,
) if global_attention else None
if zero_initialize:
self.temporal_transformer.proj_out = zero_module(self.temporal_transformer.proj_out)
if global_attention:
self.global_transformer.proj_out = zero_module(self.global_transformer.proj_out)
def forward(self, input_tensor, encoder_hidden_states=None, attention_mask=None, anchor_frame_idx=None):
hidden_states = input_tensor
hidden_states = self.temporal_transformer(hidden_states, encoder_hidden_states, attention_mask)
if self.global_transformer is not None:
hidden_states = self.global_transformer(hidden_states)
output = hidden_states
return output
class GlobalTransformer3DModel(nn.Module):
def __init__(
self,
in_channels,
num_attention_heads,
attention_head_dim,
dropout = 0.0,
attention_bias = False,
upcast_attention = False,
qk_norm = False,
):
super().__init__()
inner_dim = num_attention_heads * attention_head_dim
self.norm1 = FP32LayerNorm(inner_dim)
self.proj_in = nn.Linear(in_channels, inner_dim)
self.norm2 = FP32LayerNorm(inner_dim)
if pkg_resources.parse_version(installed_version) >= pkg_resources.parse_version("0.28.2"):
self.attention = Attention(
query_dim=inner_dim,
heads=num_attention_heads,
dim_head=attention_head_dim,
dropout=dropout,
bias=attention_bias,
upcast_attention=upcast_attention,
qk_norm="layer_norm" if qk_norm else None,
processor=HunyuanAttnProcessor2_0() if qk_norm else AttnProcessor2_0(),
)
else:
self.attention = Attention(
query_dim=inner_dim,
heads=num_attention_heads,
dim_head=attention_head_dim,
dropout=dropout,
bias=attention_bias,
upcast_attention=upcast_attention,
)
self.proj_out = nn.Linear(inner_dim, in_channels)
def forward(self, hidden_states):
assert hidden_states.dim() == 5, f"Expected hidden_states to have ndim=5, but got ndim={hidden_states.dim()}."
video_length, height, width = hidden_states.shape[2], hidden_states.shape[3], hidden_states.shape[4]
hidden_states = rearrange(hidden_states, "b c f h w -> b (f h w) c")
residual = hidden_states
hidden_states = self.norm1(hidden_states)
hidden_states = self.proj_in(hidden_states)
# Attention Blocks
hidden_states = self.norm2(hidden_states)
hidden_states = self.attention(hidden_states)
hidden_states = self.proj_out(hidden_states)
output = hidden_states + residual
output = rearrange(output, "b (f h w) c -> b c f h w", f=video_length, h=height, w=width)
return output
class TemporalTransformer3DModel(nn.Module):
def __init__(
self,
@@ -181,8 +321,6 @@ class TemporalTransformer3DModel(nn.Module):
temporal_position_encoding_max_len = 4096,
grid = False,
block_size = 1,
remove_time_embedding_in_photo = False,
qk_norm = False,
):
super().__init__()
@@ -210,8 +348,6 @@ class TemporalTransformer3DModel(nn.Module):
temporal_position_encoding_max_len=temporal_position_encoding_max_len,
block_size=block_size,
grid=grid,
remove_time_embedding_in_photo=remove_time_embedding_in_photo,
qk_norm=qk_norm
)
for d in range(num_layers)
]
@@ -262,8 +398,6 @@ class TemporalTransformerBlock(nn.Module):
temporal_position_encoding_max_len = 4096,
block_size = 1,
grid = False,
remove_time_embedding_in_photo = False,
qk_norm = False,
):
super().__init__()
@@ -288,36 +422,15 @@ class TemporalTransformerBlock(nn.Module):
temporal_position_encoding_max_len=temporal_position_encoding_max_len,
block_size=block_size,
grid=grid,
remove_time_embedding_in_photo=remove_time_embedding_in_photo,
qk_norm="layer_norm" if qk_norm else None,
processor=HunyuanAttnProcessor2_0() if qk_norm else AttnProcessor2_0(),
) if pkg_resources.parse_version(installed_version) >= pkg_resources.parse_version("0.28.2") else \
VersatileAttention(
attention_mode=block_name.split("_")[0],
cross_attention_dim=cross_attention_dim if block_name.endswith("_Cross") else None,
query_dim=dim,
heads=num_attention_heads,
dim_head=attention_head_dim,
dropout=dropout,
bias=attention_bias,
upcast_attention=upcast_attention,
cross_frame_attention_mode=cross_frame_attention_mode,
temporal_position_encoding=temporal_position_encoding,
temporal_position_encoding_max_len=temporal_position_encoding_max_len,
block_size=block_size,
grid=grid,
remove_time_embedding_in_photo=remove_time_embedding_in_photo,
)
)
norms.append(FP32LayerNorm(dim))
norms.append(nn.LayerNorm(dim))
self.attention_blocks = nn.ModuleList(attention_blocks)
self.norms = nn.ModuleList(norms)
self.ff = FeedForward(dim, dropout=dropout, activation_fn=activation_fn)
self.ff_norm = FP32LayerNorm(dim)
self.ff_norm = nn.LayerNorm(dim)
def forward(self, hidden_states, encoder_hidden_states=None, attention_mask=None, video_length=None, height=None, weight=None):
for attention_block, norm in zip(self.attention_blocks, self.norms):
@@ -355,7 +468,7 @@ class PositionalEncoding(nn.Module):
x = x + self.pe[:, :x.size(1)]
return self.dropout(x)
class VersatileAttention(Attention):
class VersatileAttention(CrossAttention):
def __init__(
self,
attention_mode = None,
@@ -364,23 +477,21 @@ class VersatileAttention(Attention):
temporal_position_encoding_max_len = 4096,
grid = False,
block_size = 1,
remove_time_embedding_in_photo = False,
*args, **kwargs
):
super().__init__(*args, **kwargs)
assert attention_mode == "Temporal" or attention_mode == "Global"
assert attention_mode == "Temporal"
self.attention_mode = attention_mode
self.is_cross_attention = kwargs["cross_attention_dim"] is not None
self.block_size = block_size
self.grid = grid
self.remove_time_embedding_in_photo = remove_time_embedding_in_photo
self.pos_encoder = PositionalEncoding(
kwargs["query_dim"],
dropout=0.,
max_len=temporal_position_encoding_max_len
) if (temporal_position_encoding and attention_mode == "Temporal") or (temporal_position_encoding and attention_mode == "Global") else None
) if (temporal_position_encoding and attention_mode == "Temporal") else None
def extra_repr(self):
return f"(Module Info) Attention_Mode: {self.attention_mode}, Is_Cross_Attention: {self.is_cross_attention}"
@@ -392,13 +503,8 @@ class VersatileAttention(Attention):
# for add pos_encoder
_, before_d, _c = hidden_states.size()
hidden_states = rearrange(hidden_states, "(b f) d c -> (b d) f c", f=video_length)
if self.remove_time_embedding_in_photo:
if self.pos_encoder is not None and video_length > 1:
hidden_states = self.pos_encoder(hidden_states)
else:
if self.pos_encoder is not None:
hidden_states = self.pos_encoder(hidden_states)
if self.pos_encoder is not None:
hidden_states = self.pos_encoder(hidden_states)
if self.grid:
hidden_states = rearrange(hidden_states, "(b d) f c -> b f d c", f=video_length, d=before_d)
@@ -409,36 +515,61 @@ class VersatileAttention(Attention):
else:
d = before_d
encoder_hidden_states = repeat(encoder_hidden_states, "b n c -> (b d) n c", d=d) if encoder_hidden_states is not None else encoder_hidden_states
elif self.attention_mode == "Global":
# for add pos_encoder
_, d, _c = hidden_states.size()
hidden_states = rearrange(hidden_states, "(b f) d c -> (b d) f c", f=video_length)
if self.pos_encoder is not None:
hidden_states = self.pos_encoder(hidden_states)
hidden_states = rearrange(hidden_states, "(b d) f c -> b (f d) c", f=video_length, d=d)
else:
raise NotImplementedError
encoder_hidden_states = encoder_hidden_states
if self.group_norm is not None:
hidden_states = self.group_norm(hidden_states.transpose(1, 2)).transpose(1, 2)
query = self.to_q(hidden_states)
dim = query.shape[-1]
query = self.reshape_heads_to_batch_dim(query)
if self.added_kv_proj_dim is not None:
raise NotImplementedError
encoder_hidden_states = encoder_hidden_states if encoder_hidden_states is not None else hidden_states
key = self.to_k(encoder_hidden_states)
value = self.to_v(encoder_hidden_states)
key = self.reshape_heads_to_batch_dim(key)
value = self.reshape_heads_to_batch_dim(value)
if attention_mask is not None:
if attention_mask.shape[-1] != query.shape[1]:
target_length = query.shape[1]
attention_mask = F.pad(attention_mask, (0, target_length), value=0.0)
attention_mask = attention_mask.repeat_interleave(self.heads, dim=0)
bs = 512
new_hidden_states = []
for i in range(0, hidden_states.shape[0], bs):
__hidden_states = super().forward(
hidden_states[i : i + bs],
encoder_hidden_states=encoder_hidden_states[i : i + bs],
attention_mask=attention_mask
)
new_hidden_states.append(__hidden_states)
for i in range(0, query.shape[0], bs):
# attention, what we cannot get enough of
if self._use_memory_efficient_attention_xformers:
hidden_states = self._memory_efficient_attention_xformers(query[i : i + bs], key[i : i + bs], value[i : i + bs], attention_mask[i : i + bs] if attention_mask is not None else attention_mask)
# Some versions of xformers return output in fp32, cast it back to the dtype of the input
hidden_states = hidden_states.to(query.dtype)
else:
if self._slice_size is None or query[i : i + bs].shape[0] // self._slice_size == 1:
hidden_states = self._attention(query[i : i + bs], key[i : i + bs], value[i : i + bs], attention_mask[i : i + bs] if attention_mask is not None else attention_mask)
else:
hidden_states = self._sliced_attention(query[i : i + bs], key[i : i + bs], value[i : i + bs], sequence_length, dim, attention_mask[i : i + bs] if attention_mask is not None else attention_mask)
new_hidden_states.append(hidden_states)
hidden_states = torch.cat(new_hidden_states, dim = 0)
# linear proj
hidden_states = self.to_out[0](hidden_states)
# dropout
hidden_states = self.to_out[1](hidden_states)
if self.attention_mode == "Temporal":
hidden_states = rearrange(hidden_states, "(b d) f c -> (b f) d c", d=d)
if self.grid:
hidden_states = rearrange(hidden_states, "(b f n m) (h w) c -> (b f) h n w m c", f=video_length, n=self.block_size, m=self.block_size, h=height // self.block_size, w=weight // self.block_size)
hidden_states = rearrange(hidden_states, "b h n w m c -> b (h n) (w m) c")
hidden_states = rearrange(hidden_states, "b h w c -> b (h w) c")
elif self.attention_mode == "Global":
hidden_states = rearrange(hidden_states, "b (f d) c -> (b f) d c", f=video_length, d=d)
return hidden_states
-166
View File
@@ -1,166 +0,0 @@
from typing import Any, Dict, Optional, Tuple
import torch
import torch.nn.functional as F
from diffusers.models.embeddings import (CombinedTimestepLabelEmbeddings,
TimestepEmbedding, Timesteps)
from torch import nn
def zero_module(module):
# Zero out the parameters of a module and return it.
for p in module.parameters():
p.detach().zero_()
return module
class FP32LayerNorm(nn.LayerNorm):
def forward(self, inputs: torch.Tensor) -> torch.Tensor:
origin_dtype = inputs.dtype
if hasattr(self, 'weight') and self.weight is not None:
return F.layer_norm(
inputs.float(), self.normalized_shape, self.weight.float(), self.bias.float(), self.eps
).to(origin_dtype)
else:
return F.layer_norm(
inputs.float(), self.normalized_shape, None, None, self.eps
).to(origin_dtype)
class EasyAnimateRMSNorm(nn.Module):
def __init__(self, hidden_size, eps=1e-6):
super().__init__()
self.weight = nn.Parameter(torch.ones(hidden_size))
self.variance_epsilon = eps
def forward(self, hidden_states):
input_dtype = hidden_states.dtype
hidden_states = hidden_states.to(torch.float32)
variance = hidden_states.pow(2).mean(-1, keepdim=True)
hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
return self.weight * hidden_states.to(input_dtype)
def extra_repr(self):
return f"{tuple(self.weight.shape)}, eps={self.variance_epsilon}"
class PixArtAlphaCombinedTimestepSizeEmbeddings(nn.Module):
"""
For PixArt-Alpha.
Reference:
https://github.com/PixArt-alpha/PixArt-alpha/blob/0f55e922376d8b797edd44d25d0e7464b260dcab/diffusion/model/nets/PixArtMS.py#L164C9-L168C29
"""
def __init__(self, embedding_dim, size_emb_dim, use_additional_conditions: bool = False):
super().__init__()
self.outdim = size_emb_dim
self.time_proj = Timesteps(num_channels=256, flip_sin_to_cos=True, downscale_freq_shift=0)
self.timestep_embedder = TimestepEmbedding(in_channels=256, time_embed_dim=embedding_dim)
self.use_additional_conditions = use_additional_conditions
if use_additional_conditions:
self.additional_condition_proj = Timesteps(num_channels=256, flip_sin_to_cos=True, downscale_freq_shift=0)
self.resolution_embedder = TimestepEmbedding(in_channels=256, time_embed_dim=size_emb_dim)
self.aspect_ratio_embedder = TimestepEmbedding(in_channels=256, time_embed_dim=size_emb_dim)
self.resolution_embedder.linear_2 = zero_module(self.resolution_embedder.linear_2)
self.aspect_ratio_embedder.linear_2 = zero_module(self.aspect_ratio_embedder.linear_2)
def forward(self, timestep, resolution, aspect_ratio, batch_size, hidden_dtype):
timesteps_proj = self.time_proj(timestep)
timesteps_emb = self.timestep_embedder(timesteps_proj.to(dtype=hidden_dtype)) # (N, D)
if self.use_additional_conditions:
resolution_emb = self.additional_condition_proj(resolution.flatten()).to(hidden_dtype)
resolution_emb = self.resolution_embedder(resolution_emb).reshape(batch_size, -1)
aspect_ratio_emb = self.additional_condition_proj(aspect_ratio.flatten()).to(hidden_dtype)
aspect_ratio_emb = self.aspect_ratio_embedder(aspect_ratio_emb).reshape(batch_size, -1)
conditioning = timesteps_emb + torch.cat([resolution_emb, aspect_ratio_emb], dim=1)
else:
conditioning = timesteps_emb
return conditioning
class AdaLayerNormSingle(nn.Module):
r"""
Norm layer adaptive layer norm single (adaLN-single).
As proposed in PixArt-Alpha (see: https://arxiv.org/abs/2310.00426; Section 2.3).
Parameters:
embedding_dim (`int`): The size of each embedding vector.
use_additional_conditions (`bool`): To use additional conditions for normalization or not.
"""
def __init__(self, embedding_dim: int, use_additional_conditions: bool = False):
super().__init__()
self.emb = PixArtAlphaCombinedTimestepSizeEmbeddings(
embedding_dim, size_emb_dim=embedding_dim // 3, use_additional_conditions=use_additional_conditions
)
self.silu = nn.SiLU()
self.linear = nn.Linear(embedding_dim, 6 * embedding_dim, bias=True)
def forward(
self,
timestep: torch.Tensor,
added_cond_kwargs: Optional[Dict[str, torch.Tensor]] = None,
batch_size: Optional[int] = None,
hidden_dtype: Optional[torch.dtype] = None,
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
# No modulation happening here.
embedded_timestep = self.emb(timestep, **added_cond_kwargs, batch_size=batch_size, hidden_dtype=hidden_dtype)
return self.linear(self.silu(embedded_timestep)), embedded_timestep
class AdaLayerNormShift(nn.Module):
r"""
Norm layer modified to incorporate timestep embeddings.
Parameters:
embedding_dim (`int`): The size of each embedding vector.
num_embeddings (`int`): The size of the embeddings dictionary.
"""
def __init__(self, embedding_dim: int, elementwise_affine=True, eps=1e-6):
super().__init__()
self.silu = nn.SiLU()
self.linear = nn.Linear(embedding_dim, embedding_dim)
self.norm = FP32LayerNorm(embedding_dim, elementwise_affine=elementwise_affine, eps=eps)
def forward(self, x: torch.Tensor, emb: torch.Tensor) -> torch.Tensor:
shift = self.linear(self.silu(emb.to(torch.float32)).to(emb.dtype))
x = self.norm(x) + shift.unsqueeze(dim=1)
return x
class EasyAnimateLayerNormZero(nn.Module):
# Modified from https://github.com/huggingface/diffusers/blob/main/src/diffusers/models/normalization.py
# Add fp32 layer norm
def __init__(
self,
conditioning_dim: int,
embedding_dim: int,
elementwise_affine: bool = True,
eps: float = 1e-5,
bias: bool = True,
norm_type: str = "fp32_layer_norm",
) -> None:
super().__init__()
self.silu = nn.SiLU()
self.linear = nn.Linear(conditioning_dim, 6 * embedding_dim, bias=bias)
if norm_type == "layer_norm":
self.norm = nn.LayerNorm(embedding_dim, elementwise_affine=elementwise_affine, eps=eps)
elif norm_type == "fp32_layer_norm":
self.norm = FP32LayerNorm(embedding_dim, elementwise_affine=elementwise_affine, eps=eps)
else:
raise ValueError(
f"Unsupported `norm_type` ({norm_type}) provided. Supported ones are: 'layer_norm', 'fp32_layer_norm'."
)
def forward(
self, hidden_states: torch.Tensor, encoder_hidden_states: torch.Tensor, temb: torch.Tensor
) -> Tuple[torch.Tensor, torch.Tensor]:
shift, scale, gate, enc_shift, enc_scale, enc_gate = self.linear(self.silu(temb)).chunk(6, dim=1)
hidden_states = self.norm(hidden_states) * (1 + scale)[:, None, :] + shift[:, None, :]
encoder_hidden_states = self.norm(encoder_hidden_states) * (1 + enc_scale)[:, None, :] + enc_shift[:, None, :]
return hidden_states, encoder_hidden_states, gate[:, None, :], enc_gate[:, None, :]
+10 -1
View File
@@ -1,10 +1,10 @@
import math
from typing import Optional
import numpy as np
import torch
import torch.nn.functional as F
import torch.nn.init as init
import math
from einops import rearrange
from torch import nn
@@ -153,6 +153,15 @@ class TemporalUpsampler3D(Upsampler):
x = torch.cat([first_frame, x], dim=2)
return x
def cast_tuple(t, length = 1):
return t if isinstance(t, tuple) else ((t,) * length)
def divisible_by(num, den):
return (num % den) == 0
def is_odd(n):
return not divisible_by(n, 2)
class CausalConv3d(nn.Conv3d):
def __init__(
self,
-459
View File
@@ -1,459 +0,0 @@
from typing import Optional
import torch
import torch.nn.functional as F
from diffusers.models.attention import Attention
from diffusers.models.embeddings import apply_rotary_emb
from einops import rearrange, repeat
class HunyuanAttnProcessor2_0:
r"""
Processor for implementing scaled dot-product attention (enabled by default if you're using PyTorch 2.0). This is
used in the HunyuanDiT model. It applies a s normalization layer and rotary embedding on query and key vector.
"""
def __init__(self):
if not hasattr(F, "scaled_dot_product_attention"):
raise ImportError("AttnProcessor2_0 requires PyTorch 2.0, to use it, please upgrade PyTorch to 2.0.")
def __call__(
self,
attn: Attention,
hidden_states: torch.Tensor,
encoder_hidden_states: Optional[torch.Tensor] = None,
attention_mask: Optional[torch.Tensor] = None,
temb: Optional[torch.Tensor] = None,
image_rotary_emb: Optional[torch.Tensor] = None,
) -> torch.Tensor:
residual = hidden_states
if attn.spatial_norm is not None:
hidden_states = attn.spatial_norm(hidden_states, temb)
input_ndim = hidden_states.ndim
if input_ndim == 4:
batch_size, channel, height, width = hidden_states.shape
hidden_states = hidden_states.view(batch_size, channel, height * width).transpose(1, 2)
batch_size, sequence_length, _ = (
hidden_states.shape if encoder_hidden_states is None else encoder_hidden_states.shape
)
if attention_mask is not None:
attention_mask = attn.prepare_attention_mask(attention_mask, sequence_length, batch_size)
# scaled_dot_product_attention expects attention_mask shape to be
# (batch, heads, source_length, target_length)
attention_mask = attention_mask.view(batch_size, attn.heads, -1, attention_mask.shape[-1])
if attn.group_norm is not None:
hidden_states = attn.group_norm(hidden_states.transpose(1, 2)).transpose(1, 2)
query = attn.to_q(hidden_states)
if encoder_hidden_states is None:
encoder_hidden_states = hidden_states
elif attn.norm_cross:
encoder_hidden_states = attn.norm_encoder_hidden_states(encoder_hidden_states)
key = attn.to_k(encoder_hidden_states)
value = attn.to_v(encoder_hidden_states)
inner_dim = key.shape[-1]
head_dim = inner_dim // attn.heads
query = query.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
key = key.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
value = value.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
if attn.norm_q is not None:
query = attn.norm_q(query)
if attn.norm_k is not None:
key = attn.norm_k(key)
# Apply RoPE if needed
if image_rotary_emb is not None:
query = apply_rotary_emb(query, image_rotary_emb)
if not attn.is_cross_attention:
key = apply_rotary_emb(key, image_rotary_emb)
# the output of sdp = (batch, num_heads, seq_len, head_dim)
# TODO: add support for attn.scale when we move to Torch 2.1
hidden_states = F.scaled_dot_product_attention(
query, key, value, attn_mask=attention_mask, dropout_p=0.0, is_causal=False
)
hidden_states = hidden_states.transpose(1, 2).reshape(batch_size, -1, attn.heads * head_dim)
hidden_states = hidden_states.to(query.dtype)
# linear proj
hidden_states = attn.to_out[0](hidden_states)
# dropout
hidden_states = attn.to_out[1](hidden_states)
if input_ndim == 4:
hidden_states = hidden_states.transpose(-1, -2).reshape(batch_size, channel, height, width)
if attn.residual_connection:
hidden_states = hidden_states + residual
hidden_states = hidden_states / attn.rescale_output_factor
return hidden_states
class LazyKVCompressionProcessor2_0:
r"""
Processor for implementing scaled dot-product attention (enabled by default if you're using PyTorch 2.0). This is
used in the KVCompression model. It applies a s normalization layer and rotary embedding on query and key vector.
"""
def __init__(self):
if not hasattr(F, "scaled_dot_product_attention"):
raise ImportError("AttnProcessor2_0 requires PyTorch 2.0, to use it, please upgrade PyTorch to 2.0.")
def __call__(
self,
attn: Attention,
hidden_states: torch.Tensor,
encoder_hidden_states: Optional[torch.Tensor] = None,
attention_mask: Optional[torch.Tensor] = None,
temb: Optional[torch.Tensor] = None,
image_rotary_emb: Optional[torch.Tensor] = None,
) -> torch.Tensor:
residual = hidden_states
if attn.spatial_norm is not None:
hidden_states = attn.spatial_norm(hidden_states, temb)
input_ndim = hidden_states.ndim
batch_size, channel, num_frames, height, width = hidden_states.shape
hidden_states = rearrange(hidden_states, "b c f h w -> b (f h w) c", f=num_frames, h=height, w=width)
batch_size, sequence_length, _ = (
hidden_states.shape if encoder_hidden_states is None else encoder_hidden_states.shape
)
if attention_mask is not None:
attention_mask = attn.prepare_attention_mask(attention_mask, sequence_length, batch_size)
# scaled_dot_product_attention expects attention_mask shape to be
# (batch, heads, source_length, target_length)
attention_mask = attention_mask.view(batch_size, attn.heads, -1, attention_mask.shape[-1])
if attn.group_norm is not None:
hidden_states = attn.group_norm(hidden_states.transpose(1, 2)).transpose(1, 2)
query = attn.to_q(hidden_states)
if encoder_hidden_states is None:
encoder_hidden_states = hidden_states
elif attn.norm_cross:
encoder_hidden_states = attn.norm_encoder_hidden_states(encoder_hidden_states)
key = attn.to_k(encoder_hidden_states)
value = attn.to_v(encoder_hidden_states)
key = rearrange(key, "b (f h w) c -> (b f) c h w", f=num_frames, h=height, w=width)
key = attn.k_compression(key)
key_shape = key.size()
key = rearrange(key, "(b f) c h w -> b (f h w) c", f=num_frames)
value = rearrange(value, "b (f h w) c -> (b f) c h w", f=num_frames, h=height, w=width)
value = attn.v_compression(value)
value = rearrange(value, "(b f) c h w -> b (f h w) c", f=num_frames)
inner_dim = key.shape[-1]
head_dim = inner_dim // attn.heads
query = query.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
key = key.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
value = value.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
if attn.norm_q is not None:
query = attn.norm_q(query)
if attn.norm_k is not None:
key = attn.norm_k(key)
# Apply RoPE if needed
if image_rotary_emb is not None:
compression_image_rotary_emb = (
rearrange(image_rotary_emb[0], "(f h w) c -> f c h w", f=num_frames, h=height, w=width),
rearrange(image_rotary_emb[1], "(f h w) c -> f c h w", f=num_frames, h=height, w=width),
)
compression_image_rotary_emb = (
F.interpolate(compression_image_rotary_emb[0], size=key_shape[-2:], mode='bilinear'),
F.interpolate(compression_image_rotary_emb[1], size=key_shape[-2:], mode='bilinear')
)
compression_image_rotary_emb = (
rearrange(compression_image_rotary_emb[0], "f c h w -> (f h w) c"),
rearrange(compression_image_rotary_emb[1], "f c h w -> (f h w) c"),
)
query = apply_rotary_emb(query, image_rotary_emb)
if not attn.is_cross_attention:
key = apply_rotary_emb(key, compression_image_rotary_emb)
# the output of sdp = (batch, num_heads, seq_len, head_dim)
# TODO: add support for attn.scale when we move to Torch 2.1
hidden_states = F.scaled_dot_product_attention(
query, key, value, attn_mask=attention_mask, dropout_p=0.0, is_causal=False
)
hidden_states = hidden_states.transpose(1, 2).reshape(batch_size, -1, attn.heads * head_dim)
hidden_states = hidden_states.to(query.dtype)
# linear proj
hidden_states = attn.to_out[0](hidden_states)
# dropout
hidden_states = attn.to_out[1](hidden_states)
if attn.residual_connection:
hidden_states = hidden_states + residual
hidden_states = hidden_states / attn.rescale_output_factor
return hidden_states
class EasyAnimateAttnProcessor2_0:
def __init__(self):
pass
def __call__(
self,
attn: Attention,
hidden_states: torch.Tensor,
encoder_hidden_states: torch.Tensor,
attention_mask: Optional[torch.Tensor] = None,
image_rotary_emb: Optional[torch.Tensor] = None,
attn2: Attention = None,
) -> torch.Tensor:
text_seq_length = encoder_hidden_states.size(1)
batch_size, sequence_length, _ = (
hidden_states.shape if encoder_hidden_states is None else encoder_hidden_states.shape
)
if attention_mask is not None:
attention_mask = attn.prepare_attention_mask(attention_mask, sequence_length, batch_size)
attention_mask = attention_mask.view(batch_size, attn.heads, -1, attention_mask.shape[-1])
if attn2 is None:
hidden_states = torch.cat([encoder_hidden_states, hidden_states], dim=1)
query = attn.to_q(hidden_states)
key = attn.to_k(hidden_states)
value = attn.to_v(hidden_states)
inner_dim = key.shape[-1]
head_dim = inner_dim // attn.heads
query = query.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
key = key.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
value = value.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
if attn.norm_q is not None:
query = attn.norm_q(query)
if attn.norm_k is not None:
key = attn.norm_k(key)
if attn2 is not None:
query_txt = attn2.to_q(encoder_hidden_states)
key_txt = attn2.to_k(encoder_hidden_states)
value_txt = attn2.to_v(encoder_hidden_states)
inner_dim = key_txt.shape[-1]
head_dim = inner_dim // attn.heads
query_txt = query_txt.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
key_txt = key_txt.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
value_txt = value_txt.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
if attn2.norm_q is not None:
query_txt = attn2.norm_q(query_txt)
if attn2.norm_k is not None:
key_txt = attn2.norm_k(key_txt)
query = torch.cat([query_txt, query], dim=2)
key = torch.cat([key_txt, key], dim=2)
value = torch.cat([value_txt, value], dim=2)
# Apply RoPE if needed
if image_rotary_emb is not None:
query[:, :, text_seq_length:] = apply_rotary_emb(query[:, :, text_seq_length:], image_rotary_emb)
if not attn.is_cross_attention:
key[:, :, text_seq_length:] = apply_rotary_emb(key[:, :, text_seq_length:], image_rotary_emb)
hidden_states = F.scaled_dot_product_attention(
query, key, value, attn_mask=attention_mask, dropout_p=0.0, is_causal=False
)
hidden_states = hidden_states.transpose(1, 2).reshape(batch_size, -1, attn.heads * head_dim)
if attn2 is None:
# linear proj
hidden_states = attn.to_out[0](hidden_states)
# dropout
hidden_states = attn.to_out[1](hidden_states)
encoder_hidden_states, hidden_states = hidden_states.split(
[text_seq_length, hidden_states.size(1) - text_seq_length], dim=1
)
else:
encoder_hidden_states, hidden_states = hidden_states.split(
[text_seq_length, hidden_states.size(1) - text_seq_length], dim=1
)
# linear proj
hidden_states = attn.to_out[0](hidden_states)
encoder_hidden_states = attn2.to_out[0](encoder_hidden_states)
# dropout
hidden_states = attn.to_out[1](hidden_states)
encoder_hidden_states = attn2.to_out[1](encoder_hidden_states)
return hidden_states, encoder_hidden_states
try:
from flash_attn import flash_attn_func, flash_attn_varlen_func
from flash_attn.bert_padding import pad_input, unpad_input
except:
print("Flash Attention is not installed. Please install with `pip install flash-attn`, if you want to use SWA.")
class EasyAnimateSWAttnProcessor2_0:
def __init__(self, cross_attention_size=1024):
self.cross_attention_size = cross_attention_size
def __call__(
self,
attn: Attention,
hidden_states: torch.Tensor,
encoder_hidden_states: torch.Tensor,
attention_mask: Optional[torch.Tensor] = None,
image_rotary_emb: Optional[torch.Tensor] = None,
num_frames: int = None,
height: int = None,
width: int = None,
attn2: Attention = None,
) -> torch.Tensor:
text_seq_length = encoder_hidden_states.size(1)
windows_size = height * width
batch_size, sequence_length, _ = (
hidden_states.shape if encoder_hidden_states is None else encoder_hidden_states.shape
)
if attn2 is None:
hidden_states = torch.cat([encoder_hidden_states, hidden_states], dim=1)
query = attn.to_q(hidden_states)
key = attn.to_k(hidden_states)
value = attn.to_v(hidden_states)
inner_dim = key.shape[-1]
head_dim = inner_dim // attn.heads
query = query.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
key = key.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
value = value.view(batch_size, -1, attn.heads, head_dim)
if attn.norm_q is not None:
query = attn.norm_q(query)
if attn.norm_k is not None:
key = attn.norm_k(key)
if attn2 is not None:
query_txt = attn2.to_q(encoder_hidden_states)
key_txt = attn2.to_k(encoder_hidden_states)
value_txt = attn2.to_v(encoder_hidden_states)
inner_dim = key_txt.shape[-1]
head_dim = inner_dim // attn.heads
query_txt = query_txt.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
key_txt = key_txt.view(batch_size, -1, attn.heads, head_dim).transpose(1, 2)
value_txt = value_txt.view(batch_size, -1, attn.heads, head_dim)
if attn2.norm_q is not None:
query_txt = attn2.norm_q(query_txt)
if attn2.norm_k is not None:
key_txt = attn2.norm_k(key_txt)
query = torch.cat([query_txt, query], dim=2)
key = torch.cat([key_txt, key], dim=2)
value = torch.cat([value_txt, value], dim=1)
# Apply RoPE if needed
if image_rotary_emb is not None:
query[:, :, text_seq_length:] = apply_rotary_emb(query[:, :, text_seq_length:], image_rotary_emb)
if not attn.is_cross_attention:
key[:, :, text_seq_length:] = apply_rotary_emb(key[:, :, text_seq_length:], image_rotary_emb)
query = query.transpose(1, 2).to(value)
key = key.transpose(1, 2).to(value)
interval = max((query.size(1) - text_seq_length) // (self.cross_attention_size - text_seq_length), 1)
cross_key = torch.cat([key[:, :text_seq_length], key[:, text_seq_length::interval]], dim=1)
cross_val = torch.cat([value[:, :text_seq_length], value[:, text_seq_length::interval]], dim=1)
cross_hidden_states = flash_attn_func(query, cross_key, cross_val, dropout_p=0.0, causal=False)
# Split and rearrange to six directions
querys = torch.tensor_split(query[:, text_seq_length:], 6, 2)
keys = torch.tensor_split(key[:, text_seq_length:], 6, 2)
values = torch.tensor_split(value[:, text_seq_length:], 6, 2)
new_querys = [querys[0]]
new_keys = [keys[0]]
new_values = [values[0]]
for index, mode in enumerate(
[
"bs (f h w) hn hd -> bs (f w h) hn hd",
"bs (f h w) hn hd -> bs (h f w) hn hd",
"bs (f h w) hn hd -> bs (h w f) hn hd",
"bs (f h w) hn hd -> bs (w f h) hn hd",
"bs (f h w) hn hd -> bs (w h f) hn hd"
]
):
new_querys.append(rearrange(querys[index + 1], mode, f=num_frames, h=height, w=width))
new_keys.append(rearrange(keys[index + 1], mode, f=num_frames, h=height, w=width))
new_values.append(rearrange(values[index + 1], mode, f=num_frames, h=height, w=width))
query = torch.cat(new_querys, dim=2)
key = torch.cat(new_keys, dim=2)
value = torch.cat(new_values, dim=2)
# apply attention
hidden_states = flash_attn_func(query, key, value, dropout_p=0.0, causal=False, window_size=(windows_size, windows_size))
hidden_states = torch.tensor_split(hidden_states, 6, 2)
new_hidden_states = [hidden_states[0]]
for index, mode in enumerate(
[
"bs (f w h) hn hd -> bs (f h w) hn hd",
"bs (h f w) hn hd -> bs (f h w) hn hd",
"bs (h w f) hn hd -> bs (f h w) hn hd",
"bs (w f h) hn hd -> bs (f h w) hn hd",
"bs (w h f) hn hd -> bs (f h w) hn hd"
]
):
new_hidden_states.append(rearrange(hidden_states[index + 1], mode, f=num_frames, h=height, w=width))
hidden_states = torch.cat([cross_hidden_states[:, :text_seq_length], torch.cat(new_hidden_states, dim=2)], dim=1) + cross_hidden_states
hidden_states = hidden_states.reshape(batch_size, -1, attn.heads * head_dim)
if attn2 is None:
# linear proj
hidden_states = attn.to_out[0](hidden_states)
# dropout
hidden_states = attn.to_out[1](hidden_states)
encoder_hidden_states, hidden_states = hidden_states.split(
[text_seq_length, hidden_states.size(1) - text_seq_length], dim=1
)
else:
encoder_hidden_states, hidden_states = hidden_states.split(
[text_seq_length, hidden_states.size(1) - text_seq_length], dim=1
)
# linear proj
hidden_states = attn.to_out[0](hidden_states)
encoder_hidden_states = attn2.to_out[0](encoder_hidden_states)
# dropout
hidden_states = attn.to_out[1](hidden_states)
encoder_hidden_states = attn2.to_out[1](encoder_hidden_states)
return hidden_states, encoder_hidden_states
-146
View File
@@ -1,146 +0,0 @@
# Copyright (c) Alibaba Cloud.
#
# This source code is licensed under the license found in the
# LICENSE file in the root directory of this source tree.
import math
import numpy as np
import torch
from torch import nn
from torch.nn import functional as F
from torch.nn.init import normal_
def get_abs_pos(abs_pos, tgt_size):
# abs_pos: L, C
# tgt_size: M
# return: M, C
src_size = int(math.sqrt(abs_pos.size(0)))
tgt_size = int(math.sqrt(tgt_size))
dtype = abs_pos.dtype
if src_size != tgt_size:
return F.interpolate(
abs_pos.float().reshape(1, src_size, src_size, -1).permute(0, 3, 1, 2),
size=(tgt_size, tgt_size),
mode="bicubic",
align_corners=False,
).permute(0, 2, 3, 1).flatten(0, 2).to(dtype=dtype)
else:
return abs_pos
# https://github.com/facebookresearch/mae/blob/efb2a8062c206524e35e47d04501ed4f544c0ae8/util/pos_embed.py#L20
def get_2d_sincos_pos_embed(embed_dim, grid_size, cls_token=False):
"""
grid_size: int of the grid height and width
return:
pos_embed: [grid_size*grid_size, embed_dim] or [1+grid_size*grid_size, embed_dim] (w/ or w/o cls_token)
"""
grid_h = np.arange(grid_size, dtype=np.float32)
grid_w = np.arange(grid_size, dtype=np.float32)
grid = np.meshgrid(grid_w, grid_h) # here w goes first
grid = np.stack(grid, axis=0)
grid = grid.reshape([2, 1, grid_size, grid_size])
pos_embed = get_2d_sincos_pos_embed_from_grid(embed_dim, grid)
if cls_token:
pos_embed = np.concatenate([np.zeros([1, embed_dim]), pos_embed], axis=0)
return pos_embed
def get_2d_sincos_pos_embed_from_grid(embed_dim, grid):
assert embed_dim % 2 == 0
# use half of dimensions to encode grid_h
emb_h = get_1d_sincos_pos_embed_from_grid(embed_dim // 2, grid[0]) # (H*W, D/2)
emb_w = get_1d_sincos_pos_embed_from_grid(embed_dim // 2, grid[1]) # (H*W, D/2)
emb = np.concatenate([emb_h, emb_w], axis=1) # (H*W, D)
return emb
def get_1d_sincos_pos_embed_from_grid(embed_dim, pos):
"""
embed_dim: output dimension for each position
pos: a list of positions to be encoded: size (M,)
out: (M, D)
"""
assert embed_dim % 2 == 0
omega = np.arange(embed_dim // 2, dtype=np.float32)
omega /= embed_dim / 2.
omega = 1. / 10000**omega # (D/2,)
pos = pos.reshape(-1) # (M,)
out = np.einsum('m,d->md', pos, omega) # (M, D/2), outer product
emb_sin = np.sin(out) # (M, D/2)
emb_cos = np.cos(out) # (M, D/2)
emb = np.concatenate([emb_sin, emb_cos], axis=1) # (M, D)
return emb
class Resampler(nn.Module):
"""
A 2D perceiver-resampler network with one cross attention layers by
(grid_size**2) learnable queries and 2d sincos pos_emb
Outputs:
A tensor with the shape of (grid_size**2, embed_dim)
"""
def __init__(
self,
grid_size,
embed_dim,
num_heads,
kv_dim=None,
norm_layer=nn.LayerNorm
):
super().__init__()
self.num_queries = grid_size ** 2
self.embed_dim = embed_dim
self.num_heads = num_heads
self.pos_embed = nn.Parameter(
torch.from_numpy(get_2d_sincos_pos_embed(embed_dim, grid_size)).float()
).requires_grad_(False)
self.query = nn.Parameter(torch.zeros(self.num_queries, embed_dim))
normal_(self.query, std=.02)
if kv_dim is not None and kv_dim != embed_dim:
self.kv_proj = nn.Linear(kv_dim, embed_dim, bias=False)
else:
self.kv_proj = nn.Identity()
self.attn = nn.MultiheadAttention(embed_dim, num_heads)
self.ln_q = norm_layer(embed_dim)
self.ln_kv = norm_layer(embed_dim)
self.apply(self._init_weights)
def _init_weights(self, m):
if isinstance(m, nn.Linear):
normal_(m.weight, std=.02)
if isinstance(m, nn.Linear) and m.bias is not None:
nn.init.constant_(m.bias, 0)
elif isinstance(m, nn.LayerNorm):
nn.init.constant_(m.bias, 0)
nn.init.constant_(m.weight, 1.0)
def forward(self, x, key_padding_mask=None):
pos_embed = get_abs_pos(self.pos_embed, x.size(1))
x = self.kv_proj(x)
x = self.ln_kv(x).permute(1, 0, 2)
N = x.shape[1]
q = self.ln_q(self.query)
out = self.attn(
self._repeat(q, N) + self.pos_embed.unsqueeze(1),
x + pos_embed.unsqueeze(1),
x,
key_padding_mask=key_padding_mask)[0]
return out.permute(1, 0, 2)
def _repeat(self, query, N: int):
return query.unsqueeze(1).repeat(1, N, 1)
+59 -24
View File
@@ -37,6 +37,10 @@ except:
from diffusers.models.embeddings import \
CaptionProjection as PixArtAlphaTextProjection
from .attention import (KVCompressionTransformerBlock,
SelfAttentionTemporalTransformerBlock,
TemporalTransformerBlock)
@dataclass
class Transformer2DModelOutput(BaseOutput):
@@ -192,29 +196,58 @@ class Transformer2DModel(ModelMixin, ConfigMixin):
interpolation_scale=interpolation_scale,
)
# 3. Define transformers blocks
self.transformer_blocks = nn.ModuleList(
[
BasicTransformerBlock(
inner_dim,
num_attention_heads,
attention_head_dim,
dropout=dropout,
cross_attention_dim=cross_attention_dim,
activation_fn=activation_fn,
num_embeds_ada_norm=num_embeds_ada_norm,
attention_bias=attention_bias,
only_cross_attention=only_cross_attention,
double_self_attention=double_self_attention,
upcast_attention=upcast_attention,
norm_type=norm_type,
norm_elementwise_affine=norm_elementwise_affine,
norm_eps=norm_eps,
attention_type=attention_type,
)
for d in range(num_layers)
]
)
basic_block = {
"basic": BasicTransformerBlock,
"kvcompression": KVCompressionTransformerBlock,
}[self.basic_block_type]
if self.basic_block_type == "kvcompression":
self.transformer_blocks = nn.ModuleList(
[
basic_block(
inner_dim,
num_attention_heads,
attention_head_dim,
dropout=dropout,
cross_attention_dim=cross_attention_dim,
activation_fn=activation_fn,
num_embeds_ada_norm=num_embeds_ada_norm,
attention_bias=attention_bias,
only_cross_attention=only_cross_attention,
double_self_attention=double_self_attention,
upcast_attention=upcast_attention,
norm_type=norm_type,
norm_elementwise_affine=norm_elementwise_affine,
norm_eps=norm_eps,
attention_type=attention_type,
kvcompression=False if d < 14 else True,
)
for d in range(num_layers)
]
)
else:
# 3. Define transformers blocks
self.transformer_blocks = nn.ModuleList(
[
BasicTransformerBlock(
inner_dim,
num_attention_heads,
attention_head_dim,
dropout=dropout,
cross_attention_dim=cross_attention_dim,
activation_fn=activation_fn,
num_embeds_ada_norm=num_embeds_ada_norm,
attention_bias=attention_bias,
only_cross_attention=only_cross_attention,
double_self_attention=double_self_attention,
upcast_attention=upcast_attention,
norm_type=norm_type,
norm_elementwise_affine=norm_elementwise_affine,
norm_eps=norm_eps,
attention_type=attention_type,
)
for d in range(num_layers)
]
)
# 4. Define output layers
self.out_channels = in_channels if out_channels is None else out_channels
@@ -377,9 +410,10 @@ class Transformer2DModel(ModelMixin, ConfigMixin):
encoder_hidden_states = encoder_hidden_states.view(batch_size, -1, hidden_states.shape[-1])
for block in self.transformer_blocks:
if torch.is_grad_enabled() and self.gradient_checkpointing:
if self.training and self.gradient_checkpointing:
args = {
"basic": [],
"kvcompression": [1, height, width],
}[self.basic_block_type]
hidden_states = torch.utils.checkpoint.checkpoint(
block,
@@ -396,6 +430,7 @@ class Transformer2DModel(ModelMixin, ConfigMixin):
else:
kwargs = {
"basic": {},
"kvcompression": {"num_frames":1, "height":height, "width":width},
}[self.basic_block_type]
hidden_states = block(
hidden_states,
+126 -1197
View File
File diff suppressed because it is too large Load Diff
+504 -806
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+594 -1215
View File
File diff suppressed because it is too large Load Diff
-1
View File
@@ -1 +0,0 @@
This folder is modified from the official [MPS](https://github.com/Kwai-Kolors/MPS/tree/main) repository.
@@ -1,7 +0,0 @@
from dataclasses import dataclass
@dataclass
class BaseModelConfig:
pass
@@ -1,154 +0,0 @@
from dataclasses import dataclass
from transformers import CLIPModel as HFCLIPModel
from transformers import AutoTokenizer
from torch import nn, einsum
# Modified: import
# from trainer.models.base_model import BaseModelConfig
from .base_model import BaseModelConfig
from transformers import CLIPConfig
from typing import Any, Optional, Tuple, Union
import torch
# Modified: import
# from trainer.models.cross_modeling import Cross_model
from .cross_modeling import Cross_model
import gc
class XCLIPModel(HFCLIPModel):
def __init__(self, config: CLIPConfig):
super().__init__(config)
def get_text_features(
self,
input_ids: Optional[torch.Tensor] = None,
attention_mask: Optional[torch.Tensor] = None,
position_ids: Optional[torch.Tensor] = None,
output_attentions: Optional[bool] = None,
output_hidden_states: Optional[bool] = None,
return_dict: Optional[bool] = None,
) -> torch.FloatTensor:
# Use CLIP model's config for some fields (if specified) instead of those of vision & text components.
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
output_hidden_states = (
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
)
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
text_outputs = self.text_model(
input_ids=input_ids,
attention_mask=attention_mask,
position_ids=position_ids,
output_attentions=output_attentions,
output_hidden_states=output_hidden_states,
return_dict=return_dict,
)
# pooled_output = text_outputs[1]
# text_features = self.text_projection(pooled_output)
last_hidden_state = text_outputs[0]
text_features = self.text_projection(last_hidden_state)
pooled_output = text_outputs[1]
text_features_EOS = self.text_projection(pooled_output)
# del last_hidden_state, text_outputs
# gc.collect()
return text_features, text_features_EOS
def get_image_features(
self,
pixel_values: Optional[torch.FloatTensor] = None,
output_attentions: Optional[bool] = None,
output_hidden_states: Optional[bool] = None,
return_dict: Optional[bool] = None,
) -> torch.FloatTensor:
# Use CLIP model's config for some fields (if specified) instead of those of vision & text components.
output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
output_hidden_states = (
output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
)
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
vision_outputs = self.vision_model(
pixel_values=pixel_values,
output_attentions=output_attentions,
output_hidden_states=output_hidden_states,
return_dict=return_dict,
)
# pooled_output = vision_outputs[1] # pooled_output
# image_features = self.visual_projection(pooled_output)
last_hidden_state = vision_outputs[0]
image_features = self.visual_projection(last_hidden_state)
return image_features
@dataclass
class ClipModelConfig(BaseModelConfig):
_target_: str = "trainer.models.clip_model.CLIPModel"
pretrained_model_name_or_path: str ="openai/clip-vit-base-patch32"
class CLIPModel(nn.Module):
def __init__(self, config):
super().__init__()
# Modified: We convert the original ckpt (contains the entire model) to a `state_dict`.
# self.model = XCLIPModel.from_pretrained(ckpt)
self.model = XCLIPModel(config)
self.cross_model = Cross_model(dim=1024, layer_num=4, heads=16)
def get_text_features(self, *args, **kwargs):
return self.model.get_text_features(*args, **kwargs)
def get_image_features(self, *args, **kwargs):
return self.model.get_image_features(*args, **kwargs)
def forward(self, text_inputs=None, image_inputs=None, condition_inputs=None):
outputs = ()
text_f, text_EOS = self.model.get_text_features(text_inputs) # B*77*1024
outputs += text_EOS,
image_f = self.model.get_image_features(image_inputs.half()) # 2B*257*1024
# [B, 77, 1024]
condition_f, _ = self.model.get_text_features(condition_inputs) # B*5*1024
sim_text_condition = einsum('b i d, b j d -> b j i', text_f, condition_f)
sim_text_condition = torch.max(sim_text_condition, dim=1, keepdim=True)[0]
sim_text_condition = sim_text_condition / sim_text_condition.max()
mask = torch.where(sim_text_condition > 0.01, 0, float('-inf')) # B*1*77
# Modified: Support both torch.float16 and torch.bfloat16
# mask = mask.repeat(1,image_f.shape[1],1) # B*257*77
model_dtype = next(self.cross_model.parameters()).dtype
mask = mask.repeat(1,image_f.shape[1],1).to(model_dtype) # B*257*77
# bc = int(image_f.shape[0]/2)
# Modified: The original input consists of a (batch of) text and two (batches of) images,
# primarily used to compute which (batch of) image is more consistent with the text.
# The modified input consists of a (batch of) text and a (batch of) images.
# sim0 = self.cross_model(image_f[:bc,:,:], text_f,mask.half())
# sim1 = self.cross_model(image_f[bc:,:,:], text_f,mask.half())
# outputs += sim0[:,0,:],
# outputs += sim1[:,0,:],
sim = self.cross_model(image_f, text_f,mask)
outputs += sim[:,0,:],
return outputs
@property
def logit_scale(self):
return self.model.logit_scale
def save(self, path):
self.model.save_pretrained(path)
@@ -1,291 +0,0 @@
import torch
from torch import einsum, nn
import torch.nn.functional as F
from einops import rearrange, repeat
# helper functions
def exists(val):
return val is not None
def default(val, d):
return val if exists(val) else d
# normalization
# they use layernorm without bias, something that pytorch does not offer
class LayerNorm(nn.Module):
def __init__(self, dim):
super().__init__()
self.weight = nn.Parameter(torch.ones(dim))
self.register_buffer("bias", torch.zeros(dim))
def forward(self, x):
return F.layer_norm(x, x.shape[-1:], self.weight, self.bias)
# residual
class Residual(nn.Module):
def __init__(self, fn):
super().__init__()
self.fn = fn
def forward(self, x, *args, **kwargs):
return self.fn(x, *args, **kwargs) + x
# rotary positional embedding
# https://arxiv.org/abs/2104.09864
class RotaryEmbedding(nn.Module):
def __init__(self, dim):
super().__init__()
inv_freq = 1.0 / (10000 ** (torch.arange(0, dim, 2).float() / dim))
self.register_buffer("inv_freq", inv_freq)
def forward(self, max_seq_len, *, device):
seq = torch.arange(max_seq_len, device=device, dtype=self.inv_freq.dtype)
freqs = einsum("i , j -> i j", seq, self.inv_freq)
return torch.cat((freqs, freqs), dim=-1)
def rotate_half(x):
x = rearrange(x, "... (j d) -> ... j d", j=2)
x1, x2 = x.unbind(dim=-2)
return torch.cat((-x2, x1), dim=-1)
def apply_rotary_pos_emb(pos, t):
return (t * pos.cos()) + (rotate_half(t) * pos.sin())
# classic Noam Shazeer paper, except here they use SwiGLU instead of the more popular GEGLU for gating the feedforward
# https://arxiv.org/abs/2002.05202
class SwiGLU(nn.Module):
def forward(self, x):
x, gate = x.chunk(2, dim=-1)
return F.silu(gate) * x
# parallel attention and feedforward with residual
# discovered by Wang et al + EleutherAI from GPT-J fame
class ParallelTransformerBlock(nn.Module):
def __init__(self, dim, dim_head=64, heads=8, ff_mult=4):
super().__init__()
self.norm = LayerNorm(dim)
attn_inner_dim = dim_head * heads
ff_inner_dim = dim * ff_mult
self.fused_dims = (attn_inner_dim, dim_head, dim_head, (ff_inner_dim * 2))
self.heads = heads
self.scale = dim_head**-0.5
self.rotary_emb = RotaryEmbedding(dim_head)
self.fused_attn_ff_proj = nn.Linear(dim, sum(self.fused_dims), bias=False)
self.attn_out = nn.Linear(attn_inner_dim, dim, bias=False)
self.ff_out = nn.Sequential(
SwiGLU(),
nn.Linear(ff_inner_dim, dim, bias=False)
)
self.register_buffer("pos_emb", None, persistent=False)
def get_rotary_embedding(self, n, device):
if self.pos_emb is not None and self.pos_emb.shape[-2] >= n:
return self.pos_emb[:n]
pos_emb = self.rotary_emb(n, device=device)
self.register_buffer("pos_emb", pos_emb, persistent=False)
return pos_emb
def forward(self, x, attn_mask=None):
"""
einstein notation
b - batch
h - heads
n, i, j - sequence length (base sequence length, source, target)
d - feature dimension
"""
n, device, h = x.shape[1], x.device, self.heads
# pre layernorm
x = self.norm(x)
# attention queries, keys, values, and feedforward inner
q, k, v, ff = self.fused_attn_ff_proj(x).split(self.fused_dims, dim=-1)
# split heads
# they use multi-query single-key-value attention, yet another Noam Shazeer paper
# they found no performance loss past a certain scale, and more efficient decoding obviously
# https://arxiv.org/abs/1911.02150
q = rearrange(q, "b n (h d) -> b h n d", h=h)
# rotary embeddings
positions = self.get_rotary_embedding(n, device)
q, k = map(lambda t: apply_rotary_pos_emb(positions, t), (q, k))
# scale
q = q * self.scale
# similarity
sim = einsum("b h i d, b j d -> b h i j", q, k)
# extra attention mask - for masking out attention from text CLS token to padding
if exists(attn_mask):
attn_mask = rearrange(attn_mask, 'b i j -> b 1 i j')
sim = sim.masked_fill(~attn_mask, -torch.finfo(sim.dtype).max)
# attention
sim = sim - sim.amax(dim=-1, keepdim=True).detach()
attn = sim.softmax(dim=-1)
# aggregate values
out = einsum("b h i j, b j d -> b h i d", attn, v)
# merge heads
out = rearrange(out, "b h n d -> b n (h d)")
return self.attn_out(out) + self.ff_out(ff)
# cross attention - using multi-query + one-headed key / values as in PaLM w/ optional parallel feedforward
class CrossAttention(nn.Module):
def __init__(
self,
dim,
*,
context_dim=None,
dim_head=64,
heads=12,
parallel_ff=False,
ff_mult=4,
norm_context=False
):
super().__init__()
self.heads = heads
self.scale = dim_head ** -0.5
inner_dim = heads * dim_head
context_dim = default(context_dim, dim)
self.norm = LayerNorm(dim)
self.context_norm = LayerNorm(context_dim) if norm_context else nn.Identity()
self.to_q = nn.Linear(dim, inner_dim, bias=False)
self.to_kv = nn.Linear(context_dim, dim_head * 2, bias=False)
self.to_out = nn.Linear(inner_dim, dim, bias=False)
# whether to have parallel feedforward
ff_inner_dim = ff_mult * dim
self.ff = nn.Sequential(
nn.Linear(dim, ff_inner_dim * 2, bias=False),
SwiGLU(),
nn.Linear(ff_inner_dim, dim, bias=False)
) if parallel_ff else None
def forward(self, x, context, mask):
"""
einstein notation
b - batch
h - heads
n, i, j - sequence length (base sequence length, source, target)
d - feature dimension
"""
# pre-layernorm, for queries and context
x = self.norm(x)
context = self.context_norm(context)
# get queries
q = self.to_q(x)
q = rearrange(q, 'b n (h d) -> b h n d', h = self.heads)
# scale
q = q * self.scale
# get key / values
k, v = self.to_kv(context).chunk(2, dim=-1)
# query / key similarity
sim = einsum('b h i d, b j d -> b h i j', q, k)
# attention
mask = mask.unsqueeze(1).repeat(1,self.heads,1,1)
sim = sim + mask # context mask
sim = sim - sim.amax(dim=-1, keepdim=True)
attn = sim.softmax(dim=-1)
# aggregate
out = einsum('b h i j, b j d -> b h i d', attn, v)
# merge and combine heads
out = rearrange(out, 'b h n d -> b n (h d)')
out = self.to_out(out)
# add parallel feedforward (for multimodal layers)
if exists(self.ff):
out = out + self.ff(x)
return out
class Cross_model(nn.Module):
def __init__(
self,
dim=512,
layer_num=4,
dim_head=64,
heads=8,
ff_mult=4
):
super().__init__()
self.layers = nn.ModuleList([])
for ind in range(layer_num):
self.layers.append(nn.ModuleList([
Residual(CrossAttention(dim=dim, dim_head=dim_head, heads=heads, parallel_ff=True, ff_mult=ff_mult)),
Residual(ParallelTransformerBlock(dim=dim, dim_head=dim_head, heads=heads, ff_mult=ff_mult))
]))
def forward(
self,
query_tokens,
context_tokens,
mask
):
for cross_attn, self_attn_ff in self.layers:
query_tokens = cross_attn(query_tokens, context_tokens,mask)
query_tokens = self_attn_ff(query_tokens)
return query_tokens
@@ -1,13 +0,0 @@
from .siglip_v2_5 import (
AestheticPredictorV2_5Head,
AestheticPredictorV2_5Model,
AestheticPredictorV2_5Processor,
convert_v2_5_from_siglip,
)
__all__ = [
"AestheticPredictorV2_5Head",
"AestheticPredictorV2_5Model",
"AestheticPredictorV2_5Processor",
"convert_v2_5_from_siglip",
]
@@ -1,133 +0,0 @@
# Borrowed from https://github.com/discus0434/aesthetic-predictor-v2-5/blob/3125a9e/src/aesthetic_predictor_v2_5/siglip_v2_5.py
import os
from collections import OrderedDict
from os import PathLike
from typing import Final
import torch
import torch.nn as nn
import torchvision.transforms as transforms
from transformers import (
SiglipImageProcessor,
SiglipVisionConfig,
SiglipVisionModel,
logging,
)
from transformers.image_processing_utils import BatchFeature
from transformers.modeling_outputs import ImageClassifierOutputWithNoAttention
logging.set_verbosity_error()
URL: Final[str] = (
"https://github.com/discus0434/aesthetic-predictor-v2-5/raw/main/models/aesthetic_predictor_v2_5.pth"
)
class AestheticPredictorV2_5Head(nn.Module):
def __init__(self, config: SiglipVisionConfig) -> None:
super().__init__()
self.scoring_head = nn.Sequential(
nn.Linear(config.hidden_size, 1024),
nn.Dropout(0.5),
nn.Linear(1024, 128),
nn.Dropout(0.5),
nn.Linear(128, 64),
nn.Dropout(0.5),
nn.Linear(64, 16),
nn.Dropout(0.2),
nn.Linear(16, 1),
)
def forward(self, image_embeds: torch.Tensor) -> torch.Tensor:
return self.scoring_head(image_embeds)
class AestheticPredictorV2_5Model(SiglipVisionModel):
PATCH_SIZE = 14
def __init__(self, config: SiglipVisionConfig, *args, **kwargs) -> None:
super().__init__(config, *args, **kwargs)
self.layers = AestheticPredictorV2_5Head(config)
self.post_init()
self.transforms = transforms.Compose([
transforms.Resize((384, 384)),
transforms.ToTensor(),
transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5]),
])
def forward(
self,
pixel_values: torch.FloatTensor | None = None,
labels: torch.Tensor | None = None,
return_dict: bool | None = None,
) -> tuple | ImageClassifierOutputWithNoAttention:
return_dict = (
return_dict if return_dict is not None else self.config.use_return_dict
)
outputs = super().forward(
pixel_values=pixel_values,
return_dict=return_dict,
)
image_embeds = outputs.pooler_output
image_embeds_norm = image_embeds / image_embeds.norm(dim=-1, keepdim=True)
prediction = self.layers(image_embeds_norm)
loss = None
if labels is not None:
loss_fct = nn.MSELoss()
loss = loss_fct()
if not return_dict:
return (loss, prediction, image_embeds)
return ImageClassifierOutputWithNoAttention(
loss=loss,
logits=prediction,
hidden_states=image_embeds,
)
class AestheticPredictorV2_5Processor(SiglipImageProcessor):
def __init__(self, *args, **kwargs) -> None:
super().__init__(*args, **kwargs)
def __call__(self, *args, **kwargs) -> BatchFeature:
return super().__call__(*args, **kwargs)
@classmethod
def from_pretrained(
self,
pretrained_model_name_or_path: str
| PathLike = "google/siglip-so400m-patch14-384",
*args,
**kwargs,
) -> "AestheticPredictorV2_5Processor":
return super().from_pretrained(pretrained_model_name_or_path, *args, **kwargs)
def convert_v2_5_from_siglip(
predictor_name_or_path: str | PathLike | None = None,
encoder_model_name: str = "google/siglip-so400m-patch14-384",
*args,
**kwargs,
) -> tuple[AestheticPredictorV2_5Model, AestheticPredictorV2_5Processor]:
model = AestheticPredictorV2_5Model.from_pretrained(
encoder_model_name, *args, **kwargs
)
processor = AestheticPredictorV2_5Processor.from_pretrained(
encoder_model_name, *args, **kwargs
)
if predictor_name_or_path is None or not os.path.exists(predictor_name_or_path):
state_dict = torch.hub.load_state_dict_from_url(URL, map_location="cpu")
else:
state_dict = torch.load(predictor_name_or_path, map_location="cpu")
assert isinstance(state_dict, OrderedDict)
model.layers.load_state_dict(state_dict)
model.eval()
return model, processor
@@ -1,49 +0,0 @@
import os
import torch
import torch.nn as nn
from transformers import CLIPModel
from torchvision.datasets.utils import download_url
URL = "https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Third_Party/sac%2Blogos%2Bava1-l14-linearMSE.pth"
FILENAME = "sac+logos+ava1-l14-linearMSE.pth"
MD5 = "b1047fd767a00134b8fd6529bf19521a"
class MLP(nn.Module):
def __init__(self):
super().__init__()
self.layers = nn.Sequential(
nn.Linear(768, 1024),
nn.Dropout(0.2),
nn.Linear(1024, 128),
nn.Dropout(0.2),
nn.Linear(128, 64),
nn.Dropout(0.1),
nn.Linear(64, 16),
nn.Linear(16, 1),
)
def forward(self, embed):
return self.layers(embed)
class ImprovedAestheticPredictor(nn.Module):
def __init__(self, encoder_path="openai/clip-vit-large-patch14", predictor_path=None):
super().__init__()
self.encoder = CLIPModel.from_pretrained(encoder_path)
self.predictor = MLP()
if predictor_path is None or not os.path.exists(predictor_path):
download_url(URL, torch.hub.get_dir(), FILENAME, md5=MD5)
predictor_path = os.path.join(torch.hub.get_dir(), FILENAME)
state_dict = torch.load(predictor_path, map_location="cpu")
self.predictor.load_state_dict(state_dict)
self.eval()
def forward(self, pixel_values):
embed = self.encoder.get_image_features(pixel_values=pixel_values)
embed = embed / torch.linalg.vector_norm(embed, dim=-1, keepdim=True)
return self.predictor(embed).squeeze(1)
-385
View File
@@ -1,385 +0,0 @@
import os
from abc import ABC, abstractmethod
import torch
import torchvision.transforms as transforms
from einops import rearrange
from torchvision.datasets.utils import download_url
from typing import Optional, Tuple
# All reward models.
__all__ = ["AestheticReward", "HPSReward", "PickScoreReward", "MPSReward"]
class BaseReward(ABC):
"""An base class for reward models. A custom Reward class must implement two functions below.
"""
def __init__(self):
"""Define your reward model and image transformations (optional) here.
"""
pass
@abstractmethod
def __call__(self, batch_frames: torch.Tensor, batch_prompt: Optional[list[str]]=None) -> Tuple[torch.Tensor, torch.Tensor]:
"""Given batch frames with shape `[B, C, T, H, W]` extracted from a list of videos and a list of prompts
(optional) correspondingly, return the loss and reward computed by your reward model (reduction by mean).
"""
pass
class AestheticReward(BaseReward):
"""Aesthetic Predictor [V2](https://github.com/christophschuhmann/improved-aesthetic-predictor)
and [V2.5](https://github.com/discus0434/aesthetic-predictor-v2-5) reward model.
"""
def __init__(
self,
encoder_path="openai/clip-vit-large-patch14",
predictor_path=None,
version="v2",
device="cpu",
dtype=torch.float16,
max_reward=10,
loss_scale=0.1,
):
from .improved_aesthetic_predictor import ImprovedAestheticPredictor
from ..video_caption.utils.siglip_v2_5 import convert_v2_5_from_siglip
self.encoder_path = encoder_path
self.predictor_path = predictor_path
self.version = version
self.device = device
self.dtype = dtype
self.max_reward = max_reward
self.loss_scale = loss_scale
if self.version != "v2" and self.version != "v2.5":
raise ValueError("Only v2 and v2.5 are supported.")
if self.version == "v2":
assert "clip-vit-large-patch14" in encoder_path.lower()
self.model = ImprovedAestheticPredictor(encoder_path=self.encoder_path, predictor_path=self.predictor_path)
# https://huggingface.co/openai/clip-vit-large-patch14/blob/main/preprocessor_config.json
# TODO: [transforms.Resize(224), transforms.CenterCrop(224)] for any aspect ratio.
self.transform = transforms.Compose([
transforms.Resize((224, 224), interpolation=transforms.InterpolationMode.BICUBIC),
transforms.Normalize(mean=[0.48145466, 0.4578275, 0.40821073], std=[0.26862954, 0.26130258, 0.27577711]),
])
elif self.version == "v2.5":
assert "siglip-so400m-patch14-384" in encoder_path.lower()
self.model, _ = convert_v2_5_from_siglip(encoder_model_name=self.encoder_path)
# https://huggingface.co/google/siglip-so400m-patch14-384/blob/main/preprocessor_config.json
self.transform = transforms.Compose([
transforms.Resize((384, 384), interpolation=transforms.InterpolationMode.BICUBIC),
transforms.Normalize(mean=[0.5, 0.5, 0.5], std=[0.5, 0.5, 0.5]),
])
self.model.to(device=self.device, dtype=self.dtype)
self.model.requires_grad_(False)
def __call__(self, batch_frames: torch.Tensor, batch_prompt: Optional[list[str]]=None) -> Tuple[torch.Tensor, torch.Tensor]:
batch_frames = rearrange(batch_frames, "b c t h w -> t b c h w")
batch_loss, batch_reward = 0, 0
for frames in batch_frames:
pixel_values = torch.stack([self.transform(frame) for frame in frames])
pixel_values = pixel_values.to(self.device, dtype=self.dtype)
if self.version == "v2":
reward = self.model(pixel_values)
elif self.version == "v2.5":
reward = self.model(pixel_values).logits.squeeze()
# Convert reward to loss in [0, 1].
if self.max_reward is None:
loss = (-1 * reward) * self.loss_scale
else:
loss = abs(reward - self.max_reward) * self.loss_scale
batch_loss, batch_reward = batch_loss + loss.mean(), batch_reward + reward.mean()
return batch_loss / batch_frames.shape[0], batch_reward / batch_frames.shape[0]
class HPSReward(BaseReward):
"""[HPS](https://github.com/tgxs002/HPSv2) v2 and v2.1 reward model.
"""
def __init__(
self,
model_path=None,
version="v2.0",
device="cpu",
dtype=torch.float16,
max_reward=1,
loss_scale=1,
):
from hpsv2.src.open_clip import create_model_and_transforms, get_tokenizer
self.model_path = model_path
self.version = version
self.device = device
self.dtype = dtype
self.max_reward = max_reward
self.loss_scale = loss_scale
self.model, _, _ = create_model_and_transforms(
"ViT-H-14",
"laion2B-s32B-b79K",
precision=self.dtype,
device=self.device,
jit=False,
force_quick_gelu=False,
force_custom_text=False,
force_patch_dropout=False,
force_image_size=None,
pretrained_image=False,
image_mean=None,
image_std=None,
light_augmentation=True,
aug_cfg={},
output_dict=True,
with_score_predictor=False,
with_region_predictor=False,
)
self.tokenizer = get_tokenizer("ViT-H-14")
# https://huggingface.co/laion/CLIP-ViT-H-14-laion2B-s32B-b79K/blob/main/preprocessor_config.json
# TODO: [transforms.Resize(224), transforms.CenterCrop(224)] for any aspect ratio.
self.transform = transforms.Compose([
transforms.Resize((224, 224), interpolation=transforms.InterpolationMode.BICUBIC),
transforms.Normalize(mean=[0.48145466, 0.4578275, 0.40821073], std=[0.26862954, 0.26130258, 0.27577711]),
])
if version == "v2.0":
url = "https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Third_Party/HPS_v2_compressed.pt"
filename = "HPS_v2_compressed.pt"
md5 = "fd9180de357abf01fdb4eaad64631db4"
elif version == "v2.1":
url = "https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Third_Party/HPS_v2.1_compressed.pt"
filename = "HPS_v2.1_compressed.pt"
md5 = "4067542e34ba2553a738c5ac6c1d75c0"
else:
raise ValueError("Only v2.0 and v2.1 are supported.")
if self.model_path is None or not os.path.exists(self.model_path):
download_url(url, torch.hub.get_dir(), md5=md5)
model_path = os.path.join(torch.hub.get_dir(), filename)
state_dict = torch.load(model_path, map_location="cpu")["state_dict"]
self.model.load_state_dict(state_dict)
self.model.to(device=self.device, dtype=self.dtype)
self.model.requires_grad_(False)
self.model.eval()
def __call__(self, batch_frames: torch.Tensor, batch_prompt: list[str]) -> Tuple[torch.Tensor, torch.Tensor]:
assert batch_frames.shape[0] == len(batch_prompt)
# Compute batch reward and loss in frame-wise.
batch_frames = rearrange(batch_frames, "b c t h w -> t b c h w")
batch_loss, batch_reward = 0, 0
for frames in batch_frames:
image_inputs = torch.stack([self.transform(frame) for frame in frames])
image_inputs = image_inputs.to(device=self.device, dtype=self.dtype)
text_inputs = self.tokenizer(batch_prompt).to(device=self.device)
outputs = self.model(image_inputs, text_inputs)
image_features, text_features = outputs["image_features"], outputs["text_features"]
logits = image_features @ text_features.T
reward = torch.diagonal(logits)
# Convert reward to loss in [0, 1].
if self.max_reward is None:
loss = (-1 * reward) * self.loss_scale
else:
loss = abs(reward - self.max_reward) * self.loss_scale
batch_loss, batch_reward = batch_loss + loss.mean(), batch_reward + reward.mean()
return batch_loss / batch_frames.shape[0], batch_reward / batch_frames.shape[0]
class PickScoreReward(BaseReward):
"""[PickScore](https://github.com/yuvalkirstain/PickScore) reward model.
"""
def __init__(
self,
model_path="yuvalkirstain/PickScore_v1",
device="cpu",
dtype=torch.float16,
max_reward=1,
loss_scale=1,
):
from transformers import AutoProcessor, AutoModel
self.model_path = model_path
self.device = device
self.dtype = dtype
self.max_reward = max_reward
self.loss_scale = loss_scale
# https://huggingface.co/yuvalkirstain/PickScore_v1/blob/main/preprocessor_config.json
self.transform = transforms.Compose([
transforms.Resize(224, interpolation=transforms.InterpolationMode.BICUBIC),
transforms.CenterCrop(224),
transforms.Normalize(mean=[0.48145466, 0.4578275, 0.40821073], std=[0.26862954, 0.26130258, 0.27577711]),
])
self.processor = AutoProcessor.from_pretrained("laion/CLIP-ViT-H-14-laion2B-s32B-b79K", torch_dtype=self.dtype)
self.model = AutoModel.from_pretrained(model_path, torch_dtype=self.dtype).eval().to(device)
self.model.requires_grad_(False)
self.model.eval()
def __call__(self, batch_frames: torch.Tensor, batch_prompt: list[str]) -> Tuple[torch.Tensor, torch.Tensor]:
assert batch_frames.shape[0] == len(batch_prompt)
# Compute batch reward and loss in frame-wise.
batch_frames = rearrange(batch_frames, "b c t h w -> t b c h w")
batch_loss, batch_reward = 0, 0
for frames in batch_frames:
image_inputs = torch.stack([self.transform(frame) for frame in frames])
image_inputs = image_inputs.to(device=self.device, dtype=self.dtype)
text_inputs = self.processor(
text=batch_prompt,
padding=True,
truncation=True,
max_length=77,
return_tensors="pt",
).to(self.device)
image_features = self.model.get_image_features(pixel_values=image_inputs)
text_features = self.model.get_text_features(**text_inputs)
image_features = image_features / torch.norm(image_features, dim=-1, keepdim=True)
text_features = text_features / torch.norm(text_features, dim=-1, keepdim=True)
logits = image_features @ text_features.T
reward = torch.diagonal(logits)
# Convert reward to loss in [0, 1].
if self.max_reward is None:
loss = (-1 * reward) * self.loss_scale
else:
loss = abs(reward - self.max_reward) * self.loss_scale
batch_loss, batch_reward = batch_loss + loss.mean(), batch_reward + reward.mean()
return batch_loss / batch_frames.shape[0], batch_reward / batch_frames.shape[0]
class MPSReward(BaseReward):
"""[MPS](https://github.com/Kwai-Kolors/MPS) reward model.
"""
def __init__(
self,
model_path=None,
device="cpu",
dtype=torch.float16,
max_reward=1,
loss_scale=1,
):
from transformers import AutoTokenizer, AutoConfig
from .MPS.trainer.models.clip_model import CLIPModel
self.model_path = model_path
self.device = device
self.dtype = dtype
self.condition = "light, color, clarity, tone, style, ambiance, artistry, shape, face, hair, hands, limbs, structure, instance, texture, quantity, attributes, position, number, location, word, things."
self.max_reward = max_reward
self.loss_scale = loss_scale
processor_name_or_path = "laion/CLIP-ViT-H-14-laion2B-s32B-b79K"
# https://huggingface.co/laion/CLIP-ViT-H-14-laion2B-s32B-b79K/blob/main/preprocessor_config.json
# TODO: [transforms.Resize(224), transforms.CenterCrop(224)] for any aspect ratio.
self.transform = transforms.Compose([
transforms.Resize((224, 224), interpolation=transforms.InterpolationMode.BICUBIC),
transforms.Normalize(mean=[0.48145466, 0.4578275, 0.40821073], std=[0.26862954, 0.26130258, 0.27577711]),
])
# We convert the original [ckpt](http://drive.google.com/file/d/17qrK_aJkVNM75ZEvMEePpLj6L867MLkN/view?usp=sharing)
# (contains the entire model) to a `state_dict`.
url = "https://pai-aigc-photog.oss-cn-hangzhou.aliyuncs.com/easyanimate/Third_Party/MPS_overall.pth"
filename = "MPS_overall.pth"
md5 = "1491cbbbd20565747fe07e7572e2ac56"
if self.model_path is None or not os.path.exists(self.model_path):
download_url(url, torch.hub.get_dir(), md5=md5)
model_path = os.path.join(torch.hub.get_dir(), filename)
self.tokenizer = AutoTokenizer.from_pretrained(processor_name_or_path, trust_remote_code=True)
config = AutoConfig.from_pretrained(processor_name_or_path)
self.model = CLIPModel(config)
state_dict = torch.load(model_path, map_location="cpu")
self.model.load_state_dict(state_dict, strict=False)
self.model.to(device=self.device, dtype=self.dtype)
self.model.requires_grad_(False)
self.model.eval()
def _tokenize(self, caption):
input_ids = self.tokenizer(
caption,
max_length=self.tokenizer.model_max_length,
padding="max_length",
truncation=True,
return_tensors="pt"
).input_ids
return input_ids
def __call__(
self,
batch_frames: torch.Tensor,
batch_prompt: list[str],
batch_condition: Optional[list[str]] = None
) -> Tuple[torch.Tensor, torch.Tensor]:
if batch_condition is None:
batch_condition = [self.condition] * len(batch_prompt)
batch_frames = rearrange(batch_frames, "b c t h w -> t b c h w")
batch_loss, batch_reward = 0, 0
for frames in batch_frames:
image_inputs = torch.stack([self.transform(frame) for frame in frames])
image_inputs = image_inputs.to(device=self.device, dtype=self.dtype)
text_inputs = self._tokenize(batch_prompt).to(self.device)
condition_inputs = self._tokenize(batch_condition).to(device=self.device)
text_features, image_features = self.model(text_inputs, image_inputs, condition_inputs)
text_features = text_features / text_features.norm(dim=-1, keepdim=True)
image_features = image_features / image_features.norm(dim=-1, keepdim=True)
# reward = self.model.logit_scale.exp() * torch.diag(torch.einsum('bd,cd->bc', text_features, image_features))
logits = image_features @ text_features.T
reward = torch.diagonal(logits)
# Convert reward to loss in [0, 1].
if self.max_reward is None:
loss = (-1 * reward) * self.loss_scale
else:
loss = abs(reward - self.max_reward) * self.loss_scale
batch_loss, batch_reward = batch_loss + loss.mean(), batch_reward + reward.mean()
return batch_loss / batch_frames.shape[0], batch_reward / batch_frames.shape[0]
if __name__ == "__main__":
import numpy as np
from decord import VideoReader
video_path_list = ["your_video_path_1.mp4", "your_video_path_2.mp4"]
prompt_list = ["your_prompt_1", "your_prompt_2"]
num_sampled_frames = 8
to_tensor = transforms.ToTensor()
sampled_frames_list = []
for video_path in video_path_list:
vr = VideoReader(video_path)
sampled_frame_indices = np.linspace(0, len(vr), num_sampled_frames, endpoint=False, dtype=int)
sampled_frames = vr.get_batch(sampled_frame_indices).asnumpy()
sampled_frames = torch.stack([to_tensor(frame) for frame in sampled_frames])
sampled_frames_list.append(sampled_frames)
sampled_frames = torch.stack(sampled_frames_list)
sampled_frames = rearrange(sampled_frames, "b t c h w -> b c t h w")
aesthetic_reward_v2 = AestheticReward(device="cuda", dtype=torch.bfloat16)
print(f"aesthetic_reward_v2: {aesthetic_reward_v2(sampled_frames)}")
aesthetic_reward_v2_5 = AestheticReward(
encoder_path="google/siglip-so400m-patch14-384", version="v2.5", device="cuda", dtype=torch.bfloat16
)
print(f"aesthetic_reward_v2_5: {aesthetic_reward_v2_5(sampled_frames)}")
hps_reward_v2 = HPSReward(device="cuda", dtype=torch.bfloat16)
print(f"hps_reward_v2: {hps_reward_v2(sampled_frames, prompt_list)}")
hps_reward_v2_1 = HPSReward(version="v2.1", device="cuda", dtype=torch.bfloat16)
print(f"hps_reward_v2_1: {hps_reward_v2_1(sampled_frames, prompt_list)}")
pick_score = PickScoreReward(device="cuda", dtype=torch.bfloat16)
print(f"pick_score_reward: {pick_score(sampled_frames, prompt_list)}")
mps_score = MPSReward(device="cuda", dtype=torch.bfloat16)
print(f"mps_reward: {mps_score(sampled_frames, prompt_list)}")
Executable → Regular
+198 -1660
View File
File diff suppressed because it is too large Load Diff
-46
View File
@@ -1,46 +0,0 @@
"""Modified from https://github.com/THUDM/CogVideo/blob/3710a612d8760f5cdb1741befeebb65b9e0f2fe0/sat/sgm/modules/diffusionmodules/sigma_sampling.py
"""
import torch
class DiscreteSampling:
def __init__(self, num_idx, uniform_sampling=False):
self.num_idx = num_idx
self.uniform_sampling = uniform_sampling
self.is_distributed = torch.distributed.is_available() and torch.distributed.is_initialized()
if self.is_distributed and self.uniform_sampling:
world_size = torch.distributed.get_world_size()
self.rank = torch.distributed.get_rank()
i = 1
while True:
if world_size % i != 0 or num_idx % (world_size // i) != 0:
i += 1
else:
self.group_num = world_size // i
break
assert self.group_num > 0
assert world_size % self.group_num == 0
# the number of rank in one group
self.group_width = world_size // self.group_num
self.sigma_interval = self.num_idx // self.group_num
print('rank=%d world_size=%d group_num=%d group_width=%d sigma_interval=%s' % (
self.rank, world_size, self.group_num,
self.group_width, self.sigma_interval))
def __call__(self, n_samples, generator=None, device=None):
if self.is_distributed and self.uniform_sampling:
group_index = self.rank // self.group_width
idx = torch.randint(
group_index * self.sigma_interval,
(group_index + 1) * self.sigma_interval,
(n_samples,),
generator=generator, device=device,
)
print('proc[%d] idx=%s' % (self.rank, idx))
else:
idx = torch.randint(
0, self.num_idx, (n_samples,),
generator=generator, device=device,
)
return idx
-35
View File
@@ -1,35 +0,0 @@
"""Modified from https://github.com/kijai/ComfyUI-MochiWrapper
"""
import torch
import torch.nn as nn
def autocast_model_forward(cls, origin_dtype, *inputs, **kwargs):
weight_dtype = cls.weight.dtype
cls.to(origin_dtype)
# Convert all inputs to the original dtype
inputs = [input.to(origin_dtype) for input in inputs]
out = cls.original_forward(*inputs, **kwargs)
cls.to(weight_dtype)
return out
def convert_model_weight_to_float8(model, exclude_module_name='embed_tokens'):
for name, module in model.named_modules():
if exclude_module_name not in name:
for param_name, param in module.named_parameters():
if exclude_module_name not in param_name:
param.data = param.data.to(torch.float8_e4m3fn)
def convert_weight_dtype_wrapper(module, origin_dtype):
for name, module in module.named_modules():
if name == "" or "embed_tokens" in name:
continue
original_forward = module.forward
if hasattr(module, "weight"):
setattr(module, "original_forward", original_forward)
setattr(
module,
"forward",
lambda *inputs, m=module, **kwargs: autocast_model_forward(m, origin_dtype, *inputs, **kwargs)
)
+47 -65
View File
@@ -156,8 +156,8 @@ def precalculate_safetensors_hashes(tensors, metadata):
class LoRANetwork(torch.nn.Module):
TRANSFORMER_TARGET_REPLACE_MODULE = ["Transformer2DModel", "Transformer3DModel", "HunyuanTransformer3DModel", "EasyAnimateTransformer3DModel"]
TEXT_ENCODER_TARGET_REPLACE_MODULE = ["T5LayerSelfAttention", "T5LayerFF", "BertEncoder"]
TRANSFORMER_TARGET_REPLACE_MODULE = ["Transformer2DModel", "Transformer3DModel"]
TEXT_ENCODER_TARGET_REPLACE_MODULE = ["T5LayerSelfAttention", "T5LayerFF"]
LORA_PREFIX_TRANSFORMER = "lora_unet"
LORA_PREFIX_TEXT_ENCODER = "lora_te"
def __init__(
@@ -238,10 +238,9 @@ class LoRANetwork(torch.nn.Module):
self.text_encoder_loras = []
skipped_te = []
for i, text_encoder in enumerate(text_encoders):
if text_encoder is not None:
text_encoder_loras, skipped = create_modules(False, text_encoder, LoRANetwork.TEXT_ENCODER_TARGET_REPLACE_MODULE)
self.text_encoder_loras.extend(text_encoder_loras)
skipped_te += skipped
text_encoder_loras, skipped = create_modules(False, text_encoder, LoRANetwork.TEXT_ENCODER_TARGET_REPLACE_MODULE)
self.text_encoder_loras.extend(text_encoder_loras)
skipped_te += skipped
print(f"create LoRA for Text Encoder: {len(self.text_encoder_loras)} modules.")
self.unet_loras, skipped_un = create_modules(True, unet, LoRANetwork.TRANSFORMER_TARGET_REPLACE_MODULE)
@@ -390,44 +389,36 @@ def merge_lora(pipeline, lora_path, multiplier, device='cpu', dtype=torch.float3
layer_infos = layer.split(LORA_PREFIX_TRANSFORMER + "_")[-1].split("_")
curr_layer = pipeline.transformer
try:
curr_layer = curr_layer.__getattr__("_".join(layer_infos[1:]))
except Exception:
temp_name = layer_infos.pop(0)
while len(layer_infos) > -1:
try:
curr_layer = curr_layer.__getattr__(temp_name)
if len(layer_infos) > 0:
temp_name = layer_infos.pop(0)
elif len(layer_infos) == 0:
break
except Exception:
if len(layer_infos) == 0:
print('Error loading layer')
if len(temp_name) > 0:
temp_name += "_" + layer_infos.pop(0)
else:
temp_name = layer_infos.pop(0)
temp_name = layer_infos.pop(0)
while len(layer_infos) > -1:
try:
curr_layer = curr_layer.__getattr__(temp_name)
if len(layer_infos) > 0:
temp_name = layer_infos.pop(0)
elif len(layer_infos) == 0:
break
except Exception:
if len(layer_infos) == 0:
print('Error loading layer')
if len(temp_name) > 0:
temp_name += "_" + layer_infos.pop(0)
else:
temp_name = layer_infos.pop(0)
origin_dtype = curr_layer.weight.data.dtype
origin_device = curr_layer.weight.data.device
curr_layer = curr_layer.to(device, dtype)
weight_up = elems['lora_up.weight'].to(device, dtype)
weight_down = elems['lora_down.weight'].to(device, dtype)
weight_up = elems['lora_up.weight'].to(dtype)
weight_down = elems['lora_down.weight'].to(dtype)
if 'alpha' in elems.keys():
alpha = elems['alpha'].item() / weight_up.shape[1]
else:
alpha = 1.0
curr_layer.weight.data = curr_layer.weight.data.to(device)
if len(weight_up.shape) == 4:
curr_layer.weight.data += multiplier * alpha * torch.mm(
weight_up.squeeze(3).squeeze(2), weight_down.squeeze(3).squeeze(2)
).unsqueeze(2).unsqueeze(3)
curr_layer.weight.data += multiplier * alpha * torch.mm(weight_up.squeeze(3).squeeze(2),
weight_down.squeeze(3).squeeze(2)).unsqueeze(
2).unsqueeze(3)
else:
curr_layer.weight.data += multiplier * alpha * torch.mm(weight_up, weight_down)
curr_layer = curr_layer.to(origin_device, origin_dtype)
return pipeline
@@ -452,43 +443,34 @@ def unmerge_lora(pipeline, lora_path, multiplier=1, device="cpu", dtype=torch.fl
layer_infos = layer.split(LORA_PREFIX_UNET + "_")[-1].split("_")
curr_layer = pipeline.transformer
try:
curr_layer = curr_layer.__getattr__("_".join(layer_infos[1:]))
except Exception:
temp_name = layer_infos.pop(0)
while len(layer_infos) > -1:
try:
curr_layer = curr_layer.__getattr__(temp_name)
if len(layer_infos) > 0:
temp_name = layer_infos.pop(0)
elif len(layer_infos) == 0:
break
except Exception:
if len(layer_infos) == 0:
print('Error loading layer')
if len(temp_name) > 0:
temp_name += "_" + layer_infos.pop(0)
else:
temp_name = layer_infos.pop(0)
temp_name = layer_infos.pop(0)
while len(layer_infos) > -1:
try:
curr_layer = curr_layer.__getattr__(temp_name)
if len(layer_infos) > 0:
temp_name = layer_infos.pop(0)
elif len(layer_infos) == 0:
break
except Exception:
if len(layer_infos) == 0:
print('Error loading layer')
if len(temp_name) > 0:
temp_name += "_" + layer_infos.pop(0)
else:
temp_name = layer_infos.pop(0)
origin_dtype = curr_layer.weight.data.dtype
origin_device = curr_layer.weight.data.device
curr_layer = curr_layer.to(device, dtype)
weight_up = elems['lora_up.weight'].to(device, dtype)
weight_down = elems['lora_down.weight'].to(device, dtype)
weight_up = elems['lora_up.weight'].to(dtype)
weight_down = elems['lora_down.weight'].to(dtype)
if 'alpha' in elems.keys():
alpha = elems['alpha'].item() / weight_up.shape[1]
else:
alpha = 1.0
curr_layer.weight.data = curr_layer.weight.data.to(device)
if len(weight_up.shape) == 4:
curr_layer.weight.data -= multiplier * alpha * torch.mm(
weight_up.squeeze(3).squeeze(2), weight_down.squeeze(3).squeeze(2)
).unsqueeze(2).unsqueeze(3)
curr_layer.weight.data -= multiplier * alpha * torch.mm(weight_up.squeeze(3).squeeze(2),
weight_down.squeeze(3).squeeze(2)).unsqueeze(2).unsqueeze(3)
else:
curr_layer.weight.data -= multiplier * alpha * torch.mm(weight_up, weight_down)
curr_layer = curr_layer.to(origin_device, origin_dtype)
return pipeline
return pipeline
+1 -172
View File
@@ -1,23 +1,14 @@
import gc
import os
import cv2
import imageio
import numpy as np
import torch
import torchvision
import cv2
from einops import rearrange
from PIL import Image
def get_width_and_height_from_image_and_base_resolution(image, base_resolution):
target_pixels = int(base_resolution) * int(base_resolution)
original_width, original_height = Image.open(image).size
ratio = (target_pixels / (original_width * original_height)) ** 0.5
width_slider = round(original_width * ratio)
height_slider = round(original_height * ratio)
return height_slider, width_slider
def color_transfer(sc, dc):
"""
Transfer color distribution from of sc, referred to dc.
@@ -71,165 +62,3 @@ def save_videos_grid(videos: torch.Tensor, path: str, rescale=False, n_rows=6, f
if path.endswith("mp4"):
path = path.replace('.mp4', '.gif')
outputs[0].save(path, format='GIF', append_images=outputs, save_all=True, duration=100, loop=0)
def get_image_to_video_latent(validation_image_start, validation_image_end, video_length, sample_size):
if validation_image_start is not None and validation_image_end is not None:
if type(validation_image_start) is str and os.path.isfile(validation_image_start):
image_start = clip_image = Image.open(validation_image_start).convert("RGB")
image_start = image_start.resize([sample_size[1], sample_size[0]])
clip_image = clip_image.resize([sample_size[1], sample_size[0]])
else:
image_start = clip_image = validation_image_start
image_start = [_image_start.resize([sample_size[1], sample_size[0]]) for _image_start in image_start]
clip_image = [_clip_image.resize([sample_size[1], sample_size[0]]) for _clip_image in clip_image]
if type(validation_image_end) is str and os.path.isfile(validation_image_end):
image_end = Image.open(validation_image_end).convert("RGB")
image_end = image_end.resize([sample_size[1], sample_size[0]])
else:
image_end = validation_image_end
image_end = [_image_end.resize([sample_size[1], sample_size[0]]) for _image_end in image_end]
if type(image_start) is list:
clip_image = clip_image[0]
start_video = torch.cat(
[torch.from_numpy(np.array(_image_start)).permute(2, 0, 1).unsqueeze(1).unsqueeze(0) for _image_start in image_start],
dim=2
)
input_video = torch.tile(start_video[:, :, :1], [1, 1, video_length, 1, 1])
input_video[:, :, :len(image_start)] = start_video
input_video_mask = torch.zeros_like(input_video[:, :1])
input_video_mask[:, :, len(image_start):] = 255
else:
input_video = torch.tile(
torch.from_numpy(np.array(image_start)).permute(2, 0, 1).unsqueeze(1).unsqueeze(0),
[1, 1, video_length, 1, 1]
)
input_video_mask = torch.zeros_like(input_video[:, :1])
input_video_mask[:, :, 1:] = 255
if type(image_end) is list:
image_end = [_image_end.resize(image_start[0].size if type(image_start) is list else image_start.size) for _image_end in image_end]
end_video = torch.cat(
[torch.from_numpy(np.array(_image_end)).permute(2, 0, 1).unsqueeze(1).unsqueeze(0) for _image_end in image_end],
dim=2
)
input_video[:, :, -len(end_video):] = end_video
input_video_mask[:, :, -len(image_end):] = 0
else:
image_end = image_end.resize(image_start[0].size if type(image_start) is list else image_start.size)
input_video[:, :, -1:] = torch.from_numpy(np.array(image_end)).permute(2, 0, 1).unsqueeze(1).unsqueeze(0)
input_video_mask[:, :, -1:] = 0
input_video = input_video / 255
elif validation_image_start is not None:
if type(validation_image_start) is str and os.path.isfile(validation_image_start):
image_start = clip_image = Image.open(validation_image_start).convert("RGB")
image_start = image_start.resize([sample_size[1], sample_size[0]])
clip_image = clip_image.resize([sample_size[1], sample_size[0]])
else:
image_start = clip_image = validation_image_start
image_start = [_image_start.resize([sample_size[1], sample_size[0]]) for _image_start in image_start]
clip_image = [_clip_image.resize([sample_size[1], sample_size[0]]) for _clip_image in clip_image]
image_end = None
if type(image_start) is list:
clip_image = clip_image[0]
start_video = torch.cat(
[torch.from_numpy(np.array(_image_start)).permute(2, 0, 1).unsqueeze(1).unsqueeze(0) for _image_start in image_start],
dim=2
)
input_video = torch.tile(start_video[:, :, :1], [1, 1, video_length, 1, 1])
input_video[:, :, :len(image_start)] = start_video
input_video = input_video / 255
input_video_mask = torch.zeros_like(input_video[:, :1])
input_video_mask[:, :, len(image_start):] = 255
else:
input_video = torch.tile(
torch.from_numpy(np.array(image_start)).permute(2, 0, 1).unsqueeze(1).unsqueeze(0),
[1, 1, video_length, 1, 1]
) / 255
input_video_mask = torch.zeros_like(input_video[:, :1])
input_video_mask[:, :, 1:, ] = 255
else:
image_start = None
image_end = None
input_video = torch.zeros([1, 3, video_length, sample_size[0], sample_size[1]])
input_video_mask = torch.ones([1, 1, video_length, sample_size[0], sample_size[1]]) * 255
clip_image = None
del image_start
del image_end
gc.collect()
return input_video, input_video_mask, clip_image
def get_video_to_video_latent(input_video_path, video_length, sample_size, fps=None, validation_video_mask=None, ref_image=None):
if input_video_path is not None:
if isinstance(input_video_path, str):
cap = cv2.VideoCapture(input_video_path)
input_video = []
original_fps = cap.get(cv2.CAP_PROP_FPS)
frame_skip = 1 if fps is None else int(original_fps // fps)
frame_count = 0
while True:
ret, frame = cap.read()
if not ret:
break
if frame_count % frame_skip == 0:
frame = cv2.resize(frame, (sample_size[1], sample_size[0]))
input_video.append(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))
frame_count += 1
cap.release()
else:
input_video = input_video_path
input_video = torch.from_numpy(np.array(input_video))[:video_length]
input_video = input_video.permute([3, 0, 1, 2]).unsqueeze(0) / 255
if validation_video_mask is not None:
validation_video_mask = Image.open(validation_video_mask).convert('L').resize((sample_size[1], sample_size[0]))
input_video_mask = np.where(np.array(validation_video_mask) < 240, 0, 255)
input_video_mask = torch.from_numpy(np.array(input_video_mask)).unsqueeze(0).unsqueeze(-1).permute([3, 0, 1, 2]).unsqueeze(0)
input_video_mask = torch.tile(input_video_mask, [1, 1, input_video.size()[2], 1, 1])
input_video_mask = input_video_mask.to(input_video.device, input_video.dtype)
else:
input_video_mask = torch.zeros_like(input_video[:, :1])
input_video_mask[:, :, :] = 255
else:
input_video, input_video_mask = None, None
if ref_image is not None:
if isinstance(ref_image, str):
ref_image = Image.open(ref_image).convert("RGB")
ref_image = ref_image.resize((sample_size[1], sample_size[0]))
ref_image = torch.from_numpy(np.array(ref_image))
ref_image = ref_image.unsqueeze(0).permute([3, 0, 1, 2]).unsqueeze(0) / 255
else:
ref_image = torch.from_numpy(np.array(ref_image))
ref_image = ref_image.unsqueeze(0).permute([3, 0, 1, 2]).unsqueeze(0) / 255
return input_video, input_video_mask, ref_image
def get_image_latent(ref_image=None, sample_size=None):
if ref_image is not None:
if isinstance(ref_image, str):
ref_image = Image.open(ref_image).convert("RGB")
ref_image = ref_image.resize((sample_size[1], sample_size[0]))
ref_image = torch.from_numpy(np.array(ref_image))
ref_image = ref_image.unsqueeze(0).permute([3, 0, 1, 2]).unsqueeze(0) / 255
else:
ref_image = torch.from_numpy(np.array(ref_image))
ref_image = ref_image.unsqueeze(0).permute([3, 0, 1, 2]).unsqueeze(0) / 255
return ref_image
@@ -1,64 +0,0 @@
model:
base_learning_rate: 1.0e-04
target: easyanimate.vae.ldm.models.cogvideox_casual3dcnn.AutoencoderKLMagvit_CogVideoX
params:
latent_channels: 16
temporal_compression_ratio: 4
monitor: train/rec_loss
ckpt_path: vae/diffusion_pytorch_model.safetensors
down_block_types: ("CogVideoXDownBlock3D", "CogVideoXDownBlock3D", "CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",)
up_block_types: ("CogVideoXUpBlock3D", "CogVideoXUpBlock3D", "CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",)
lossconfig:
target: easyanimate.vae.ldm.modules.losses.LPIPSWithDiscriminator
params:
disc_start: 50001
kl_weight: 1.0e-06
disc_weight: 0.5
l2_loss_weight: 0.1
l1_loss_weight: 1.0
perceptual_weight: 1.0
data:
target: train_vae.DataModuleFromConfig
params:
batch_size: 1
wrap: true
num_workers: 8
train:
target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRTrain
params:
data_json_path: pretrain.json
data_root: /your_data_root # This is used in relative path
size: 256
degradation: pil_nearest
video_size: 256
video_len: 49
slice_interval: 1
validation:
target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRValidation
params:
data_json_path: pretrain.json
data_root: /your_data_root # This is used in relative path
size: 256
degradation: pil_nearest
video_size: 256
video_len: 49
slice_interval: 1
lightning:
callbacks:
image_logger:
target: train_vae.ImageLogger
params:
batch_frequency: 5000
max_images: 8
increase_log_steps: True
trainer:
benchmark: True
accumulate_grad_batches: 1
gpus: "0"
num_nodes: 1
@@ -1,65 +0,0 @@
model:
base_learning_rate: 1.0e-04
target: easyanimate.vae.ldm.models.omnigen_casual3dcnn.AutoencoderKLMagvit_fromOmnigen
params:
spatial_group_norm: true
mid_block_attention_type: "spatial"
latent_channels: 16
monitor: train/rec_loss
ckpt_path: vae/diffusion_pytorch_model.safetensors
down_block_types: ("SpatialDownBlock3D", "SpatialTemporalDownBlock3D", "SpatialTemporalDownBlock3D",
"SpatialTemporalDownBlock3D",)
up_block_types: ("SpatialUpBlock3D", "SpatialTemporalUpBlock3D", "SpatialTemporalUpBlock3D",
"SpatialTemporalUpBlock3D",)
lossconfig:
target: easyanimate.vae.ldm.modules.losses.LPIPSWithDiscriminator
params:
disc_start: 50001
kl_weight: 1.0e-06
disc_weight: 0.5
l2_loss_weight: 0.1
l1_loss_weight: 1.0
perceptual_weight: 1.0
data:
target: train_vae.DataModuleFromConfig
params:
batch_size: 1
wrap: true
num_workers: 8
train:
target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRTrain
params:
data_json_path: pretrain.json
data_root: /your_data_root # This is used in relative path
size: 256
degradation: pil_nearest
video_size: 256
video_len: 49
slice_interval: 1
validation:
target: easyanimate.vae.ldm.data.dataset_image_video.CustomSRValidation
params:
data_json_path: pretrain.json
data_root: /your_data_root # This is used in relative path
size: 256
degradation: pil_nearest
video_size: 256
video_len: 49
slice_interval: 1
lightning:
callbacks:
image_logger:
target: train_vae.ImageLogger
params:
batch_frequency: 5000
max_images: 8
increase_log_steps: True
trainer:
benchmark: True
accumulate_grad_batches: 1
gpus: "0"
num_nodes: 1
@@ -1,7 +1,6 @@
#-*- encoding:utf-8 -*-
from pytorch_lightning.callbacks import Callback
class DatasetCallback(Callback):
def __init__(self):
self.sampler_pos_start = 0
@@ -17,7 +17,7 @@ from decord import VideoReader
from func_timeout import FunctionTimedOut, func_set_timeout
from omegaconf import OmegaConf
from PIL import Image
from torch.utils.data import BatchSampler, Dataset, Sampler
from torch.utils.data import (BatchSampler, Dataset, Sampler)
from tqdm import tqdm
from ..modules.image_degradation import (degradation_fn_bsr,
@@ -164,18 +164,15 @@ class ImageVideoDataset(Dataset):
return self.base[index].get('type', 'image')
def __getitem__(self, i):
@func_set_timeout(15) # time wait 3 seconds
@func_set_timeout(3) # time wait 3 seconds
def get_video_item(example):
if self.data_root is not None:
video_reader = VideoReader(os.path.join(self.data_root, example['file_path']))
else:
video_reader = VideoReader(example['file_path'])
video_length = len(video_reader)
if self.slice_interval == "rand":
slice_interval = np.random.choice([1, 2, 3, 4, 5, 6, 7, 8])
else:
slice_interval = int(self.slice_interval)
clip_length = min(video_length, (self.video_len - 1) * slice_interval + 1)
clip_length = min(video_length, (self.video_len - 1) * self.slice_interval + 1)
start_idx = random.randint(0, video_length - clip_length)
batch_index = np.linspace(start_idx, start_idx + clip_length - 1, self.video_len, dtype=int)
+4 -4
View File
@@ -126,13 +126,13 @@ class AutoencoderKLMagvit(pl.LightningModule):
def configure_optimizers(self):
lr = self.learning_rate
opt_ae = torch.optim.AdamW(list(self.encoder.parameters())+
opt_ae = torch.optim.Adam(list(self.encoder.parameters())+
list(self.decoder.parameters())+
list(self.quant_conv.parameters())+
list(self.post_quant_conv.parameters()),
lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
opt_disc = torch.optim.AdamW(self.loss.discriminator.parameters(),
lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
lr=lr, betas=(0.5, 0.9))
opt_disc = torch.optim.Adam(self.loss.discriminator.parameters(),
lr=lr, betas=(0.5, 0.9))
return [opt_ae, opt_disc], []
def get_last_layer(self):
-337
View File
@@ -1,337 +0,0 @@
import time
from contextlib import contextmanager
import pytorch_lightning as pl
import torch
import torch.nn.functional as F
from ..modules.diffusionmodules.model import Decoder, Encoder
from ..modules.distributions.distributions import DiagonalGaussianDistribution
from ..util import instantiate_from_config
from .enc_dec import Decoder as Mag_Decoder
from .enc_dec import Encoder as Mag_Encoder
class AutoencoderKLMagvit(pl.LightningModule):
def __init__(self,
ddconfig,
lossconfig,
embed_dim,
ckpt_path=None,
ignore_keys=[],
image_key="image",
colorize_nlabels=None,
monitor=None,
):
super().__init__()
self.image_key = image_key
self.encoder = Mag_Encoder()
self.decoder = Mag_Decoder()
self.loss = instantiate_from_config(lossconfig)
self.quant_conv = torch.nn.Conv3d(16, 16, 1)
self.post_quant_conv = torch.nn.Conv3d(8, 8, 1)
self.embed_dim = embed_dim
if colorize_nlabels is not None:
assert type(colorize_nlabels)==int
self.register_buffer("colorize", torch.randn(3, colorize_nlabels, 1, 1))
if monitor is not None:
self.monitor = monitor
if ckpt_path is not None:
self.init_from_ckpt(ckpt_path, ignore_keys=ignore_keys)
def init_from_ckpt(self, path, ignore_keys=list()):
sd = torch.load(path, map_location="cpu")["state_dict"]
keys = list(sd.keys())
for k in keys:
for ik in ignore_keys:
if k.startswith(ik):
print("Deleting key {} from state_dict.".format(k))
del sd[k]
self.load_state_dict(sd, strict=False)
print(f"Restored from {path}")
def encode(self, x):
h = self.encoder(x)
moments = self.quant_conv(h)
posterior = DiagonalGaussianDistribution(moments)
return posterior
def decode(self, z):
z = self.post_quant_conv(z)
dec = self.decoder(z)
return dec
def forward(self, input, sample_posterior=True):
if input.ndim==4:
input = input.unsqueeze(2)
posterior = self.encode(input)
if sample_posterior:
z = posterior.sample()
else:
z = posterior.mode()
dec = self.decode(z)
return dec, posterior
def get_input(self, batch, k):
x = batch[k]
if x.ndim==5:
x = x.permute(0, 4, 1, 2, 3).to(memory_format=torch.contiguous_format).float()
return x
if len(x.shape) == 3:
x = x[..., None]
x = x.permute(0, 3, 1, 2).to(memory_format=torch.contiguous_format).float()
return x
def training_step(self, batch, batch_idx, optimizer_idx):
# tic = time.time()
inputs = self.get_input(batch, self.image_key)
# print(f"get_input time {time.time() - tic}")
# tic = time.time()
reconstructions, posterior = self(inputs)
# print(f"model forward time {time.time() - tic}")
if optimizer_idx == 0:
# train encoder+decoder+logvar
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("aeloss", aeloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_ae, prog_bar=False, logger=True, on_step=True, on_epoch=False)
# print(f"cal loss time {time.time() - tic}")
return aeloss
if optimizer_idx == 1:
# train the discriminator
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("discloss", discloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_disc, prog_bar=False, logger=True, on_step=True, on_epoch=False)
# print(f"cal loss time {time.time() - tic}")
return discloss
def validation_step(self, batch, batch_idx):
with torch.no_grad():
inputs = self.get_input(batch, self.image_key)
reconstructions, posterior = self(inputs)
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, 0, self.global_step,
last_layer=self.get_last_layer(), split="val")
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, 1, self.global_step,
last_layer=self.get_last_layer(), split="val")
self.log("val/rec_loss", log_dict_ae["val/rec_loss"])
self.log_dict(log_dict_ae)
self.log_dict(log_dict_disc)
return self.log_dict
def configure_optimizers(self):
lr = self.learning_rate
opt_ae = torch.optim.Adam(list(self.encoder.parameters())+
list(self.decoder.parameters())+
list(self.quant_conv.parameters())+
list(self.post_quant_conv.parameters()),
lr=lr, betas=(0.5, 0.9))
opt_disc = torch.optim.Adam(self.loss.discriminator.parameters(),
lr=lr, betas=(0.5, 0.9))
return [opt_ae, opt_disc], []
def get_last_layer(self):
return self.decoder.conv_out.weight
@torch.no_grad()
def log_images(self, batch, only_inputs=False, **kwargs):
log = dict()
x = self.get_input(batch, self.image_key)
x = x.to(self.device)
if not only_inputs:
xrec, posterior = self(x)
if x.shape[1] > 3:
# colorize with random projection
assert xrec.shape[1] > 3
x = self.to_rgb(x)
xrec = self.to_rgb(xrec)
log["samples"] = self.decode(torch.randn_like(posterior.sample()))
log["reconstructions"] = xrec
log["inputs"] = x
return log
def to_rgb(self, x):
assert self.image_key == "segmentation"
if not hasattr(self, "colorize"):
self.register_buffer("colorize", torch.randn(3, x.shape[1], 1, 1).to(x))
x = F.conv2d(x, weight=self.colorize)
x = 2.*(x-x.min())/(x.max()-x.min()) - 1.
return x
class AutoencoderKL(pl.LightningModule):
def __init__(self,
ddconfig,
lossconfig,
embed_dim,
ckpt_path=None,
ignore_keys=[],
image_key="image",
colorize_nlabels=None,
monitor=None,
):
super().__init__()
self.image_key = image_key
self.encoder = Encoder(**ddconfig)
self.decoder = Decoder(**ddconfig)
self.loss = instantiate_from_config(lossconfig)
assert ddconfig["double_z"]
self.quant_conv = torch.nn.Conv2d(2*ddconfig["z_channels"], 2*embed_dim, 1)
self.post_quant_conv = torch.nn.Conv2d(embed_dim, ddconfig["z_channels"], 1)
self.embed_dim = embed_dim
if colorize_nlabels is not None:
assert type(colorize_nlabels)==int
self.register_buffer("colorize", torch.randn(3, colorize_nlabels, 1, 1))
if monitor is not None:
self.monitor = monitor
if ckpt_path is not None:
self.init_from_ckpt(ckpt_path, ignore_keys=ignore_keys)
def init_from_ckpt(self, path, ignore_keys=list()):
sd = torch.load(path, map_location="cpu")["state_dict"]
keys = list(sd.keys())
for k in keys:
for ik in ignore_keys:
if k.startswith(ik):
print("Deleting key {} from state_dict.".format(k))
del sd[k]
self.load_state_dict(sd, strict=False)
print(f"Restored from {path}")
def encode(self, x):
h = self.encoder(x)
moments = self.quant_conv(h)
posterior = DiagonalGaussianDistribution(moments)
return posterior
def decode(self, z):
z = self.post_quant_conv(z)
dec = self.decoder(z)
return dec
def forward(self, input, sample_posterior=True):
posterior = self.encode(input)
if sample_posterior:
z = posterior.sample()
else:
z = posterior.mode()
dec = self.decode(z)
return dec, posterior
def get_input(self, batch, k):
x = batch[k]
if len(x.shape) == 3:
x = x[..., None]
x = x.permute(0, 3, 1, 2).to(memory_format=torch.contiguous_format).float()
return x
def training_step(self, batch, batch_idx, optimizer_idx):
# tic = time.time()
inputs = self.get_input(batch, self.image_key)
# print(f"get_input time {time.time() - tic}")
# tic = time.time()
reconstructions, posterior = self(inputs)
# print(f"model forward time {time.time() - tic}")
tic = time.time()
if optimizer_idx == 0:
# train encoder+decoder+logvar
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("aeloss", aeloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_ae, prog_bar=False, logger=True, on_step=True, on_epoch=False)
# print(f"cal loss time {time.time() - tic}")
return aeloss
if optimizer_idx == 1:
# train the discriminator
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("discloss", discloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_disc, prog_bar=False, logger=True, on_step=True, on_epoch=False)
# print(f"cal loss time {time.time() - tic}")
return discloss
def validation_step(self, batch, batch_idx):
tic = time.time()
inputs = self.get_input(batch, self.image_key)
print(f"get_input time {time.time() - tic}")
tic = time.time()
reconstructions, posterior = self(inputs)
print(f"val forward time {time.time() - tic}")
tic = time.time()
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, 0, self.global_step,
last_layer=self.get_last_layer(), split="val")
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, 1, self.global_step,
last_layer=self.get_last_layer(), split="val")
self.log("val/rec_loss", log_dict_ae["val/rec_loss"])
self.log_dict(log_dict_ae)
self.log_dict(log_dict_disc)
print(f"val end time {time.time() - tic}")
return self.log_dict
def configure_optimizers(self):
lr = self.learning_rate
opt_ae = torch.optim.AdamW(list(self.encoder.parameters())+
list(self.decoder.parameters())+
list(self.quant_conv.parameters())+
list(self.post_quant_conv.parameters()), \
lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
opt_disc = torch.optim.AdamW(self.loss.discriminator.parameters(),
lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
return [opt_ae, opt_disc], []
def get_last_layer(self):
return self.decoder.conv_out.weight
@torch.no_grad()
def log_images(self, batch, only_inputs=False, **kwargs):
log = dict()
x = self.get_input(batch, self.image_key)
x = x.to(self.device)
if not only_inputs:
xrec, posterior = self(x)
if x.shape[1] > 3:
# colorize with random projection
assert xrec.shape[1] > 3
x = self.to_rgb(x)
xrec = self.to_rgb(xrec)
log["samples"] = self.decode(torch.randn_like(posterior.sample()))
log["reconstructions"] = xrec
log["inputs"] = x
return log
def to_rgb(self, x):
assert self.image_key == "segmentation"
if not hasattr(self, "colorize"):
self.register_buffer("colorize", torch.randn(3, x.shape[1], 1, 1).to(x))
x = F.conv2d(x, weight=self.colorize)
x = 2.*(x-x.min())/(x.max()-x.min()) - 1.
return x
class IdentityFirstStage(torch.nn.Module):
def __init__(self, *args, vq_interface=False, **kwargs):
self.vq_interface = vq_interface # TODO: Should be true by default but check to not break older stuff
super().__init__()
def encode(self, x, *args, **kwargs):
return x
def decode(self, x, *args, **kwargs):
return x
def quantize(self, x, *args, **kwargs):
if self.vq_interface:
return x, None, [None, None, None]
return x
def forward(self, x, *args, **kwargs):
return x
@@ -1,326 +0,0 @@
from dataclasses import dataclass
from typing import Dict, Optional, Tuple
import numpy as np
import pytorch_lightning as pl
import torch
import torch.nn as nn
import torch.nn.functional as F
from einops import rearrange
from ..util import instantiate_from_config
from .cogvideox_enc_dec import (CogVideoXDecoder3D, CogVideoXEncoder3D,
CogVideoXSafeConv3d)
class DiagonalGaussianDistribution:
def __init__(
self,
mean: torch.Tensor,
logvar: torch.Tensor,
deterministic: bool = False,
):
self.mean = mean
self.logvar = torch.clamp(logvar, -30.0, 20.0)
self.deterministic = deterministic
if deterministic:
self.var = self.std = torch.zeros_like(self.mean)
else:
self.std = torch.exp(0.5 * self.logvar)
self.var = torch.exp(self.logvar)
def sample(self, generator = None) -> torch.FloatTensor:
x = torch.randn(
self.mean.shape,
generator=generator,
device=self.mean.device,
dtype=self.mean.dtype,
)
return self.mean + self.std * x
def mode(self):
return self.mean
def kl(self, other: Optional["DiagonalGaussianDistribution"] = None) -> torch.Tensor:
dims = list(range(1, self.mean.ndim))
if self.deterministic:
return torch.Tensor([0.0])
else:
if other is None:
return 0.5 * torch.sum(
torch.pow(self.mean, 2) + self.var - 1.0 - self.logvar,
dim=dims,
)
else:
return 0.5 * torch.sum(
torch.pow(self.mean - other.mean, 2) / other.var
+ self.var / other.var
- 1.0
- self.logvar
+ other.logvar,
dim=dims,
)
def nll(self, sample: torch.Tensor) -> torch.Tensor:
dims = list(range(1, self.mean.ndim))
if self.deterministic:
return torch.Tensor([0.0])
logtwopi = np.log(2.0 * np.pi)
return 0.5 * torch.sum(
logtwopi + self.logvar + torch.pow(sample - self.mean, 2) / self.var,
dim=dims,
)
@dataclass
class EncoderOutput:
latent_dist: DiagonalGaussianDistribution
@dataclass
class DecoderOutput:
sample: torch.Tensor
def str_eval(item):
if type(item) == str:
return eval(item)
else:
return item
class AutoencoderKLMagvit_CogVideoX(pl.LightningModule):
def __init__(
self,
in_channels: int = 3,
out_channels: int = 3,
down_block_types: Tuple[str] = (
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
),
up_block_types: Tuple[str] = (
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
),
block_out_channels: Tuple[int] = (128, 256, 256, 512),
latent_channels: int = 16,
layers_per_block: int = 3,
act_fn: str = "silu",
norm_eps: float = 1e-6,
norm_num_groups: int = 32,
temporal_compression_ratio: float = 4,
use_quant_conv: bool = False,
use_post_quant_conv: bool = False,
mini_batch_encoder=4,
mini_batch_decoder=1,
image_key="image",
train_decoder_only=False,
train_encoder_only=False,
monitor=None,
ckpt_path=None,
lossconfig=None,
):
super().__init__()
self.image_key = image_key
down_block_types = str_eval(down_block_types)
up_block_types = str_eval(up_block_types)
self.encoder = CogVideoXEncoder3D(
in_channels=in_channels,
out_channels=latent_channels,
down_block_types=down_block_types,
block_out_channels=block_out_channels,
layers_per_block=layers_per_block,
act_fn=act_fn,
norm_eps=norm_eps,
norm_num_groups=norm_num_groups,
temporal_compression_ratio=temporal_compression_ratio,
)
self.decoder = CogVideoXDecoder3D(
in_channels=latent_channels,
out_channels=out_channels,
up_block_types=up_block_types,
block_out_channels=block_out_channels,
layers_per_block=layers_per_block,
act_fn=act_fn,
norm_eps=norm_eps,
norm_num_groups=norm_num_groups,
temporal_compression_ratio=temporal_compression_ratio,
)
self.quant_conv = CogVideoXSafeConv3d(2 * out_channels, 2 * out_channels, 1) if use_quant_conv else None
self.post_quant_conv = CogVideoXSafeConv3d(out_channels, out_channels, 1) if use_post_quant_conv else None
self.mini_batch_encoder = mini_batch_encoder
self.mini_batch_decoder = mini_batch_decoder
self.train_decoder_only = train_decoder_only
self.train_encoder_only = train_encoder_only
if train_decoder_only:
self.encoder.requires_grad_(False)
if self.quant_conv is not None:
self.quant_conv.requires_grad_(False)
if train_encoder_only:
self.decoder.requires_grad_(False)
if self.post_quant_conv is not None:
self.post_quant_conv.requires_grad_(False)
if monitor is not None:
self.monitor = monitor
if ckpt_path is not None:
self.init_from_ckpt(ckpt_path, ignore_keys="loss")
if lossconfig is not None:
self.loss = instantiate_from_config(lossconfig)
def init_from_ckpt(self, path, ignore_keys=list()):
if path.endswith("safetensors"):
from safetensors.torch import load_file, safe_open
sd = load_file(path)
else:
sd = torch.load(path, map_location="cpu")
if "state_dict" in list(sd.keys()):
sd = sd["state_dict"]
keys = list(sd.keys())
for k in keys:
for ik in ignore_keys:
if k.startswith(ik):
print("Deleting key {} from state_dict.".format(k))
del sd[k]
m, u = self.load_state_dict(sd, strict=False) # loss.item can be ignored successfully
print(f"Restored from {path}")
print(f"missing keys: {str(m)}, unexpected keys: {str(u)}")
def encode(self, x: torch.Tensor) -> EncoderOutput:
h = self.encoder(x)
self.encoder._clear_fake_context_parallel_cache()
if self.quant_conv is not None:
moments: torch.Tensor = self.quant_conv(h)
else:
moments: torch.Tensor = h
mean, logvar = moments.chunk(2, dim=1)
posterior = DiagonalGaussianDistribution(mean, logvar)
return posterior
def decode(self, z: torch.Tensor) -> DecoderOutput:
if self.post_quant_conv is not None:
z = self.post_quant_conv(z)
decoded = self.decoder(z)
self.decoder._clear_fake_context_parallel_cache()
return decoded
def forward(self, input, sample_posterior=True):
if input.ndim==4:
input = input.unsqueeze(2)
posterior = self.encode(input)
if sample_posterior:
z = posterior.sample()
else:
z = posterior.mode()
# print("stt latent shape", z.shape)
dec = self.decode(z)
return dec, posterior
def get_input(self, batch, k):
x = batch[k]
if x.ndim==5:
x = x.permute(0, 4, 1, 2, 3).to(memory_format=torch.contiguous_format).float()
return x
if len(x.shape) == 3:
x = x[..., None]
x = x.permute(0, 3, 1, 2).to(memory_format=torch.contiguous_format).float()
return x
def training_step(self, batch, batch_idx, optimizer_idx):
inputs = self.get_input(batch, self.image_key)
reconstructions, posterior = self(inputs)
if optimizer_idx == 0:
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("aeloss", aeloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_ae, prog_bar=False, logger=True, on_step=True, on_epoch=False)
return aeloss
if optimizer_idx == 1:
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("discloss", discloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_disc, prog_bar=False, logger=True, on_step=True, on_epoch=False)
return discloss
def validation_step(self, batch, batch_idx):
with torch.no_grad():
inputs = self.get_input(batch, self.image_key)
reconstructions, posterior = self(inputs)
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, 0, self.global_step,
last_layer=self.get_last_layer(), split="val")
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, 1, self.global_step,
last_layer=self.get_last_layer(), split="val")
self.log("val/rec_loss", log_dict_ae["val/rec_loss"])
self.log_dict(log_dict_ae)
self.log_dict(log_dict_disc)
return self.log_dict
def configure_optimizers(self):
lr = self.learning_rate
if self.train_decoder_only:
if self.post_quant_conv is not None:
training_list = list(self.decoder.parameters()) + list(self.post_quant_conv.parameters())
else:
training_list = list(self.decoder.parameters())
opt_ae = torch.optim.AdamW(training_list, lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
elif self.train_encoder_only:
if self.quant_conv is not None:
training_list = list(self.encoder.parameters()) + list(self.quant_conv.parameters())
else:
training_list = list(self.encoder.parameters())
opt_ae = torch.optim.AdamW(training_list, lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
else:
training_list = list(self.encoder.parameters()) + list(self.decoder.parameters())
if self.quant_conv is not None:
training_list = training_list + list(self.quant_conv.parameters())
if self.post_quant_conv is not None:
training_list = training_list + list(self.post_quant_conv.parameters())
opt_ae = torch.optim.AdamW(training_list, lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
opt_disc = torch.optim.AdamW(
list(self.loss.discriminator3d.parameters()) + list(self.loss.discriminator.parameters()),
lr=lr, betas=(0.9, 0.999), weight_decay=5e-2
)
return [opt_ae, opt_disc], []
def get_last_layer(self):
return self.decoder.conv_out.conv.weight
@torch.no_grad()
def log_images(self, batch, only_inputs=False, **kwargs):
log = dict()
x = self.get_input(batch, self.image_key)
x = x.to(self.device)
if not only_inputs:
xrec, posterior = self(x)
if x.shape[1] > 3:
# colorize with random projection
assert xrec.shape[1] > 3
x = self.to_rgb(x)
xrec = self.to_rgb(xrec)
log["samples"] = self.decode(torch.randn_like(posterior.sample()))
log["reconstructions"] = xrec
log["inputs"] = x
return log
def to_rgb(self, x):
assert self.image_key == "segmentation"
if not hasattr(self, "colorize"):
self.register_buffer("colorize", torch.randn(3, x.shape[1], 1, 1).to(x))
x = F.conv2d(x, weight=self.colorize)
x = 2.*(x-x.min())/(x.max()-x.min()) - 1.
return x
@@ -1,312 +0,0 @@
# Copyright 2024 The CogVideoX team, Tsinghua University & ZhipuAI and The HuggingFace Team.
# All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
from typing import Optional, Tuple
import numpy as np
import torch
import torch.nn as nn
from diffusers.models.autoencoders.autoencoder_kl_cogvideox import (
CogVideoXCausalConv3d, CogVideoXDownBlock3D, CogVideoXMidBlock3D,
CogVideoXSafeConv3d, CogVideoXSpatialNorm3D, CogVideoXUpBlock3D)
from diffusers.utils import logging
logger = logging.get_logger(__name__) # pylint: disable=invalid-name
class CogVideoXEncoder3D(nn.Module):
r"""
The `CogVideoXEncoder3D` layer of a variational autoencoder that encodes its input into a latent representation.
Args:
in_channels (`int`, *optional*, defaults to 3):
The number of input channels.
out_channels (`int`, *optional*, defaults to 3):
The number of output channels.
down_block_types (`Tuple[str, ...]`, *optional*, defaults to `("DownEncoderBlock2D",)`):
The types of down blocks to use. See `~diffusers.models.unet_2d_blocks.get_down_block` for available
options.
block_out_channels (`Tuple[int, ...]`, *optional*, defaults to `(64,)`):
The number of output channels for each block.
act_fn (`str`, *optional*, defaults to `"silu"`):
The activation function to use. See `~diffusers.models.activations.get_activation` for available options.
layers_per_block (`int`, *optional*, defaults to 2):
The number of layers per block.
norm_num_groups (`int`, *optional*, defaults to 32):
The number of groups for normalization.
"""
_supports_gradient_checkpointing = True
def __init__(
self,
in_channels: int = 3,
out_channels: int = 16,
down_block_types: Tuple[str, ...] = (
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
"CogVideoXDownBlock3D",
),
block_out_channels: Tuple[int, ...] = (128, 256, 256, 512),
layers_per_block: int = 3,
act_fn: str = "silu",
norm_eps: float = 1e-6,
norm_num_groups: int = 32,
dropout: float = 0.0,
pad_mode: str = "first",
temporal_compression_ratio: float = 4,
):
super().__init__()
# log2 of temporal_compress_times
temporal_compress_level = int(np.log2(temporal_compression_ratio))
self.conv_in = CogVideoXCausalConv3d(in_channels, block_out_channels[0], kernel_size=3, pad_mode=pad_mode)
self.down_blocks = nn.ModuleList([])
# down blocks
output_channel = block_out_channels[0]
for i, down_block_type in enumerate(down_block_types):
input_channel = output_channel
output_channel = block_out_channels[i]
is_final_block = i == len(block_out_channels) - 1
compress_time = i < temporal_compress_level
if down_block_type == "CogVideoXDownBlock3D":
down_block = CogVideoXDownBlock3D(
in_channels=input_channel,
out_channels=output_channel,
temb_channels=0,
dropout=dropout,
num_layers=layers_per_block,
resnet_eps=norm_eps,
resnet_act_fn=act_fn,
resnet_groups=norm_num_groups,
add_downsample=not is_final_block,
compress_time=compress_time,
)
else:
raise ValueError("Invalid `down_block_type` encountered. Must be `CogVideoXDownBlock3D`")
self.down_blocks.append(down_block)
# mid block
self.mid_block = CogVideoXMidBlock3D(
in_channels=block_out_channels[-1],
temb_channels=0,
dropout=dropout,
num_layers=2,
resnet_eps=norm_eps,
resnet_act_fn=act_fn,
resnet_groups=norm_num_groups,
pad_mode=pad_mode,
)
self.norm_out = nn.GroupNorm(norm_num_groups, block_out_channels[-1], eps=1e-6)
self.conv_act = nn.SiLU()
self.conv_out = CogVideoXCausalConv3d(
block_out_channels[-1], 2 * out_channels, kernel_size=3, pad_mode=pad_mode
)
self.gradient_checkpointing = False
def _clear_fake_context_parallel_cache(self):
for name, module in self.named_modules():
if isinstance(module, CogVideoXCausalConv3d):
logger.debug(f"Clearing fake Context Parallel cache for layer: {name}")
module._clear_fake_context_parallel_cache()
def forward(self, sample: torch.Tensor, temb: Optional[torch.Tensor] = None) -> torch.Tensor:
r"""The forward method of the `CogVideoXEncoder3D` class."""
hidden_states = self.conv_in(sample)
if self.training and self.gradient_checkpointing:
def create_custom_forward(module):
def custom_forward(*inputs):
return module(*inputs)
return custom_forward
# 1. Down
for down_block in self.down_blocks:
hidden_states = torch.utils.checkpoint.checkpoint(
create_custom_forward(down_block), hidden_states, temb, None
)
# 2. Mid
hidden_states = torch.utils.checkpoint.checkpoint(
create_custom_forward(self.mid_block), hidden_states, temb, None
)
else:
# 1. Down
for down_block in self.down_blocks:
hidden_states = down_block(hidden_states, temb, None)
# 2. Mid
hidden_states = self.mid_block(hidden_states, temb, None)
# 3. Post-process
hidden_states = self.norm_out(hidden_states)
hidden_states = self.conv_act(hidden_states)
hidden_states = self.conv_out(hidden_states)
return hidden_states
class CogVideoXDecoder3D(nn.Module):
r"""
The `CogVideoXDecoder3D` layer of a variational autoencoder that decodes its latent representation into an output
sample.
Args:
in_channels (`int`, *optional*, defaults to 3):
The number of input channels.
out_channels (`int`, *optional*, defaults to 3):
The number of output channels.
up_block_types (`Tuple[str, ...]`, *optional*, defaults to `("UpDecoderBlock2D",)`):
The types of up blocks to use. See `~diffusers.models.unet_2d_blocks.get_up_block` for available options.
block_out_channels (`Tuple[int, ...]`, *optional*, defaults to `(64,)`):
The number of output channels for each block.
act_fn (`str`, *optional*, defaults to `"silu"`):
The activation function to use. See `~diffusers.models.activations.get_activation` for available options.
layers_per_block (`int`, *optional*, defaults to 2):
The number of layers per block.
norm_num_groups (`int`, *optional*, defaults to 32):
The number of groups for normalization.
"""
_supports_gradient_checkpointing = True
def __init__(
self,
in_channels: int = 16,
out_channels: int = 3,
up_block_types: Tuple[str, ...] = (
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
"CogVideoXUpBlock3D",
),
block_out_channels: Tuple[int, ...] = (128, 256, 256, 512),
layers_per_block: int = 3,
act_fn: str = "silu",
norm_eps: float = 1e-6,
norm_num_groups: int = 32,
dropout: float = 0.0,
pad_mode: str = "first",
temporal_compression_ratio: float = 4,
):
super().__init__()
reversed_block_out_channels = list(reversed(block_out_channels))
self.conv_in = CogVideoXCausalConv3d(
in_channels, reversed_block_out_channels[0], kernel_size=3, pad_mode=pad_mode
)
# mid block
self.mid_block = CogVideoXMidBlock3D(
in_channels=reversed_block_out_channels[0],
temb_channels=0,
num_layers=2,
resnet_eps=norm_eps,
resnet_act_fn=act_fn,
resnet_groups=norm_num_groups,
spatial_norm_dim=in_channels,
pad_mode=pad_mode,
)
# up blocks
self.up_blocks = nn.ModuleList([])
output_channel = reversed_block_out_channels[0]
temporal_compress_level = int(np.log2(temporal_compression_ratio))
for i, up_block_type in enumerate(up_block_types):
prev_output_channel = output_channel
output_channel = reversed_block_out_channels[i]
is_final_block = i == len(block_out_channels) - 1
compress_time = i < temporal_compress_level
if up_block_type == "CogVideoXUpBlock3D":
up_block = CogVideoXUpBlock3D(
in_channels=prev_output_channel,
out_channels=output_channel,
temb_channels=0,
dropout=dropout,
num_layers=layers_per_block + 1,
resnet_eps=norm_eps,
resnet_act_fn=act_fn,
resnet_groups=norm_num_groups,
spatial_norm_dim=in_channels,
add_upsample=not is_final_block,
compress_time=compress_time,
pad_mode=pad_mode,
)
prev_output_channel = output_channel
else:
raise ValueError("Invalid `up_block_type` encountered. Must be `CogVideoXUpBlock3D`")
self.up_blocks.append(up_block)
self.norm_out = CogVideoXSpatialNorm3D(reversed_block_out_channels[-1], in_channels, groups=norm_num_groups)
self.conv_act = nn.SiLU()
self.conv_out = CogVideoXCausalConv3d(
reversed_block_out_channels[-1], out_channels, kernel_size=3, pad_mode=pad_mode
)
self.gradient_checkpointing = False
def _clear_fake_context_parallel_cache(self):
for name, module in self.named_modules():
if isinstance(module, CogVideoXCausalConv3d):
logger.debug(f"Clearing fake Context Parallel cache for layer: {name}")
module._clear_fake_context_parallel_cache()
def forward(self, sample: torch.Tensor, temb: Optional[torch.Tensor] = None) -> torch.Tensor:
r"""The forward method of the `CogVideoXDecoder3D` class."""
hidden_states = self.conv_in(sample)
if self.training and self.gradient_checkpointing:
def create_custom_forward(module):
def custom_forward(*inputs):
return module(*inputs)
return custom_forward
# 1. Mid
hidden_states = torch.utils.checkpoint.checkpoint(
create_custom_forward(self.mid_block), hidden_states, temb, sample
)
# 2. Up
for up_block in self.up_blocks:
hidden_states = torch.utils.checkpoint.checkpoint(
create_custom_forward(up_block), hidden_states, temb, sample
)
else:
# 1. Mid
hidden_states = self.mid_block(hidden_states, temb, sample)
# 2. Up
for up_block in self.up_blocks:
hidden_states = up_block(hidden_states, temb, sample)
# 3. Post-process
hidden_states = self.norm_out(hidden_states, sample)
hidden_states = self.conv_act(hidden_states)
hidden_states = self.conv_out(hidden_states)
return hidden_states
@@ -1,3 +1,4 @@
import itertools
from dataclasses import dataclass
from typing import Optional
@@ -95,7 +96,6 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
out_channels: int = 3,
ch = 128,
ch_mult = [ 1,2,4,4 ],
block_out_channels = [128, 256, 512, 512],
use_gc_blocks = None,
down_block_types: tuple = None,
up_block_types: tuple = None,
@@ -112,15 +112,10 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
monitor=None,
ckpt_path=None,
lossconfig=None,
slice_mag_vae=False,
slice_compression_vae=False,
cache_compression_vae=False,
cache_mag_vae=False,
spatial_group_norm=False,
mini_batch_encoder=9,
mini_batch_decoder=3,
train_decoder_only=False,
train_encoder_only=False,
):
super().__init__()
self.image_key = image_key
@@ -130,9 +125,8 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
in_channels=in_channels,
out_channels=latent_channels,
down_block_types=down_block_types,
ch=ch,
ch_mult=ch_mult,
block_out_channels=block_out_channels,
ch = ch,
ch_mult = ch_mult,
use_gc_blocks=use_gc_blocks,
mid_block_type=mid_block_type,
mid_block_use_attention=mid_block_use_attention,
@@ -143,11 +137,7 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
act_fn=act_fn,
num_attention_heads=num_attention_heads,
double_z=True,
slice_mag_vae=slice_mag_vae,
slice_compression_vae=slice_compression_vae,
cache_compression_vae=cache_compression_vae,
cache_mag_vae=cache_mag_vae,
spatial_group_norm=spatial_group_norm,
mini_batch_encoder=mini_batch_encoder,
)
@@ -155,9 +145,8 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
in_channels=latent_channels,
out_channels=out_channels,
up_block_types=up_block_types,
ch=ch,
ch_mult=ch_mult,
block_out_channels=block_out_channels,
ch = ch,
ch_mult = ch_mult,
use_gc_blocks=use_gc_blocks,
mid_block_type=mid_block_type,
mid_block_use_attention=mid_block_use_attention,
@@ -167,11 +156,7 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
norm_num_groups=norm_num_groups,
act_fn=act_fn,
num_attention_heads=num_attention_heads,
slice_mag_vae=slice_mag_vae,
slice_compression_vae=slice_compression_vae,
cache_compression_vae=cache_compression_vae,
cache_mag_vae=cache_mag_vae,
spatial_group_norm=spatial_group_norm,
mini_batch_decoder=mini_batch_decoder,
)
@@ -181,15 +166,9 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
self.mini_batch_encoder = mini_batch_encoder
self.mini_batch_decoder = mini_batch_decoder
self.train_decoder_only = train_decoder_only
self.train_encoder_only = train_encoder_only
if train_decoder_only:
self.encoder.requires_grad_(False)
if self.quant_conv is not None:
self.quant_conv.requires_grad_(False)
if train_encoder_only:
self.decoder.requires_grad_(False)
if self.post_quant_conv is not None:
self.post_quant_conv.requires_grad_(False)
self.quant_conv.requires_grad_(False)
if monitor is not None:
self.monitor = monitor
if ckpt_path is not None:
@@ -211,28 +190,28 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
if k.startswith(ik):
print("Deleting key {} from state_dict.".format(k))
del sd[k]
m, u = self.load_state_dict(sd, strict=False) # loss.item can be ignored successfully
self.load_state_dict(sd, strict=False) # loss.item can be ignored successfully
print(f"Restored from {path}")
print(f"missing keys: {str(m)}, unexpected keys: {str(u)}")
def encode(self, x: torch.Tensor) -> EncoderOutput:
h = self.encoder(x)
if self.quant_conv is not None:
moments: torch.Tensor = self.quant_conv(h)
else:
moments: torch.Tensor = h
moments: torch.Tensor = self.quant_conv(h)
mean, logvar = moments.chunk(2, dim=1)
posterior = DiagonalGaussianDistribution(mean, logvar)
# return EncoderOutput(latent_dist=posterior)
return posterior
def decode(self, z: torch.Tensor) -> DecoderOutput:
if self.post_quant_conv is not None:
z = self.post_quant_conv(z)
z = self.post_quant_conv(z)
decoded = self.decoder(z)
# return DecoderOutput(sample=decoded)
return decoded
def forward(self, input, sample_posterior=True):
if input.ndim==4:
input = input.unsqueeze(2)
@@ -256,22 +235,30 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
return x
def training_step(self, batch, batch_idx, optimizer_idx):
# tic = time.time()
inputs = self.get_input(batch, self.image_key)
# print(f"get_input time {time.time() - tic}")
# tic = time.time()
reconstructions, posterior = self(inputs)
# print(f"model forward time {time.time() - tic}")
if optimizer_idx == 0:
# train encoder+decoder+logvar
aeloss, log_dict_ae = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("aeloss", aeloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_ae, prog_bar=False, logger=True, on_step=True, on_epoch=False)
# print(f"cal loss time {time.time() - tic}")
return aeloss
if optimizer_idx == 1:
# train the discriminator
discloss, log_dict_disc = self.loss(inputs, reconstructions, posterior, optimizer_idx, self.global_step,
last_layer=self.get_last_layer(), split="train")
self.log("discloss", discloss, prog_bar=True, logger=True, on_step=True, on_epoch=True)
self.log_dict(log_dict_disc, prog_bar=False, logger=True, on_step=True, on_epoch=False)
# print(f"cal loss time {time.time() - tic}")
return discloss
def validation_step(self, batch, batch_idx):
@@ -292,28 +279,17 @@ class AutoencoderKLMagvit_fromOmnigen(pl.LightningModule):
def configure_optimizers(self):
lr = self.learning_rate
if self.train_decoder_only:
if self.post_quant_conv is not None:
training_list = list(self.decoder.parameters()) + list(self.post_quant_conv.parameters())
else:
training_list = list(self.decoder.parameters())
opt_ae = torch.optim.AdamW(training_list, lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
elif self.train_encoder_only:
if self.quant_conv is not None:
training_list = list(self.encoder.parameters()) + list(self.quant_conv.parameters())
else:
training_list = list(self.encoder.parameters())
opt_ae = torch.optim.AdamW(training_list, lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
opt_ae = torch.optim.Adam(list(self.decoder.parameters())+
list(self.post_quant_conv.parameters()),
lr=lr, betas=(0.5, 0.9))
else:
training_list = list(self.encoder.parameters()) + list(self.decoder.parameters())
if self.quant_conv is not None:
training_list = training_list + list(self.quant_conv.parameters())
if self.post_quant_conv is not None:
training_list = training_list + list(self.post_quant_conv.parameters())
opt_ae = torch.optim.AdamW(training_list, lr=lr, betas=(0.9, 0.999), weight_decay=5e-2)
opt_disc = torch.optim.AdamW(
list(self.loss.discriminator3d.parameters()) + list(self.loss.discriminator.parameters()),
lr=lr, betas=(0.9, 0.999), weight_decay=5e-2
)
opt_ae = torch.optim.Adam(list(self.encoder.parameters())+
list(self.decoder.parameters())+
list(self.quant_conv.parameters())+
list(self.post_quant_conv.parameters()),
lr=lr, betas=(0.5, 0.9))
opt_disc = torch.optim.Adam(list(self.loss.discriminator3d.parameters()) + list(self.loss.discriminator.parameters()),
lr=lr, betas=(0.5, 0.9))
return [opt_ae, opt_disc], []
def get_last_layer(self):
+29 -310
View File
@@ -1,10 +1,6 @@
from typing import Any, Dict
import torch
import torch.nn as nn
from diffusers.utils import is_torch_version
from einops import rearrange
import numpy as np
from ..modules.vaemodules.activations import get_activation
from ..modules.vaemodules.common import CausalConv3d
from ..modules.vaemodules.down_blocks import get_down_block
@@ -12,16 +8,6 @@ from ..modules.vaemodules.mid_blocks import get_mid_block
from ..modules.vaemodules.up_blocks import get_up_block
def create_custom_forward(module, return_dict=None):
def custom_forward(*inputs):
if return_dict is not None:
return module(*inputs, return_dict=return_dict)
else:
return module(*inputs)
return custom_forward
class Encoder(nn.Module):
r"""
The `Encoder` layer of a variational autoencoder that encodes its input into a latent representation.
@@ -51,8 +37,6 @@ class Encoder(nn.Module):
Whether to double the number of output channels for the last block.
"""
_supports_gradient_checkpointing = True
def __init__(
self,
in_channels: int = 3,
@@ -60,7 +44,6 @@ class Encoder(nn.Module):
down_block_types = ("SpatialDownBlock3D",),
ch = 128,
ch_mult = [1,2,4,4,],
block_out_channels = [128, 256, 512, 512],
use_gc_blocks = None,
mid_block_type: str = "MidBlock3D",
mid_block_use_attention: bool = True,
@@ -71,17 +54,12 @@ class Encoder(nn.Module):
act_fn: str = "silu",
num_attention_heads: int = 1,
double_z: bool = True,
slice_mag_vae: bool = False,
slice_compression_vae: bool = False,
cache_compression_vae: bool = False,
cache_mag_vae: bool = False,
spatial_group_norm: bool = False,
mini_batch_encoder: int = 9,
verbose = False,
):
super().__init__()
if block_out_channels is None:
block_out_channels = [ch * i for i in ch_mult]
block_out_channels = [ch * i for i in ch_mult]
assert len(down_block_types) == len(block_out_channels), (
"Number of down block types must match number of block output channels."
)
@@ -140,16 +118,11 @@ class Encoder(nn.Module):
conv_out_channels = 2 * out_channels if double_z else out_channels
self.conv_out = CausalConv3d(block_out_channels[-1], conv_out_channels, kernel_size=3)
self.slice_mag_vae = slice_mag_vae
self.slice_compression_vae = slice_compression_vae
self.cache_compression_vae = cache_compression_vae
self.cache_mag_vae = cache_mag_vae
self.mini_batch_encoder = mini_batch_encoder
self.spatial_group_norm = spatial_group_norm
self.features_share = False
self.verbose = verbose
self.gradient_checkpointing = False
def set_padding_one_frame(self):
def _set_padding_one_frame(name, module):
if hasattr(module, 'padding_flag'):
@@ -172,124 +145,36 @@ class Encoder(nn.Module):
for name, module in self.named_children():
_set_padding_more_frame(name, module)
def set_magvit_padding_one_frame(self):
def _set_magvit_padding_one_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 3
for sub_name, sub_mod in module.named_children():
_set_magvit_padding_one_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_magvit_padding_one_frame(name, module)
def set_magvit_padding_more_frame(self):
def _set_magvit_padding_more_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 4
for sub_name, sub_mod in module.named_children():
_set_magvit_padding_more_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_magvit_padding_more_frame(name, module)
def set_cache_slice_vae_padding_one_frame(self):
def _set_cache_slice_vae_padding_one_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 5
for sub_name, sub_mod in module.named_children():
_set_cache_slice_vae_padding_one_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_cache_slice_vae_padding_one_frame(name, module)
def set_cache_slice_vae_padding_more_frame(self):
def _set_cache_slice_vae_padding_more_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 6
for sub_name, sub_mod in module.named_children():
_set_cache_slice_vae_padding_more_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_cache_slice_vae_padding_more_frame(name, module)
def set_3dgroupnorm_for_submodule(self):
def _set_3dgroupnorm_for_submodule(name, module):
if hasattr(module, 'set_3dgroupnorm'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.set_3dgroupnorm = True
for sub_name, sub_mod in module.named_children():
_set_3dgroupnorm_for_submodule(sub_name, sub_mod)
for name, module in self.named_children():
_set_3dgroupnorm_for_submodule(name, module)
def single_forward(self, x: torch.Tensor, previous_features: torch.Tensor, after_features: torch.Tensor) -> torch.Tensor:
# x: (B, C, T, H, W)
if torch.is_grad_enabled() and self.gradient_checkpointing:
ckpt_kwargs: Dict[str, Any] = {"use_reentrant": False} if is_torch_version(">=", "1.11.0") else {}
if previous_features is not None and after_features is None:
if self.features_share and previous_features is not None and after_features is None:
x = torch.concat([previous_features, x], 2)
elif previous_features is None and after_features is not None:
elif self.features_share and previous_features is None and after_features is not None:
x = torch.concat([x, after_features], 2)
elif previous_features is not None and after_features is not None:
elif self.features_share and previous_features is not None and after_features is not None:
x = torch.concat([previous_features, x, after_features], 2)
if torch.is_grad_enabled() and self.gradient_checkpointing:
x = torch.utils.checkpoint.checkpoint(
create_custom_forward(self.conv_in),
x,
**ckpt_kwargs,
)
else:
x = self.conv_in(x)
x = self.conv_in(x)
for down_block in self.down_blocks:
if torch.is_grad_enabled() and self.gradient_checkpointing:
x = torch.utils.checkpoint.checkpoint(
create_custom_forward(down_block),
x,
**ckpt_kwargs,
)
else:
x = down_block(x)
x = down_block(x)
x = self.mid_block(x)
if self.spatial_group_norm:
batch_size = x.shape[0]
x = rearrange(x, "b c t h w -> (b t) c h w")
x = self.conv_norm_out(x)
x = rearrange(x, "(b t) c h w -> b c t h w", b=batch_size)
else:
x = self.conv_norm_out(x)
x = self.conv_norm_out(x)
x = self.conv_act(x)
x = self.conv_out(x)
if previous_features is not None and after_features is None:
if self.features_share and previous_features is not None and after_features is None:
x = x[:, :, 1:]
elif previous_features is None and after_features is not None:
elif self.features_share and previous_features is None and after_features is not None:
x = x[:, :, :2]
elif previous_features is not None and after_features is not None:
elif self.features_share and previous_features is not None and after_features is not None:
x = x[:, :, 1:3]
return x
def forward(self, x: torch.Tensor) -> torch.Tensor:
if self.spatial_group_norm:
self.set_3dgroupnorm_for_submodule()
if self.cache_mag_vae:
self.set_magvit_padding_one_frame()
first_frames = self.single_forward(x[:, :, 0:1, :, :], None, None)
self.set_magvit_padding_more_frame()
new_pixel_values = [first_frames]
for i in range(1, x.shape[2], self.mini_batch_encoder):
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_encoder, :, :], None, None)
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
elif self.cache_compression_vae:
if self.slice_compression_vae:
_, _, f, _, _ = x.size()
if f % 2 != 0:
self.set_padding_one_frame()
@@ -303,33 +188,11 @@ class Encoder(nn.Module):
new_pixel_values = []
start_index = 0
previous_features = None
for i in range(start_index, x.shape[2], self.mini_batch_encoder):
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_encoder, :, :], None, None)
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
elif self.slice_compression_vae:
_, _, f, _, _ = x.size()
if f % 2 != 0:
self.set_padding_one_frame()
first_frames = self.single_forward(x[:, :, 0:1, :, :], None, None)
self.set_padding_more_frame()
new_pixel_values = [first_frames]
start_index = 1
else:
self.set_padding_more_frame()
new_pixel_values = []
start_index = 0
for i in range(start_index, x.shape[2], self.mini_batch_encoder):
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_encoder, :, :], None, None)
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
elif self.slice_mag_vae:
_, _, f, _, _ = x.size()
new_pixel_values = []
for i in range(0, x.shape[2], self.mini_batch_encoder):
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_encoder, :, :], None, None)
after_features = x[:, :, i + self.mini_batch_encoder: i + self.mini_batch_encoder + 4, :, :] if i + self.mini_batch_encoder < x.shape[2] else None
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_encoder, :, :], previous_features, after_features)
previous_features = x[:, :, i + self.mini_batch_encoder - 4: i + self.mini_batch_encoder, :, :]
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
else:
@@ -363,8 +226,6 @@ class Decoder(nn.Module):
The number of attention heads to use.
"""
_supports_gradient_checkpointing = True
def __init__(
self,
in_channels: int = 8,
@@ -372,7 +233,6 @@ class Decoder(nn.Module):
up_block_types = ("SpatialUpBlock3D",),
ch = 128,
ch_mult = [1,2,4,4,],
block_out_channels = [128, 256, 512, 512],
use_gc_blocks = None,
mid_block_type: str = "MidBlock3D",
mid_block_use_attention: bool = True,
@@ -382,17 +242,12 @@ class Decoder(nn.Module):
norm_num_groups: int = 32,
act_fn: str = "silu",
num_attention_heads: int = 1,
slice_mag_vae: bool = False,
slice_compression_vae: bool = False,
cache_compression_vae: bool = False,
cache_mag_vae: bool = False,
spatial_group_norm: bool = False,
mini_batch_decoder: int = 3,
verbose = False,
):
super().__init__()
if block_out_channels is None:
block_out_channels = [ch * i for i in ch_mult]
block_out_channels = [ch * i for i in ch_mult]
assert len(up_block_types) == len(block_out_channels), (
"Number of up block types must match number of block output channels."
)
@@ -454,16 +309,11 @@ class Decoder(nn.Module):
self.conv_out = CausalConv3d(block_out_channels[0], out_channels, kernel_size=3)
self.slice_mag_vae = slice_mag_vae
self.slice_compression_vae = slice_compression_vae
self.cache_compression_vae = cache_compression_vae
self.cache_mag_vae = cache_mag_vae
self.mini_batch_decoder = mini_batch_decoder
self.spatial_group_norm = spatial_group_norm
self.features_share = True
self.verbose = verbose
self.gradient_checkpointing = False
def set_padding_one_frame(self):
def _set_padding_one_frame(name, module):
if hasattr(module, 'padding_flag'):
@@ -485,90 +335,22 @@ class Decoder(nn.Module):
_set_padding_more_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_padding_more_frame(name, module)
def set_magvit_padding_one_frame(self):
def _set_magvit_padding_one_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 3
for sub_name, sub_mod in module.named_children():
_set_magvit_padding_one_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_magvit_padding_one_frame(name, module)
def set_magvit_padding_more_frame(self):
def _set_magvit_padding_more_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 4
for sub_name, sub_mod in module.named_children():
_set_magvit_padding_more_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_magvit_padding_more_frame(name, module)
def set_cache_slice_vae_padding_one_frame(self):
def _set_cache_slice_vae_padding_one_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 5
for sub_name, sub_mod in module.named_children():
_set_cache_slice_vae_padding_one_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_cache_slice_vae_padding_one_frame(name, module)
def set_cache_slice_vae_padding_more_frame(self):
def _set_cache_slice_vae_padding_more_frame(name, module):
if hasattr(module, 'padding_flag'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 6
for sub_name, sub_mod in module.named_children():
_set_cache_slice_vae_padding_more_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_cache_slice_vae_padding_more_frame(name, module)
def set_3dgroupnorm_for_submodule(self):
def _set_3dgroupnorm_for_submodule(name, module):
if hasattr(module, 'set_3dgroupnorm'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.set_3dgroupnorm = True
for sub_name, sub_mod in module.named_children():
_set_3dgroupnorm_for_submodule(sub_name, sub_mod)
for name, module in self.named_children():
_set_3dgroupnorm_for_submodule(name, module)
def clear_cache(self):
def _clear_cache(name, module):
if hasattr(module, 'prev_features'):
if self.verbose:
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.prev_features = None
for sub_name, sub_mod in module.named_children():
_clear_cache(sub_name, sub_mod)
for name, module in self.named_children():
_clear_cache(name, module)
def single_forward(self, x: torch.Tensor, previous_features: torch.Tensor, after_features: torch.Tensor) -> torch.Tensor:
# x: (B, C, T, H, W)
if torch.is_grad_enabled() and self.gradient_checkpointing:
ckpt_kwargs: Dict[str, Any] = {"use_reentrant": False} if is_torch_version(">=", "1.11.0") else {}
if previous_features is not None and after_features is None:
if self.features_share and previous_features is not None and after_features is None:
b, c, t, h, w = x.size()
x = torch.concat([previous_features, x], 2)
x = self.conv_in(x)
x = self.mid_block(x)
x = x[:, :, -t:]
elif previous_features is None and after_features is not None:
elif self.features_share and previous_features is None and after_features is not None:
b, c, t, h, w = x.size()
x = torch.concat([x, after_features], 2)
x = self.conv_in(x)
x = self.mid_block(x)
x = x[:, :, :t]
elif previous_features is not None and after_features is not None:
elif self.features_share and previous_features is not None and after_features is not None:
_, _, t_1, _, _ = previous_features.size()
_, _, t_2, _, _ = x.size()
x = torch.concat([previous_features, x, after_features], 2)
@@ -576,76 +358,20 @@ class Decoder(nn.Module):
x = self.mid_block(x)
x = x[:, :, t_1:(t_1 + t_2)]
else:
if torch.is_grad_enabled() and self.gradient_checkpointing:
x = torch.utils.checkpoint.checkpoint(
create_custom_forward(self.conv_in),
x,
**ckpt_kwargs,
)
x = torch.utils.checkpoint.checkpoint(
create_custom_forward(self.mid_block),
x,
**ckpt_kwargs,
)
else:
x = self.conv_in(x)
x = self.mid_block(x)
x = self.conv_in(x)
x = self.mid_block(x)
for up_block in self.up_blocks:
if torch.is_grad_enabled() and self.gradient_checkpointing:
x = torch.utils.checkpoint.checkpoint(
create_custom_forward(up_block),
x,
**ckpt_kwargs,
)
else:
x = up_block(x)
if self.spatial_group_norm:
batch_size = x.shape[0]
x = rearrange(x, "b c t h w -> (b t) c h w")
x = self.conv_norm_out(x)
x = rearrange(x, "(b t) c h w -> b c t h w", b=batch_size)
else:
x = self.conv_norm_out(x)
x = up_block(x)
x = self.conv_norm_out(x)
x = self.conv_act(x)
x = self.conv_out(x)
return x
def forward(self, x: torch.Tensor) -> torch.Tensor:
if self.spatial_group_norm:
self.set_3dgroupnorm_for_submodule()
if self.cache_mag_vae:
self.set_magvit_padding_one_frame()
first_frames = self.single_forward(x[:, :, 0:1, :, :], None, None)
new_pixel_values = [first_frames]
for i in range(1, x.shape[2], self.mini_batch_decoder):
self.set_magvit_padding_more_frame()
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_decoder, :, :], None, None)
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
elif self.cache_compression_vae:
_, _, f, _, _ = x.size()
if f == 1:
self.set_padding_one_frame()
first_frames = self.single_forward(x[:, :, :1, :, :], None, None)
new_pixel_values = [first_frames]
start_index = 1
else:
self.set_cache_slice_vae_padding_one_frame()
first_frames = self.single_forward(x[:, :, :self.mini_batch_decoder, :, :], None, None)
new_pixel_values = [first_frames]
start_index = self.mini_batch_decoder
for i in range(start_index, x.shape[2], self.mini_batch_decoder):
self.set_cache_slice_vae_padding_more_frame()
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_decoder, :, :], None, None)
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
elif self.slice_compression_vae:
if self.slice_compression_vae:
_, _, f, _, _ = x.size()
if f % 2 != 0:
self.set_padding_one_frame()
@@ -665,13 +391,6 @@ class Decoder(nn.Module):
previous_features = x[:, :, i: i + self.mini_batch_decoder, :, :]
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
elif self.slice_mag_vae:
_, _, f, _, _ = x.size()
new_pixel_values = []
for i in range(0, x.shape[2], self.mini_batch_decoder):
next_frames = self.single_forward(x[:, :, i: i + self.mini_batch_decoder, :, :], None, None)
new_pixel_values.append(next_frames)
new_pixel_values = torch.cat(new_pixel_values, dim=2)
else:
new_pixel_values = self.single_forward(x, None, None)
return new_pixel_values
+1 -2
View File
@@ -1,8 +1,7 @@
#-*- encoding:utf-8 -*-
import torch
from pytorch_lightning.callbacks import Callback
from torch import nn
from pytorch_lightning.callbacks import Callback
class LitEma(nn.Module):
def __init__(self, model, decay=0.9999, use_num_upates=True):
@@ -2,15 +2,12 @@ import torch
import torch.nn as nn
import torch.nn.functional as F
from taming.modules.losses.vqperceptual import * # TODO: taming dependency yes/no?
from ..vaemodules.discriminator import Discriminator3D
class LPIPSWithDiscriminator(nn.Module):
def __init__(self, disc_start, logvar_init=0.0, kl_weight=1.0, pixelloss_weight=1.0,
disc_num_layers=3, disc_in_channels=3, disc_factor=1.0, disc_weight=1.0,
perceptual_weight=1.0, use_actnorm=False, disc_conditional=False,
outlier_penalty_loss_r=3.0, outlier_penalty_loss_weight=1e5,
perceptual_weight=1.0, use_actnorm=False, disc_conditional=False,
disc_loss="hinge", l2_loss_weight=0.0, l1_loss_weight=1.0):
super().__init__()
@@ -35,8 +32,6 @@ class LPIPSWithDiscriminator(nn.Module):
self.disc_factor = disc_factor
self.discriminator_weight = disc_weight
self.disc_conditional = disc_conditional
self.outlier_penalty_loss_r = outlier_penalty_loss_r
self.outlier_penalty_loss_weight = outlier_penalty_loss_weight
self.l1_loss_weight = l1_loss_weight
self.l2_loss_weight = l2_loss_weight
@@ -53,18 +48,6 @@ class LPIPSWithDiscriminator(nn.Module):
d_weight = d_weight * self.discriminator_weight
return d_weight
def outlier_penalty_loss(self, posteriors, r):
batch_size, channels, frames, height, width = posteriors.shape
mean_X = posteriors.mean(dim=(3, 4), keepdim=True)
std_X = posteriors.std(dim=(3, 4), keepdim=True)
diff = torch.abs(posteriors - mean_X)
penalty = torch.maximum(diff - r * std_X, torch.zeros_like(diff))
opl = penalty.sum(dim=(3, 4)) / (height * width)
opl_final = opl.mean(dim=(0, 1, 2))
return opl_final
def forward(self, inputs, reconstructions, posteriors, optimizer_idx,
global_step, last_layer=None, cond=None, split="train",
weights=None):
@@ -79,6 +62,15 @@ class LPIPSWithDiscriminator(nn.Module):
# get new loss_weight
loss_weights = 1
# b, _ ,f, _, _ = reconstructions.size()
# loss_weights = torch.ones([b, f]).view(b, 1, f, 1, 1)
# loss_weights[:, :, 0] = 3
# for i in range(1, f, 8):
# loss_weights[:, :, i - 1] = 3
# loss_weights[:, :, i] = 3
# loss_weights[:, :, -1] = 3
# loss_weights = loss_weights.permute(0, 2, 1, 3, 4).flatten(0, 1).to(reconstructions.device)
inputs = inputs.permute(0, 2, 1, 3, 4).flatten(0, 1)
reconstructions = reconstructions.permute(0, 2, 1, 3, 4).flatten(0, 1)
@@ -101,8 +93,6 @@ class LPIPSWithDiscriminator(nn.Module):
kl_loss = posteriors.kl()
kl_loss = torch.sum(kl_loss) / kl_loss.shape[0]
outlier_penalty_loss = self.outlier_penalty_loss(posteriors.mode(), self.outlier_penalty_loss_r) * self.outlier_penalty_loss_weight
# now the GAN part
if optimizer_idx == 0:
# generator update
@@ -119,13 +109,13 @@ class LPIPSWithDiscriminator(nn.Module):
try:
d_weight = self.calculate_adaptive_weight(nll_loss, g_loss, last_layer=last_layer)
except RuntimeError:
# assert not self.training
assert not self.training
d_weight = torch.tensor(0.0)
else:
d_weight = torch.tensor(0.0)
disc_factor = adopt_weight(self.disc_factor, global_step, threshold=self.discriminator_iter_start)
loss = weighted_nll_loss + self.kl_weight * kl_loss + d_weight * disc_factor * g_loss + outlier_penalty_loss
loss = weighted_nll_loss + self.kl_weight * kl_loss + d_weight * disc_factor * g_loss
log = {"{}/total_loss".format(split): loss.clone().detach().mean(), "{}/logvar".format(split): self.logvar.detach(),
"{}/kl_loss".format(split): kl_loss.detach().mean(), "{}/nll_loss".format(split): nll_loss.detach().mean(),
View File
View File
+27 -140
View File
@@ -8,17 +8,6 @@ from einops import rearrange, repeat
from .activations import get_activation
try:
current_version = torch.__version__
version_numbers = [int(x) for x in current_version.split('.')[:2]]
if version_numbers[0] < 2 or (version_numbers[0] == 2 and version_numbers[1] < 2):
need_to_float = True
else:
need_to_float = False
except Exception as e:
print("Encountered an error with Torch version. Set the data type to float in the VAE. ")
need_to_float = False
def cast_tuple(t, length = 1):
return t if isinstance(t, tuple) else ((t,) * length)
@@ -49,7 +38,7 @@ class CausalConv3d(nn.Conv3d):
assert len(dilation) == 3, f"Dilation must be a 3-tuple, got {dilation} instead."
t_ks, h_ks, w_ks = kernel_size
self.t_stride, h_stride, w_stride = stride
_, h_stride, w_stride = stride
t_dilation, h_dilation, w_dilation = dilation
t_pad = (t_ks - 1) * t_dilation
@@ -65,7 +54,6 @@ class CausalConv3d(nn.Conv3d):
self.temporal_padding = t_pad
self.temporal_padding_origin = math.ceil(((t_ks - 1) * w_dilation + (1 - w_stride)) / 2)
self.padding_flag = 0
self.prev_features = None
super().__init__(
in_channels=in_channels,
@@ -77,106 +65,40 @@ class CausalConv3d(nn.Conv3d):
**kwargs,
)
def _clear_conv_cache(self):
del self.prev_features
self.prev_features = None
def forward(self, x: torch.Tensor) -> torch.Tensor:
# x: (B, C, T, H, W)
dtype = x.dtype
if need_to_float:
x = x.float()
if self.padding_flag == 0:
x = F.pad(
x,
pad=(0, 0, 0, 0, self.temporal_padding, 0),
mode="replicate", # TODO: check if this is necessary
)
x = x.to(dtype=dtype)
return super().forward(x)
elif self.padding_flag == 3:
x = F.pad(
x,
pad=(0, 0, 0, 0, self.temporal_padding, 0),
mode="replicate", # TODO: check if this is necessary
)
x = x.to(dtype=dtype)
# Clear cache before
self._clear_conv_cache()
# We could move these to the cpu for a lower VRAM
self.prev_features = x[:, :, -self.temporal_padding:].clone()
b, c, f, h, w = x.size()
outputs = []
i = 0
while i + self.temporal_padding + 1 <= f:
out = super().forward(x[:, :, i:i + self.temporal_padding + 1])
i += self.t_stride
outputs.append(out)
return torch.concat(outputs, 2)
elif self.padding_flag == 4:
if self.t_stride == 2:
x = torch.concat(
[self.prev_features[:, :, -(self.temporal_padding - 1):], x], dim = 2
)
else:
x = torch.concat(
[self.prev_features, x], dim = 2
)
x = x.to(dtype=dtype)
# Clear cache before
self._clear_conv_cache()
# We could move these to the cpu for a lower VRAM
self.prev_features = x[:, :, -self.temporal_padding:].clone()
b, c, f, h, w = x.size()
outputs = []
i = 0
while i + self.temporal_padding + 1 <= f:
out = super().forward(x[:, :, i:i + self.temporal_padding + 1])
i += self.t_stride
outputs.append(out)
return torch.concat(outputs, 2)
elif self.padding_flag == 5:
x = F.pad(
x,
pad=(0, 0, 0, 0, self.temporal_padding, 0),
mode="replicate", # TODO: check if this is necessary
)
x = x.to(dtype=dtype)
# Clear cache before
self._clear_conv_cache()
# We could move these to the cpu for a lower VRAM
self.prev_features = x[:, :, -self.temporal_padding:].clone()
return super().forward(x)
elif self.padding_flag == 6:
if self.t_stride == 2:
x = torch.concat(
[self.prev_features[:, :, -(self.temporal_padding - 1):], x], dim = 2
)
else:
x = torch.concat(
[self.prev_features, x], dim = 2
)
# Clear cache before
self._clear_conv_cache()
# We could move these to the cpu for a lower VRAM
self.prev_features = x[:, :, -self.temporal_padding:].clone()
x = x.to(dtype=dtype)
return super().forward(x)
else:
x = F.pad(
x,
pad=(0, 0, 0, 0, self.temporal_padding_origin, self.temporal_padding_origin),
)
x = x.to(dtype=dtype)
return super().forward(x)
return super().forward(x)
def set_padding_one_frame(self):
def _set_padding_one_frame(name, module):
if hasattr(module, 'padding_flag'):
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 1
for sub_name, sub_mod in module.named_children():
_set_padding_one_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_padding_one_frame(name, module)
def set_padding_more_frame(self):
def _set_padding_more_frame(name, module):
if hasattr(module, 'padding_flag'):
print('Set pad mode for module[%s] type=%s' % (name, str(type(module))))
module.padding_flag = 2
for sub_name, sub_mod in module.named_children():
_set_padding_more_frame(sub_name, sub_mod)
for name, module in self.named_children():
_set_padding_more_frame(name, module)
class ResidualBlock2D(nn.Module):
def __init__(
@@ -220,29 +142,15 @@ class ResidualBlock2D(nn.Module):
else:
self.shortcut = nn.Identity()
self.set_3dgroupnorm = False
def forward(self, x: torch.Tensor) -> torch.Tensor:
shortcut = self.shortcut(x)
if self.set_3dgroupnorm:
batch_size = x.shape[0]
x = rearrange(x, "b c t h w -> (b t) c h w")
x = self.norm1(x)
x = rearrange(x, "(b t) c h w -> b c t h w", b=batch_size)
else:
x = self.norm1(x)
x = self.norm1(x)
x = self.nonlinearity(x)
x = self.conv1(x)
if self.set_3dgroupnorm:
batch_size = x.shape[0]
x = rearrange(x, "b c t h w -> (b t) c h w")
x = self.norm2(x)
x = rearrange(x, "(b t) c h w -> b c t h w", b=batch_size)
else:
x = self.norm2(x)
x = self.norm2(x)
x = self.nonlinearity(x)
x = self.dropout(x)
@@ -293,29 +201,15 @@ class ResidualBlock3D(nn.Module):
else:
self.shortcut = nn.Identity()
self.set_3dgroupnorm = False
def forward(self, x: torch.Tensor) -> torch.Tensor:
shortcut = self.shortcut(x)
if self.set_3dgroupnorm:
batch_size = x.shape[0]
x = rearrange(x, "b c t h w -> (b t) c h w")
x = self.norm1(x)
x = rearrange(x, "(b t) c h w -> b c t h w", b=batch_size)
else:
x = self.norm1(x)
x = self.norm1(x)
x = self.nonlinearity(x)
x = self.conv1(x)
if self.set_3dgroupnorm:
batch_size = x.shape[0]
x = rearrange(x, "b c t h w -> (b t) c h w")
x = self.norm2(x)
x = rearrange(x, "(b t) c h w -> b c t h w", b=batch_size)
else:
x = self.norm2(x)
x = self.norm2(x)
x = self.nonlinearity(x)
x = self.dropout(x)
@@ -344,18 +238,11 @@ class SpatialNorm2D(nn.Module):
self.norm = nn.GroupNorm(num_channels=f_channels, num_groups=32, eps=1e-6, affine=True)
self.conv_y = nn.Conv2d(zq_channels, f_channels, kernel_size=1, stride=1, padding=0)
self.conv_b = nn.Conv2d(zq_channels, f_channels, kernel_size=1, stride=1, padding=0)
self.set_3dgroupnorm = False
def forward(self, f: torch.FloatTensor, zq: torch.FloatTensor) -> torch.FloatTensor:
f_size = f.shape[-2:]
zq = F.interpolate(zq, size=f_size, mode="nearest")
if self.set_3dgroupnorm:
batch_size = f.shape[0]
f = rearrange(f, "b c t h w -> (b t) c h w")
norm_f = self.norm(f)
norm_f = rearrange(norm_f, "(b t) c h w -> b c t h w", b=batch_size)
else:
norm_f = self.norm(f)
norm_f = self.norm(f)
new_f = norm_f * self.conv_y(zq) + self.conv_b(zq)
return new_f
View File
View File

Some files were not shown because too many files have changed in this diff Show More