Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a6d4d815f8 | ||
|
|
42da276006 | ||
|
|
7b3c4fbdd1 | ||
|
|
f041bd3f49 | ||
|
|
fbb9a73ab4 | ||
|
|
daea023383 | ||
|
|
0422d18377 | ||
|
|
f46c6d923f | ||
|
|
79a31b0b45 | ||
|
|
ef98769b46 | ||
|
|
869c5f7370 | ||
|
|
74529f22a4 | ||
|
|
2a2d67e792 | ||
|
|
bb3367c9ac |
@@ -163,7 +163,6 @@ steps:
|
||||
- path:
|
||||
- "csrc/attn/vsa/**"
|
||||
- "csrc/attn/tk/**"
|
||||
- "csrc/attn/tests/test_vsa.py"
|
||||
- "csrc/attn/setup_vsa.py"
|
||||
- "csrc/attn/config_vsa.py"
|
||||
- "csrc/attn/vsa.cpp"
|
||||
|
||||
@@ -18,12 +18,6 @@ on:
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
python_3_12_cuda_12_9:
|
||||
description: 'Build Python 3.12 image Cuda 12.9'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -55,13 +49,4 @@ jobs:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile.python3.12
|
||||
tag_suffix: py3.12
|
||||
secrets: inherit
|
||||
|
||||
build-python-3-12-cuda-12-9:
|
||||
if: ${{ github.event.inputs.python_3_12_cuda_12_9 == 'true' }}
|
||||
uses: ./.github/workflows/build-image-template.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile.python3.12.cuda12.9.1
|
||||
tag_suffix: py3.12-cuda12.9.1
|
||||
secrets: inherit
|
||||
@@ -1,257 +0,0 @@
|
||||
name: Publish Video Sparse Attention Kernel to PyPI on Version Change
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "csrc/attn/setup_vsa.py"
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
check-version-change:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
version-changed: ${{ steps.check-version.outputs.changed }}
|
||||
new-version: ${{ steps.check-version.outputs.new-version }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check-version
|
||||
run: |
|
||||
cd csrc/attn
|
||||
# Get current commit's version
|
||||
NEW_VERSION=$(grep -oP 'VERSION\s*=\s*"\K[^"]+' setup_vsa.py)
|
||||
echo "New version: $NEW_VERSION"
|
||||
|
||||
# Get previous version from git history
|
||||
OLD_VERSION=$(git show HEAD~1:./setup_vsa.py | grep -oP 'VERSION\s*=\s*"\K[^"]+' || echo "0.0.0")
|
||||
echo "Old version: $OLD_VERSION"
|
||||
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "Version changed from $OLD_VERSION to $NEW_VERSION"
|
||||
echo "changed=true" >> $GITHUB_OUTPUT
|
||||
echo "new-version=$NEW_VERSION" >> $GITHUB_OUTPUT
|
||||
else
|
||||
echo "Version did not change"
|
||||
echo "changed=false" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
build_wheels:
|
||||
name: Build Wheel
|
||||
needs: check-version-change
|
||||
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# Using ubuntu-20.04 instead of 22.04 for more compatibility (glibc). Ideally we'd use the
|
||||
# manylinux docker image, but I haven't figured out how to install CUDA on manylinux.
|
||||
os: [ubuntu-22.04]
|
||||
python-version: ['3.10', '3.11', '3.12', '3.13']
|
||||
# For version reference https://pytorch.org/get-started/previous-versions/
|
||||
torch-cuda:
|
||||
- torch-version: '2.5.1'
|
||||
cuda-version: '12.4.1'
|
||||
torch-cuda-short: 'cu124'
|
||||
- torch-version: '2.6.0'
|
||||
cuda-version: '12.6.3'
|
||||
torch-cuda-short: 'cu126'
|
||||
- torch-version: '2.7.1'
|
||||
cuda-version: '12.8.0'
|
||||
torch-cuda-short: 'cu128'
|
||||
|
||||
steps:
|
||||
- name: Free up disk space
|
||||
run: |
|
||||
echo "Initial disk space:"
|
||||
df -h
|
||||
|
||||
# Remove large directories
|
||||
sudo rm -rf /usr/share/dotnet
|
||||
sudo rm -rf /usr/local/lib/android
|
||||
sudo rm -rf /opt/ghc
|
||||
sudo rm -rf /usr/local/share/boost
|
||||
sudo rm -rf /usr/share/swift
|
||||
sudo rm -rf /usr/local/lib/node_modules
|
||||
sudo rm -rf /usr/local/share/powershell
|
||||
sudo rm -rf /usr/share/rust
|
||||
sudo rm -rf /usr/local/.ghcup
|
||||
|
||||
# Remove cached files
|
||||
sudo rm -rf /var/lib/apt/lists/*
|
||||
sudo rm -rf /var/cache/apt/archives/*
|
||||
|
||||
echo "Disk space after cleanup:"
|
||||
df -h
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
|
||||
- name: Install CUDA ${{ matrix.torch-cuda.cuda-version }}
|
||||
uses: Jimver/cuda-toolkit@v0.2.21
|
||||
id: cuda-toolkit
|
||||
with:
|
||||
cuda: ${{ matrix.torch-cuda.cuda-version }}
|
||||
linux-local-args: '["--toolkit"]'
|
||||
method: 'network'
|
||||
|
||||
- name: Install dependencies (GCC, Clang, CUDA Paths, Git)
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y git patchelf gcc-11 g++-11 clang-11
|
||||
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
# Allow Git to Access Safe Directory
|
||||
git config --global --add safe.directory /__w/FastVideo/FastVideo
|
||||
|
||||
# Set CUDA environment variables
|
||||
export CUDA_HOME=/usr/local/cuda-${{ matrix.torch-cuda.cuda-version }}
|
||||
export PATH=${CUDA_HOME}/bin:${PATH}
|
||||
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
|
||||
# Verify installation
|
||||
gcc --version
|
||||
g++ --version
|
||||
clang-11 --version
|
||||
nvcc --version
|
||||
|
||||
- name: Install PyTorch ${{ matrix.torch-cuda.torch-version }}+cu${{ matrix.torch-cuda.cuda-version }}
|
||||
run: |
|
||||
pip install --upgrade pip
|
||||
# With python 3.13 and torch 2.5.1, unless we update typing-extensions, we get error
|
||||
# AttributeError: attribute '__default__' of 'typing.ParamSpec' objects is not writable
|
||||
pip install typing-extensions==4.12.2
|
||||
# We want to figure out the CUDA version to download pytorch
|
||||
# e.g. we can have system CUDA version being 11.7 but if torch==1.12 then we need to download the wheel from cu116
|
||||
# see https://github.com/pytorch/pytorch/blob/main/RELEASE.md#release-compatibility-matrix
|
||||
pip install --no-cache-dir torch==${{ matrix.torch-cuda.torch-version }} --index-url https://download.pytorch.org/whl/${{matrix.torch-cuda.torch-cuda-short}}
|
||||
nvcc --version
|
||||
python --version
|
||||
python -c "import torch; print('PyTorch:', torch.__version__)"
|
||||
python -c "import torch; print('CUDA:', torch.version.cuda)"
|
||||
python -c "from torch.utils import cpp_extension; print (cpp_extension.CUDA_HOME)"
|
||||
|
||||
- name: Build wheel
|
||||
run: |
|
||||
export PYTHONPATH=$GITHUB_WORKSPACE:$PYTHONPATH
|
||||
|
||||
# We want setuptools >= 49.6.0 otherwise we can't compile the extension if system CUDA version is 11.7 and pytorch cuda version is 11.6
|
||||
# https://github.com/pytorch/pytorch/blob/664058fa83f1d8eede5d66418abff6e20bd76ca8/torch/utils/cpp_extension.py#L810
|
||||
# However this still fails so I'm using a newer version of setuptools
|
||||
pip install setuptools
|
||||
pip install ninja packaging wheel
|
||||
|
||||
cd csrc/attn # Move into the correct folder
|
||||
git submodule update --init --recursive # Ensure ThunderKittens submodule is initialized
|
||||
python setup_vsa.py bdist_wheel --dist-dir=dist
|
||||
|
||||
- name: Rename wheel file
|
||||
run: |
|
||||
cd csrc/attn
|
||||
|
||||
CUDA_SHORT_VERSION=$(echo ${{ matrix.torch-cuda.cuda-version }} | cut -d. -f1,2 | sed 's/\.//g')
|
||||
TORCH_SHORT_VERSION=$(echo ${{ matrix.torch-cuda.torch-version }} | cut -d. -f1,2)
|
||||
# Get the correct version format
|
||||
tmpname=cu${CUDA_SHORT_VERSION}torch${TORCH_SHORT_VERSION}
|
||||
wheel_name=$(ls dist/*whl | xargs -n 1 basename | sed "s/-/+$tmpname-/2")
|
||||
# Rename with version information
|
||||
ls dist/*whl |xargs -I {} mv {} dist/${wheel_name}
|
||||
echo "wheel_name=${wheel_name}" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload wheel artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: ${{ env.wheel_name }}
|
||||
path: csrc/attn/dist/*.whl
|
||||
retention-days: 90
|
||||
|
||||
publish_package:
|
||||
name: Publish package
|
||||
needs: [build_wheels, check-version-change]
|
||||
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
|
||||
runs-on: ubuntu-22.04
|
||||
permissions:
|
||||
id-token: write # Needed for OIDC Trusted Publishing
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.10'
|
||||
|
||||
- name: Install CUDA 12.4.1
|
||||
uses: Jimver/cuda-toolkit@v0.2.21
|
||||
id: cuda-toolkit
|
||||
with:
|
||||
cuda: 12.4.1
|
||||
linux-local-args: '["--toolkit"]'
|
||||
method: 'network'
|
||||
sub-packages: '["nvcc"]'
|
||||
|
||||
- name: Install dependencies (GCC, Clang, CUDA Paths, Git)
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y git patchelf gcc-11 g++-11 clang-11
|
||||
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
# Allow Git to Access Safe Directory
|
||||
git config --global --add safe.directory /__w/FastVideo/FastVideo
|
||||
|
||||
# Set CUDA environment variables
|
||||
export CUDA_HOME=/usr/local/cuda-12.4.1
|
||||
export PATH=${CUDA_HOME}/bin:${PATH}
|
||||
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
|
||||
# Verify installation
|
||||
gcc --version
|
||||
g++ --version
|
||||
clang-11 --version
|
||||
nvcc --version
|
||||
|
||||
- name: Install PyTorch 2.5.1+cu12.4.1
|
||||
run: |
|
||||
pip install --upgrade pip
|
||||
# With python 3.13 and torch 2.5.1, unless we update typing-extensions, we get error
|
||||
# AttributeError: attribute '__default__' of 'typing.ParamSpec' objects is not writable
|
||||
pip install typing-extensions==4.12.2
|
||||
# We want to figure out the CUDA version to download pytorch
|
||||
# e.g. we can have system CUDA version being 11.7 but if torch==1.12 then we need to download the wheel from cu116
|
||||
# see https://github.com/pytorch/pytorch/blob/main/RELEASE.md#release-compatibility-matrix
|
||||
export TORCH_CUDA_VERSION=124
|
||||
pip install --no-cache-dir torch==2.5.1 --index-url https://download.pytorch.org/whl/cu${TORCH_CUDA_VERSION}
|
||||
nvcc --version
|
||||
python --version
|
||||
python -c "import torch; print('PyTorch:', torch.__version__)"
|
||||
python -c "import torch; print('CUDA:', torch.version.cuda)"
|
||||
python -c "from torch.utils import cpp_extension; print (cpp_extension.CUDA_HOME)"
|
||||
|
||||
- name: Build source distribution
|
||||
run: |
|
||||
export PYTHONPATH=$GITHUB_WORKSPACE:$PYTHONPATH
|
||||
|
||||
# We want setuptools >= 49.6.0 otherwise we can't compile the extension if system CUDA version is 11.7 and pytorch cuda version is 11.6
|
||||
# https://github.com/pytorch/pytorch/blob/664058fa83f1d8eede5d66418abff6e20bd76ca8/torch/utils/cpp_extension.py#L810
|
||||
# However this still fails so I'm using a newer version of setuptools
|
||||
pip install setuptools
|
||||
pip install ninja packaging wheel
|
||||
|
||||
cd csrc/attn # Move into the correct folder
|
||||
git submodule update --init --recursive # Ensure ThunderKittens submodule is initialized
|
||||
python setup_vsa.py sdist --dist-dir=dist
|
||||
|
||||
- name: Publish release distributions to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages-dir: csrc/attn/dist/
|
||||
@@ -42,7 +42,6 @@ docs/_build/
|
||||
docs/source/getting_started/examples/
|
||||
docs/source/inference/examples/
|
||||
docs/source/training/examples/
|
||||
docs/source/distillation/examples/
|
||||
|
||||
# VSCode
|
||||
.vscode/
|
||||
@@ -64,6 +63,3 @@ docs/source/distillation/examples/
|
||||
!docs/source/_static/images/**/*.png
|
||||
!comfyui/assets/**/*.png
|
||||
!comfyui/assets/**/*.gif
|
||||
|
||||
dmd_t2v_output/
|
||||
sf_output/
|
||||
@@ -22,7 +22,6 @@ exclude: |
|
||||
examples/.*|
|
||||
.github/workflows/fastvideo-publish.yml|
|
||||
.github/workflows/sta-publish.yml|
|
||||
.github/workflows/vsa-publish.yml|
|
||||
.github/workflows/build-image-template.yml|
|
||||
docs/source/inference/support_matrix.md
|
||||
)
|
||||
|
||||
@@ -1,21 +1,21 @@
|
||||
<div align="center">
|
||||
<img src=assets/logos/logo.svg width="30%"/>
|
||||
<img src=assets/logo.jpg width="30%"/>
|
||||
</div>
|
||||
|
||||
**FastVideo is a unified post-training and inference framework for accelerated video generation.**
|
||||
**FastVideo is a unified framework for accelerated video generation.**
|
||||
|
||||
FastVideo features an end-to-end unified pipeline for accelerating diffusion models, starting from data preprocessing to model training, finetuning, distillation, and inference. FastVideo is designed to be modular and extensible, allowing users to easily add new optimizations and techniques. Whether it is training-free optimizations or post-training optimizations, FastVideo has you covered.
|
||||
It features a clean, consistent API that works across popular video models, making it easier for developers to author new models and incorporate system- or kernel-level optimizations.
|
||||
With FastVideo's optimizations, you can achieve more than 3x inference improvement compared to other systems.
|
||||
|
||||
<p align="center">
|
||||
| 🕹️ <a href="https://fastwan.fastvideo.org/"<b>Online Demo</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start.html"><b> Quick Start</b></a> | 🤗 <a href="https://huggingface.co/collections/FastVideo/fastwan-6886a305d9799c8cd1496408" target="_blank"><b>FastWan</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-38u6p1jqe-yDI1QJOCEnbtkLoaI5bjZQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://ibb.co/rG0QpZdw" target="_blank"> <b> WeChat </b> </a> |
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start.html"><b> Quick Start</b></a> | 🤗 <a href="https://huggingface.co/FastVideo/FastHunyuan" target="_blank"><b>FastHunyuan</b></a> | 🤗 <a href="https://huggingface.co/FastVideo/FastMochi-diffusers" target="_blank"><b>FastMochi</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-38u6p1jqe-yDI1QJOCEnbtkLoaI5bjZQ" target="_blank"> <b>Slack</b> </a> |
|
||||
</p>
|
||||
|
||||
<div align="center">
|
||||
<img src=assets/fastwan.png width="90%"/>
|
||||
<img src=assets/perf.png width="90%"/>
|
||||
</div>
|
||||
|
||||
## NEWS
|
||||
- ```2025/08/04```: Release [FastWan](https://hao-ai-lab.github.io/FastVideo/distillation/dmd.html) models and [Sparse-Distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/).
|
||||
- ```2025/06/14```: Release finetuning and inference code for [VSA](https://arxiv.org/pdf/2505.13389)
|
||||
- ```2025/04/24```: [FastVideo V1](https://hao-ai-lab.github.io/blogs/fastvideo/) is released!
|
||||
- ```2025/02/18```: Release the inference code for [Sliding Tile Attention](https://hao-ai-lab.github.io/blogs/sta/).
|
||||
@@ -23,19 +23,20 @@ FastVideo features an end-to-end unified pipeline for accelerating diffusion mod
|
||||
## Key Features
|
||||
|
||||
FastVideo has the following features:
|
||||
- End-to-end post-training support:
|
||||
- [Sparse distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/) for Wan2.1 and Wan2.2 to achineve >50x denoising speedup
|
||||
- Data preprocessing pipeline for video data
|
||||
- Support full finetuning and LoRA finetuning for state-of-the-art open video DiTs
|
||||
- Scalable training with FSDP2, sequence parallelism, and selective activation checkpointing, with near linear scaling to 64 GPUs
|
||||
- State-of-the-art performance optimizations for inference
|
||||
- [Video Sparse Attention](https://arxiv.org/pdf/2505.13389)
|
||||
- [Sliding Tile Attention](https://arxiv.org/pdf/2502.04507)
|
||||
- [TeaCache](https://arxiv.org/pdf/2411.19108)
|
||||
- [Sage Attention](https://arxiv.org/abs/2410.02367)
|
||||
- Diverse hardware and OS support
|
||||
- Support H100, A100, 4090
|
||||
- Support Linux, Windows, MacOS
|
||||
- Cutting edge models
|
||||
- Wan2.1 T2V, I2V
|
||||
- HunyuanVideo
|
||||
- FastHunyuan: consistency distilled video diffusion models for 8x inference speedup.
|
||||
- StepVideo T2V
|
||||
- Distillation support
|
||||
- Recipes for video DiT, based on [PCM](https://github.com/G-U-N/Phased-Consistency-Model).
|
||||
- Support distilling/finetuning/inferencing state-of-the-art open video DiTs: 1. Mochi 2. Hunyuan.
|
||||
- Scalable training with FSDP, sequence parallelism, and selective activation checkpointing, with near linear scaling to 64 GPUs.
|
||||
- Memory efficient finetuning with LoRA, precomputed latent, and precomputed text embeddings.
|
||||
|
||||
## Getting Started
|
||||
We recommend using an environment manager such as `Conda` to create a clean environment:
|
||||
@@ -51,31 +52,17 @@ pip install fastvideo
|
||||
|
||||
Please see our [docs](https://hao-ai-lab.github.io/FastVideo/getting_started/installation.html) for more detailed installation instructions.
|
||||
|
||||
## Sparse Distillation
|
||||
For our sparse distillation techniques, please see our [distillation docs](https://hao-ai-lab.github.io/FastVideo/distillation/dmd.html) and check out our [blog](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/).
|
||||
|
||||
See below for recipes and datasets:
|
||||
|
||||
| Model | Sparse Distillation | Dataset |
|
||||
|:-------------------------------------------------------------------------------------------: |:---------------------------------------------------------------------------------------------------------------: |:--------------------------------------------------------------------------------------------------------: |
|
||||
| [FastWan2.1-T2V-1.3B](https://huggingface.co/FastVideo/FastWan2.1-T2V-1.3B-Diffusers) | [Recipe](https://github.com/hao-ai-lab/FastVideo/tree/main/examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P) | [FastVideo Synthetic Wan2.1 480P](https://huggingface.co/datasets/FastVideo/Wan-Syn_77x448x832_600k) |
|
||||
| [FastWan2.1-T2V-14B-Preview](https://huggingface.co/FastVideo/FastWan2.1-T2V-14B-Diffusers) | Coming soon! | [FastVideo Synthetic Wan2.1 720P](https://huggingface.co/datasets/FastVideo/Wan-Syn_77x768x1280_250k) |
|
||||
| [FastWan2.2-TI2V-5B](https://huggingface.co/FastVideo/FastWan2.2-TI2V-5B-Diffusers) | [Recipe](https://github.com/hao-ai-lab/FastVideo/tree/main/examples/distill/Wan2.2-TI2V-5B-Diffusers/Data-free) | [FastVideo Synthetic Wan2.2 720P](https://huggingface.co/datasets/FastVideo/Wan2.2-Syn-121x704x1280_32k) |
|
||||
|
||||
## Inference
|
||||
### Generating Your First Video
|
||||
Here's a minimal example to generate a video using the default settings. Make sure VSA kernels are [installed](https://hao-ai-lab.github.io/FastVideo/video_sparse_attention/installation.html). Create a file called `example.py` with the following code:
|
||||
Here's a minimal example to generate a video using the default settings. Create a file called `example.py` with the following code:
|
||||
|
||||
```python
|
||||
import os
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
def main():
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
|
||||
|
||||
# Create a video generator with a pre-trained model
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1, # Adjust based on your hardware
|
||||
)
|
||||
|
||||
@@ -108,42 +95,42 @@ For a more detailed guide, please see our [inference quick start](https://hao-ai
|
||||
- [Contribution Guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation.html)
|
||||
|
||||
## Distillation and Finetuning
|
||||
- [Distillation Guide](https://hao-ai-lab.github.io/FastVideo/distillation/dmd.html)
|
||||
<!-- - [Finetuning Guide](https://hao-ai-lab.github.io/FastVideo/training/finetune.html) -->
|
||||
- [Distillation Guide](https://hao-ai-lab.github.io/FastVideo/training/distillation.html)
|
||||
- [Finetuning Guide](https://hao-ai-lab.github.io/FastVideo/training/finetune.html)
|
||||
|
||||
## 📑 Development Plan
|
||||
|
||||
<!-- - More distillation methods -->
|
||||
<!-- - [ ] Add Distribution Matching Distillation -->
|
||||
More FastWan Models Coming Soon!
|
||||
- [ ] Add FastWan2.1-T2V-14B
|
||||
- [ ] Add FastWan2.2-T2V-14B
|
||||
- [ ] Add FastWan2.2-I2V-14B
|
||||
<!-- - Optimization features
|
||||
- Code updates -->
|
||||
- More models support
|
||||
<!-- - [ ] Add CogvideoX model -->
|
||||
- [x] Add StepVideo to V1
|
||||
- Optimization features
|
||||
- [x] Teacache in V1
|
||||
- [x] SageAttention in V1
|
||||
- Code updates
|
||||
- [x] V1 Configuration API
|
||||
- [ ] Support Training in V1
|
||||
<!-- - [ ] fp8 support -->
|
||||
<!-- - [ ] faster load model and save model support -->
|
||||
|
||||
See details in [development roadmap](https://github.com/hao-ai-lab/FastVideo/issues/468).
|
||||
|
||||
## 🤝 Contributing
|
||||
|
||||
We welcome all contributions. Please check out our guide [here](https://hao-ai-lab.github.io/FastVideo/contributing/overview.html)
|
||||
|
||||
## Acknowledgement
|
||||
We learned and reused code from the following projects:
|
||||
- [Wan-Video](https://github.com/Wan-Video)
|
||||
- [ThunderKittens](https://github.com/HazyResearch/ThunderKittens)
|
||||
- [Triton](https://github.com/triton-lang/triton)
|
||||
- [DMD2](https://github.com/tianweiy/DMD2)
|
||||
- [PCM](https://github.com/G-U-N/Phased-Consistency-Model)
|
||||
- [diffusers](https://github.com/huggingface/diffusers)
|
||||
- [OpenSoraPlan](https://github.com/PKU-YuanGroup/Open-Sora-Plan)
|
||||
- [xDiT](https://github.com/xdit-project/xDiT)
|
||||
- [vLLM](https://github.com/vllm-project/vllm)
|
||||
- [SGLang](https://github.com/sgl-project/sglang)
|
||||
|
||||
We thank [MBZUAI](https://ifm.mbzuai.ac.ae/), [Anyscale](https://www.anyscale.com/), and [GMI Cloud](https://www.gmicloud.ai/) for their support throughout this project.
|
||||
We thank MBZUAI and [Anyscale](https://www.anyscale.com/) for their support throughout this project.
|
||||
|
||||
## Citation
|
||||
If you find FastVideo useful, please considering citing our work:
|
||||
If you use FastVideo for your research, please cite our work:
|
||||
|
||||
```bibtex
|
||||
@software{fastvideo2024,
|
||||
@@ -154,17 +141,31 @@ If you find FastVideo useful, please considering citing our work:
|
||||
year = {2024},
|
||||
}
|
||||
|
||||
@article{zhang2025vsa,
|
||||
title={VSA: Faster Video Diffusion with Trainable Sparse Attention},
|
||||
author={Zhang, Peiyuan and Huang, Haofeng and Chen, Yongqi and Lin, Will and Liu, Zhengzhong and Stoica, Ion and Xing, Eric and Zhang, Hao},
|
||||
journal={arXiv preprint arXiv:2505.13389},
|
||||
year={2025}
|
||||
@misc{zhang2025vsafastervideodiffusion,
|
||||
title={VSA: Faster Video Diffusion with Trainable Sparse Attention},
|
||||
author={Peiyuan Zhang and Haofeng Huang and Yongqi Chen and Will Lin and Zhengzhong Liu and Ion Stoica and Eric Xing and Hao Zhang},
|
||||
year={2025},
|
||||
eprint={2505.13389},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CV},
|
||||
url={https://arxiv.org/abs/2505.13389},
|
||||
}
|
||||
|
||||
@article{zhang2025fast,
|
||||
title={Fast video generation with sliding tile attention},
|
||||
author={Zhang, Peiyuan and Chen, Yongqi and Su, Runlong and Ding, Hangliang and Stoica, Ion and Liu, Zhengzhong and Zhang, Hao},
|
||||
journal={arXiv preprint arXiv:2502.04507},
|
||||
year={2025}
|
||||
@misc{zhang2025fastvideogenerationsliding,
|
||||
title={Fast Video Generation with Sliding Tile Attention},
|
||||
author={Peiyuan Zhang and Yongqi Chen and Runlong Su and Hangliang Ding and Ion Stoica and Zhenghong Liu and Hao Zhang},
|
||||
year={2025},
|
||||
eprint={2502.04507},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CV},
|
||||
url={https://arxiv.org/abs/2502.04507},
|
||||
}
|
||||
@misc{ding2025efficientvditefficientvideodiffusion,
|
||||
title={Efficient-vDiT: Efficient Video Diffusion Transformers With Attention Tile},
|
||||
author={Hangliang Ding and Dacheng Li and Runlong Su and Peiyuan Zhang and Zhijie Deng and Ion Stoica and Hao Zhang},
|
||||
year={2025},
|
||||
eprint={2502.06155},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CV},
|
||||
url={https://arxiv.org/abs/2502.06155},
|
||||
}
|
||||
```
|
||||
|
||||
|
Before Width: | Height: | Size: 194 KiB |
|
After Width: | Height: | Size: 149 KiB |
@@ -1,6 +0,0 @@
|
||||
<svg width="160" height="93" viewBox="0 0 160 93" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M28.8511 91.66L57.6319 1.86368H64.5394L35.7585 91.66H28.8511Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
<path d="M15.0376 91.66L43.8185 1.86368H46.1209L17.3401 91.66H15.0376Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
<path d="M1.22217 91.66L30.003 1.86366H31.1543L2.3734 91.66H1.22217Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.15122"/>
|
||||
<path d="M71.4465 1.86483L42.666 91.6599H69.144L78.3538 58.2746H123.251L129.007 39.855H84.1099L89.866 22.5868H152.032L157.788 1.86483H71.4465Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 691 B |
@@ -1,18 +0,0 @@
|
||||
<svg width="252" height="105" viewBox="0 0 252 105" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M89.4843 55.5457H101.361L87.7028 101H74.638L89.4843 55.5457Z" fill="#356CFF"/>
|
||||
<path fill-rule="evenodd" clip-rule="evenodd" d="M96.0167 1.00057H112.645L118.583 48.273H104.924L103.737 39.7882H79.9827L67.5117 55.5457H85.3273L43.1638 101H28.3174L22.3789 55.5457H33.6621L38.4129 91.3031L58.604 68.2729H44.3515L96.0167 1.00057ZM100.768 13.1217L87.7028 29.4852H103.143L100.768 13.1217Z" fill="#356CFF"/>
|
||||
<path d="M37.2252 1.00057L22.3789 48.273H36.0375L40.7884 30.6974L62.6727 30.6974L69.6727 21.0005L43.7576 21.0004L46.7269 11.9096L77.6727 11.9096L86 1.00057L37.2252 1.00057Z" fill="#356CFF"/>
|
||||
<path fill-rule="evenodd" clip-rule="evenodd" d="M108.488 55.5457L94.2351 101C94.2351 101 105.518 101 120.959 101C136.399 101 144.078 93.0133 148.276 79.788C152.432 68.0157 153.027 55.5457 136.399 55.5457C119.771 55.5457 108.488 55.5457 108.488 55.5457ZM109.081 90.697L116.802 65.8487C116.802 65.8487 120.959 65.8487 132.242 65.8487C143.525 65.8487 137.586 78.5759 135.211 84.0304C133.307 88.4021 127.491 90.697 122.74 90.697C117.989 90.697 109.081 90.697 109.081 90.697Z" fill="#356CFF"/>
|
||||
<path d="M173.188 1.00056L168.625 11.9096C168.625 11.9096 149.386 11.9092 142.525 11.9095C135.664 11.9098 136.586 20.3944 141.337 20.3944H159.747C168.654 20.3944 166.961 33.6899 163.904 38.5761C160.467 44.0675 157.371 48.273 148.463 48.273L125.188 48.273L124 37.97L147.87 37.97C153.808 37.97 156.184 29.4852 151.433 29.4852H131.836C120.142 29.4852 125.897 1.00043 141.337 1.00043L173.188 1.00056Z" fill="#356CFF"/>
|
||||
<path d="M179.938 1.00056L175.688 11.9096L191.221 11.9096L179.938 48.273H192.409L203.692 11.9096L219.132 11.9095L223.289 1.00043L179.938 1.00056Z" fill="#356CFF"/>
|
||||
<path d="M161.341 55.5457H202.845L198.5 65.8487H169.654L167.279 73.7268H188.5L184.749 82.8177H164.31L161.934 90.697H190.251L186.624 101H146.494L161.341 55.5457Z" fill="#356CFF"/>
|
||||
<path fill-rule="evenodd" clip-rule="evenodd" d="M230.821 54.9391C255.169 54.9391 251.776 67.0602 249.231 77.9692C246.686 88.8783 240.917 101 217.757 101C194.596 101 195.606 88.8783 199.347 77.9692C203.089 67.0602 206.473 54.9391 230.821 54.9391ZM237.948 77.9692C239.984 70.6965 240.917 65.242 228.446 65.242C215.975 65.242 211.818 71.9087 210.037 77.9692C208.255 84.0298 208.255 91.3025 219.538 91.3025C230.821 91.3025 235.911 85.2419 237.948 77.9692Z" fill="#356CFF"/>
|
||||
<path d="M173.188 1.00056L168.625 11.9096C168.625 11.9096 149.386 11.9092 142.525 11.9095C135.664 11.9098 136.586 20.3944 141.337 20.3944M173.188 1.00056C173.188 1.00056 156.777 1.00043 141.337 1.00043M173.188 1.00056L141.337 1.00043M141.337 20.3944C146.088 20.3944 150.839 20.3944 159.747 20.3944M141.337 20.3944H159.747M159.747 20.3944C168.654 20.3944 166.961 33.6899 163.904 38.5761C160.467 44.0675 157.371 48.273 148.463 48.273M148.463 48.273C139.556 48.273 125.188 48.273 125.188 48.273M148.463 48.273L125.188 48.273M125.188 48.273L124 37.97M124 37.97C124 37.97 141.931 37.97 147.87 37.97M124 37.97L147.87 37.97M147.87 37.97C153.808 37.97 156.184 29.4852 151.433 29.4852M151.433 29.4852C146.682 29.4852 138.962 29.4852 131.836 29.4852M151.433 29.4852H131.836M131.836 29.4852C120.142 29.4852 125.897 1.00043 141.337 1.00043M37.2252 1.00057L22.3789 48.273H36.0375L40.7884 30.6974L62.6727 30.6974L69.6727 21.0005L43.7576 21.0004L46.7269 11.9096L77.6727 11.9096L86 1.00057L37.2252 1.00057ZM96.0167 1.00057H112.645L118.583 48.273H104.924L103.737 39.7882H79.9827L67.5117 55.5457H85.3273L43.1638 101H28.3174L22.3789 55.5457H33.6621L38.4129 91.3031L58.604 68.2729H44.3515L96.0167 1.00057ZM87.7028 29.4852L100.768 13.1217L103.143 29.4852H87.7028ZM89.4843 55.5457H101.361L87.7028 101H74.638L89.4843 55.5457ZM108.488 55.5457L94.2351 101C94.2351 101 105.518 101 120.959 101C136.399 101 144.078 93.0133 148.276 79.788C152.432 68.0157 153.027 55.5457 136.399 55.5457C119.771 55.5457 108.488 55.5457 108.488 55.5457ZM116.802 65.8487L109.081 90.697C109.081 90.697 117.989 90.697 122.74 90.697C127.491 90.697 133.307 88.4021 135.211 84.0304C137.586 78.5759 143.525 65.8487 132.242 65.8487C120.959 65.8487 116.802 65.8487 116.802 65.8487ZM179.938 1.00056L175.688 11.9096L191.221 11.9096L179.938 48.273H192.409L203.692 11.9096L219.132 11.9095L223.289 1.00043L179.938 1.00056ZM161.341 55.5457H202.845L198.5 65.8487H169.654L167.279 73.7268H188.5L184.749 82.8177H164.31L161.934 90.697H190.251L186.624 101H146.494L161.341 55.5457ZM230.821 54.9391C255.169 54.9391 251.776 67.0602 249.231 77.9692C246.686 88.8783 240.917 101 217.757 101C194.596 101 195.606 88.8783 199.347 77.9692C203.089 67.0602 206.473 54.9391 230.821 54.9391ZM228.446 65.242C240.917 65.242 239.984 70.6965 237.948 77.9692C235.911 85.2419 230.821 91.3025 219.538 91.3025C208.255 91.3025 208.255 84.0298 210.037 77.9692C211.818 71.9087 215.975 65.242 228.446 65.242Z" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M15.2524 55.5451L21.191 100.999L24.7541 100.999L18.8156 55.5451L15.2524 55.5451Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M8.12646 55.5451L14.065 100.999L15.2527 100.999L9.31417 55.5451L8.12646 55.5451Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M1 55.5451L6.93853 100.999L7.53239 100.999L1.59385 55.5451L1 55.5451Z" fill="#356CFF" stroke="#356CFF" stroke-width="0.593853"/>
|
||||
<path d="M15.2524 48.2724L30.0988 1H33.6619L18.8156 48.2724H15.2524Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M8.12646 48.2724L22.9728 1H24.1605L9.31417 48.2724H8.12646Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M1 48.2724L15.8463 1H16.4402L1.59385 48.2724H1Z" fill="#356CFF" stroke="#356CFF" stroke-width="0.593853"/>
|
||||
<path d="M85.3271 55.5457H67.5116L87 12.7363L44.3513 68.2729H58.6038L43.1636 101L85.3271 55.5457Z" fill="#FDC717" stroke="#FDC717" stroke-width="1.18771" stroke-miterlimit="16"/>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 5.7 KiB |
@@ -1,6 +0,0 @@
|
||||
<svg width="160" height="93" viewBox="0 0 160 93" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M28.8511 91.66L57.6319 1.86368H64.5394L35.7585 91.66H28.8511Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
<path d="M15.0376 91.66L43.8185 1.86368H46.1209L17.3401 91.66H15.0376Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
<path d="M1.22217 91.66L30.003 1.86366H31.1543L2.3734 91.66H1.22217Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.15122"/>
|
||||
<path d="M71.4465 1.86483L42.666 91.6599H69.144L78.3538 58.2746H123.251L129.007 39.855H84.1099L89.866 22.5868H152.032L157.788 1.86483H71.4465Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 691 B |
|
After Width: | Height: | Size: 31 KiB |
@@ -84,7 +84,7 @@ out = sliding_tile_attention(q, k, v, window_size, 0, False)
|
||||
### Test
|
||||
```bash
|
||||
python tests/test_sta.py # test STA
|
||||
python tests/test_vsa.py # test VSA
|
||||
python tests/test_block_sparse.py # test VSA
|
||||
```
|
||||
### Benchmark
|
||||
```bash
|
||||
|
||||
@@ -86,6 +86,8 @@ def benchmark_attention(configurations):
|
||||
# print(f"Average TFLOPS: {tflops_bwd}")
|
||||
# print("=" * 60)
|
||||
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
return results
|
||||
|
||||
|
||||
|
||||
@@ -51,16 +51,19 @@ for k in kernels:
|
||||
source_files.append(sources[k]['source_files'][target])
|
||||
cpp_flags.append(f'-DTK_COMPILE_{k.replace(" ", "_").upper()}')
|
||||
|
||||
|
||||
ext_modules = [
|
||||
CUDAExtension('vsa_cuda',
|
||||
sources=source_files,
|
||||
extra_compile_args={
|
||||
'cxx': cpp_flags,
|
||||
'nvcc': cuda_flags
|
||||
},
|
||||
libraries=['cuda'])
|
||||
]
|
||||
ext_modules = []
|
||||
import torch
|
||||
major, minor = torch.cuda.get_device_capability(0)
|
||||
if major == 9 and minor == 0:# check if H100
|
||||
ext_modules = [
|
||||
CUDAExtension('vsa_cuda',
|
||||
sources=source_files,
|
||||
extra_compile_args={
|
||||
'cxx': cpp_flags,
|
||||
'nvcc': cuda_flags
|
||||
},
|
||||
libraries=['cuda'])
|
||||
]
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -13,9 +13,9 @@ BLOCK_M = 64
|
||||
BLOCK_N = 64
|
||||
|
||||
def pytorch_test(Q, K, V, block_sparse_mask, dO):
|
||||
q_ = Q.clone().float().requires_grad_()
|
||||
k_ = K.clone().float().requires_grad_()
|
||||
v_ = V.clone().float().requires_grad_()
|
||||
q_ = Q.clone().requires_grad_()
|
||||
k_ = K.clone().requires_grad_()
|
||||
v_ = V.clone().requires_grad_()
|
||||
|
||||
QK = torch.matmul(q_, k_.transpose(-2, -1))
|
||||
QK /= (q_.size(-1) ** 0.5)
|
||||
@@ -35,9 +35,9 @@ def pytorch_test(Q, K, V, block_sparse_mask, dO):
|
||||
|
||||
|
||||
def block_sparse_kernel_test(Q, K, V, block_sparse_mask, variable_block_sizes, non_pad_index, dO):
|
||||
Q = Q.detach().requires_grad_()
|
||||
K = K.detach().requires_grad_()
|
||||
V = V.detach().requires_grad_()
|
||||
Q = Q.clone().requires_grad_()
|
||||
K = K.clone().requires_grad_()
|
||||
V = V.clone().requires_grad_()
|
||||
|
||||
q_padded = vsa_pad(Q, non_pad_index, variable_block_sizes.shape[0], BLOCK_M)
|
||||
k_padded = vsa_pad(K, non_pad_index, variable_block_sizes.shape[0], BLOCK_M)
|
||||
@@ -60,9 +60,11 @@ def get_non_pad_index(
|
||||
|
||||
return index_pad[index_mask]
|
||||
|
||||
def generate_tensor(shape, dtype, device):
|
||||
def generate_tensor(shape, mean, std, dtype, device):
|
||||
tensor = torch.randn(shape, dtype=dtype, device=device)
|
||||
return tensor
|
||||
magnitude = torch.norm(tensor, dim=-1, keepdim=True)
|
||||
scaled_tensor = tensor * (torch.randn(magnitude.shape, dtype=dtype, device=device) * std + mean) / magnitude
|
||||
return scaled_tensor.contiguous()
|
||||
|
||||
def generate_variable_block_sizes(num_blocks, min_size=32, max_size=64, device="cuda"):
|
||||
return torch.randint(min_size, max_size + 1, (num_blocks,), device=device, dtype=torch.int32)
|
||||
@@ -73,7 +75,7 @@ def vsa_pad(x, non_pad_index, num_blocks, block_size):
|
||||
padded_x[:, :, non_pad_index, :] = x
|
||||
return padded_x
|
||||
|
||||
def check_correctness(h, d, num_blocks, k, num_iterations=20, error_mode='all'):
|
||||
def check_correctness(h, d, num_blocks, k, mean, std, num_iterations=20, error_mode='all'):
|
||||
results = {
|
||||
'gO': {'sum_diff': 0.0, 'sum_abs': 0.0, 'max_diff': 0.0},
|
||||
'gQ': {'sum_diff': 0.0, 'sum_abs': 0.0, 'max_diff': 0.0},
|
||||
@@ -89,10 +91,10 @@ def check_correctness(h, d, num_blocks, k, num_iterations=20, error_mode='all')
|
||||
block_mask = generate_block_sparse_mask_for_function(h, num_blocks, k, device)
|
||||
full_mask = create_full_mask_from_block_mask(block_mask, variable_block_sizes, device)
|
||||
for _ in range(num_iterations):
|
||||
Q = generate_tensor((1, h, S, d), torch.bfloat16, device)
|
||||
K = generate_tensor((1, h, S, d), torch.bfloat16, device)
|
||||
V = generate_tensor((1, h, S, d), torch.bfloat16, device)
|
||||
dO = generate_tensor((1, h, S, d), torch.bfloat16, device)
|
||||
Q = generate_tensor((1, h, S, d), mean, std, torch.bfloat16, device)
|
||||
K = generate_tensor((1, h, S, d), mean, std, torch.bfloat16, device)
|
||||
V = generate_tensor((1, h, S, d), mean, std, torch.bfloat16, device)
|
||||
dO = generate_tensor((1, h, S, d), mean, std, torch.bfloat16, device)
|
||||
|
||||
# dO_padded = torch.zeros_like(dO_padded)
|
||||
# dO_padded[:, :, non_pad_index, :] = dO
|
||||
@@ -105,8 +107,7 @@ def check_correctness(h, d, num_blocks, k, num_iterations=20, error_mode='all')
|
||||
abs_diff = torch.abs(diff)
|
||||
results[name]['sum_diff'] += torch.sum(abs_diff).item()
|
||||
results[name]['sum_abs'] += torch.sum(torch.abs(pt)).item()
|
||||
rel_max_diff = torch.max(abs_diff) / torch.mean(torch.abs(pt))
|
||||
results[name]['max_diff'] = max(results[name]['max_diff'], rel_max_diff.item())
|
||||
results[name]['max_diff'] = max(results[name]['max_diff'], torch.max(abs_diff).item())
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
@@ -118,27 +119,27 @@ def check_correctness(h, d, num_blocks, k, num_iterations=20, error_mode='all')
|
||||
|
||||
return results
|
||||
|
||||
def generate_error_graphs(h, d, error_mode='all'):
|
||||
def generate_error_graphs(h, d, mean, std, error_mode='all'):
|
||||
test_configs = [
|
||||
{"num_blocks": 16, "k": 2, "description": "Small sequence"},
|
||||
{"num_blocks": 32, "k": 4, "description": "Medium sequence"},
|
||||
{"num_blocks": 53, "k": 6, "description": "Large sequence"},
|
||||
]
|
||||
|
||||
print(f"\nError Analysis for h={h}, d={d}, mode={error_mode}")
|
||||
print(f"\nError Analysis for h={h}, d={d}, mean={mean}, std={std}, mode={error_mode}")
|
||||
print("=" * 150)
|
||||
print(f"{'Config':<20} {'Blocks':<8} {'K':<4} "
|
||||
f"{'gQ Avg':<12} {'Rel gQ Max':<12} "
|
||||
f"{'gK Avg':<12} {'Rel gK Max':<12} "
|
||||
f"{'gV Avg':<12} {'Rel gV Max':<12} "
|
||||
f"{'gO Avg':<12} {'Rel gO Max':<12}")
|
||||
f"{'gQ Avg':<12} {'gQ Max':<12} "
|
||||
f"{'gK Avg':<12} {'gK Max':<12} "
|
||||
f"{'gV Avg':<12} {'gV Max':<12} "
|
||||
f"{'gO Avg':<12} {'gO Max':<12}")
|
||||
print("-" * 150)
|
||||
|
||||
for config in test_configs:
|
||||
num_blocks = config["num_blocks"]
|
||||
k = config["k"]
|
||||
description = config["description"]
|
||||
results = check_correctness(h, d, num_blocks, k, error_mode=error_mode)
|
||||
results = check_correctness(h, d, num_blocks, k, mean, std, error_mode=error_mode)
|
||||
print(f"{description:<20} {num_blocks:<8} {k:<4} "
|
||||
f"{results['gQ']['avg_diff']:<12.6e} {results['gQ']['max_diff']:<12.6e} "
|
||||
f"{results['gK']['avg_diff']:<12.6e} {results['gK']['max_diff']:<12.6e} "
|
||||
@@ -149,8 +150,10 @@ def generate_error_graphs(h, d, error_mode='all'):
|
||||
|
||||
if __name__ == "__main__":
|
||||
h, d = 16, 128
|
||||
mean = 0.0
|
||||
std = 1
|
||||
print("Block Sparse Attention with Variable Block Sizes Analysis")
|
||||
print("=" * 60)
|
||||
for mode in ['backward']:
|
||||
generate_error_graphs(h, d, error_mode=mode)
|
||||
generate_error_graphs(h, d, mean, std, error_mode=mode)
|
||||
print("\nAnalysis completed for all modes.")
|
||||
|
||||
@@ -1,13 +1,12 @@
|
||||
import torch
|
||||
from typing import Tuple
|
||||
block_sparse_attn=None
|
||||
import torch
|
||||
major, minor = torch.cuda.get_device_capability(0)
|
||||
if major == 9 and minor == 0:# check if H100
|
||||
|
||||
try:
|
||||
from vsa_cuda import block_sparse_fwd, block_sparse_bwd
|
||||
from vsa.block_sparse_wrapper import block_sparse_attn_SM90
|
||||
block_sparse_attn = block_sparse_attn_SM90
|
||||
else:
|
||||
except ImportError:
|
||||
from vsa.block_sparse_wrapper import block_sparse_attn_triton
|
||||
block_sparse_fwd = None
|
||||
block_sparse_bwd = None
|
||||
|
||||
@@ -568,6 +568,7 @@ void bwd_attend_ker(const __grid_constant__ bwd_globals<D> g) {
|
||||
__syncthreads(); // wait for sd_smem shared memory write
|
||||
warpgroup::mm_AtB(qg_reg, ds_smem_t[0], k_smem[0]); //delat dQ = dSK
|
||||
warpgroup::mma_commit_group();
|
||||
tma::store_async_wait();
|
||||
warpgroup::mma_async_wait();
|
||||
// store qg to shared memory
|
||||
warpgroup::store(qg_smem, qg_reg);
|
||||
@@ -577,7 +578,6 @@ void bwd_attend_ker(const __grid_constant__ bwd_globals<D> g) {
|
||||
if (threadIdx.x / 32 == 0) {
|
||||
coord<qg_tile> tile_idx = {blockIdx.z, blockIdx.y, store_qg_block_index, 0};
|
||||
tma::store_add_async(g.qg, qg_smem, tile_idx);
|
||||
tma::store_async_wait();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -624,6 +624,7 @@ void bwd_attend_ker(const __grid_constant__ bwd_globals<D> g) {
|
||||
__syncthreads(); // wait for sd_smem shared memory write
|
||||
warpgroup::mm_AtB(qg_reg, ds_smem_t[0], k_smem[0]); //delat dQ = dSK
|
||||
warpgroup::mma_commit_group();
|
||||
tma::store_async_wait();
|
||||
warpgroup::mma_async_wait();
|
||||
// store qg to shared memory
|
||||
warpgroup::store(qg_smem, qg_reg);
|
||||
@@ -633,14 +634,13 @@ void bwd_attend_ker(const __grid_constant__ bwd_globals<D> g) {
|
||||
if (threadIdx.x / 32 == 0) {
|
||||
coord<qg_tile> tile_idx = {blockIdx.z, blockIdx.y, store_qg_block_index, 0};
|
||||
tma::store_add_async(g.qg, qg_smem, tile_idx);
|
||||
tma::store_async_wait();
|
||||
}
|
||||
}
|
||||
|
||||
// store kq and vq
|
||||
|
||||
// ! the following two line seems unnecessary.
|
||||
// tma::store_async_wait(); // ensure qg is finished
|
||||
tma::store_async_wait(); // ensure qg is finished
|
||||
__syncthreads();
|
||||
|
||||
warpgroup::store(kg_smem[0], kg_reg);
|
||||
@@ -1174,4 +1174,4 @@ block_sparse_attention_backward(torch::Tensor q,
|
||||
|
||||
return {qg, kg, vg};
|
||||
//cudadevicesynchronize();
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,95 @@ from typing import Tuple, Optional
|
||||
|
||||
|
||||
|
||||
@torch.library.custom_op("vsa::block_sparse_attn_SM90", mutates_args=(), device_types="cuda")
|
||||
def block_sparse_attn_SM90(
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
)-> Tuple[torch.Tensor, torch.Tensor]:
|
||||
q_padded = q_padded.contiguous()
|
||||
k_padded = k_padded.contiguous()
|
||||
v_padded = v_padded.contiguous()
|
||||
q2k_block_sparse_index, q2k_block_sparse_num = map_to_index(block_map)
|
||||
variable_block_sizes = variable_block_sizes.int()
|
||||
o_padded, lse_padded = block_sparse_fwd(q_padded, k_padded, v_padded, q2k_block_sparse_index, q2k_block_sparse_num, variable_block_sizes)
|
||||
return o_padded, lse_padded
|
||||
|
||||
|
||||
|
||||
|
||||
@torch.library.register_fake("vsa::block_sparse_attn_SM90")
|
||||
def _block_sparse_attn_SM90_fake(
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
q_padded, k_padded, v_padded = [x.contiguous() for x in (q_padded, k_padded, v_padded)]
|
||||
B, H, S, D = q_padded.shape
|
||||
o_padded = torch.empty_like(q_padded)
|
||||
lse_padded = torch.empty((B, H, S, 1), device=q_padded.device, dtype=torch.float32)
|
||||
return o_padded, lse_padded
|
||||
|
||||
|
||||
@torch.library.custom_op("vsa::block_sparse_attn_backward_SM90", mutates_args=(), device_types="cuda")
|
||||
def block_sparse_attn_backward_SM90(
|
||||
grad_output_padded: torch.Tensor,
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
o_padded: torch.Tensor,
|
||||
lse_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
)-> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
grad_output_padded = grad_output_padded.contiguous()
|
||||
k2q_block_sparse_index, k2q_block_sparse_num = map_to_index(block_map.transpose(-1, -2))
|
||||
grad_q_padded, grad_k_padded, grad_v_padded = block_sparse_bwd(
|
||||
q_padded, k_padded, v_padded, o_padded, lse_padded, grad_output_padded, k2q_block_sparse_index, k2q_block_sparse_num, variable_block_sizes
|
||||
)
|
||||
grad_q_padded = grad_q_padded.to(grad_output_padded.dtype)
|
||||
grad_k_padded = grad_k_padded.to(grad_output_padded.dtype)
|
||||
grad_v_padded = grad_v_padded.to(grad_output_padded.dtype)
|
||||
return grad_q_padded, grad_k_padded, grad_v_padded
|
||||
|
||||
@torch.library.register_fake("vsa::block_sparse_attn_backward_SM90")
|
||||
def _block_sparse_attn_backward_SM90_fake(
|
||||
grad_output_padded: torch.Tensor,
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
o_padded: torch.Tensor,
|
||||
lse_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
torch._check(grad_output_padded.dtype == torch.bfloat16)
|
||||
torch._check(lse_padded.dtype == torch.float32)
|
||||
grad_output_padded = grad_output_padded.contiguous()
|
||||
dq = torch.empty_like(grad_output_padded)
|
||||
dk = torch.empty_like(grad_output_padded)
|
||||
dv = torch.empty_like(grad_output_padded)
|
||||
return dq, dk, dv
|
||||
|
||||
|
||||
def backward_SM90(ctx, grad_output1, grad_output2):
|
||||
q_padded, k_padded, v_padded, o_padded, lse_padded, block_map, variable_block_sizes= ctx.saved_tensors
|
||||
dq, dk, dv = block_sparse_attn_backward_SM90(grad_output1, q_padded, k_padded, v_padded, o_padded, lse_padded, block_map, variable_block_sizes)
|
||||
return dq, dk, dv, None, None
|
||||
|
||||
def setup_context_SM90(ctx, inputs, output):
|
||||
q_padded, k_padded, v_padded, block_map, variable_block_sizes = inputs
|
||||
o_padded, lse_padded = output
|
||||
ctx.save_for_backward(q_padded, k_padded, v_padded, o_padded, lse_padded, block_map, variable_block_sizes)
|
||||
|
||||
|
||||
block_sparse_attn_SM90.register_autograd(backward_SM90, setup_context=setup_context_SM90)
|
||||
|
||||
|
||||
@torch.library.custom_op("vsa::block_sparse_attn_triton", mutates_args=(), device_types="cuda")
|
||||
def block_sparse_attn_triton(
|
||||
q: torch.Tensor,
|
||||
@@ -90,96 +179,4 @@ def setup_context_triton(ctx, inputs, output):
|
||||
o_padded, M = output
|
||||
ctx.save_for_backward(q_padded, k_padded, v_padded, o_padded, M, block_map, variable_block_sizes)
|
||||
|
||||
block_sparse_attn_triton.register_autograd(backward_triton, setup_context=setup_context_triton)
|
||||
|
||||
|
||||
major, minor = torch.cuda.get_device_capability(0)
|
||||
|
||||
if major == 9 and minor == 0:# check if H100
|
||||
@torch.library.custom_op("vsa::block_sparse_attn_SM90", mutates_args=(), device_types="cuda")
|
||||
def block_sparse_attn_SM90(
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
)-> Tuple[torch.Tensor, torch.Tensor]:
|
||||
q_padded = q_padded.contiguous()
|
||||
k_padded = k_padded.contiguous()
|
||||
v_padded = v_padded.contiguous()
|
||||
q2k_block_sparse_index, q2k_block_sparse_num = map_to_index(block_map)
|
||||
variable_block_sizes = variable_block_sizes.int()
|
||||
o_padded, lse_padded = block_sparse_fwd(q_padded, k_padded, v_padded, q2k_block_sparse_index, q2k_block_sparse_num, variable_block_sizes)
|
||||
return o_padded, lse_padded
|
||||
|
||||
|
||||
|
||||
|
||||
@torch.library.register_fake("vsa::block_sparse_attn_SM90")
|
||||
def _block_sparse_attn_SM90_fake(
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
q_padded, k_padded, v_padded = [x.contiguous() for x in (q_padded, k_padded, v_padded)]
|
||||
B, H, S, D = q_padded.shape
|
||||
o_padded = torch.empty_like(q_padded)
|
||||
lse_padded = torch.empty((B, H, S, 1), device=q_padded.device, dtype=torch.float32)
|
||||
return o_padded, lse_padded
|
||||
|
||||
|
||||
@torch.library.custom_op("vsa::block_sparse_attn_backward_SM90", mutates_args=(), device_types="cuda")
|
||||
def block_sparse_attn_backward_SM90(
|
||||
grad_output_padded: torch.Tensor,
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
o_padded: torch.Tensor,
|
||||
lse_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
)-> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
grad_output_padded = grad_output_padded.contiguous()
|
||||
k2q_block_sparse_index, k2q_block_sparse_num = map_to_index(block_map.transpose(-1, -2))
|
||||
grad_q_padded, grad_k_padded, grad_v_padded = block_sparse_bwd(
|
||||
q_padded, k_padded, v_padded, o_padded, lse_padded, grad_output_padded, k2q_block_sparse_index, k2q_block_sparse_num, variable_block_sizes
|
||||
)
|
||||
grad_q_padded = grad_q_padded.to(grad_output_padded.dtype)
|
||||
grad_k_padded = grad_k_padded.to(grad_output_padded.dtype)
|
||||
grad_v_padded = grad_v_padded.to(grad_output_padded.dtype)
|
||||
return grad_q_padded, grad_k_padded, grad_v_padded
|
||||
|
||||
@torch.library.register_fake("vsa::block_sparse_attn_backward_SM90")
|
||||
def _block_sparse_attn_backward_SM90_fake(
|
||||
grad_output_padded: torch.Tensor,
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
o_padded: torch.Tensor,
|
||||
lse_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
torch._check(grad_output_padded.dtype == torch.bfloat16)
|
||||
torch._check(lse_padded.dtype == torch.float32)
|
||||
grad_output_padded = grad_output_padded.contiguous()
|
||||
dq = torch.empty_like(grad_output_padded)
|
||||
dk = torch.empty_like(grad_output_padded)
|
||||
dv = torch.empty_like(grad_output_padded)
|
||||
return dq, dk, dv
|
||||
|
||||
|
||||
def backward_SM90(ctx, grad_output1, grad_output2):
|
||||
q_padded, k_padded, v_padded, o_padded, lse_padded, block_map, variable_block_sizes= ctx.saved_tensors
|
||||
dq, dk, dv = block_sparse_attn_backward_SM90(grad_output1, q_padded, k_padded, v_padded, o_padded, lse_padded, block_map, variable_block_sizes)
|
||||
return dq, dk, dv, None, None
|
||||
|
||||
def setup_context_SM90(ctx, inputs, output):
|
||||
q_padded, k_padded, v_padded, block_map, variable_block_sizes = inputs
|
||||
o_padded, lse_padded = output
|
||||
ctx.save_for_backward(q_padded, k_padded, v_padded, o_padded, lse_padded, block_map, variable_block_sizes)
|
||||
|
||||
|
||||
block_sparse_attn_SM90.register_autograd(backward_SM90, setup_context=setup_context_SM90)
|
||||
block_sparse_attn_triton.register_autograd(backward_triton, setup_context=setup_context_triton)
|
||||
@@ -1,9 +1,7 @@
|
||||
FROM nvidia/cuda:12.8.0-devel-ubuntu22.04
|
||||
FROM nvidia/cuda:12.4.1-devel-ubuntu20.04
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
WORKDIR /FastVideo
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
@@ -11,25 +9,17 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
git \
|
||||
ca-certificates \
|
||||
openssh-server \
|
||||
zsh \
|
||||
vim \
|
||||
curl \
|
||||
gcc-11 \
|
||||
g++-11 \
|
||||
clang-11 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Set up C++20 compilers for ThunderKittens
|
||||
RUN update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
RUN wget https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh && \
|
||||
bash Miniconda3-latest-Linux-x86_64.sh -b -p /opt/conda && \
|
||||
rm Miniconda3-latest-Linux-x86_64.sh
|
||||
|
||||
# Set CUDA environment variables
|
||||
ENV CUDA_HOME=/usr/local/cuda-12.8
|
||||
ENV PATH=${CUDA_HOME}/bin:${PATH}
|
||||
ENV LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
ENV PATH=/opt/conda/bin:$PATH
|
||||
|
||||
# Install uv and source its environment
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh && \
|
||||
echo 'source $HOME/.local/bin/env' >> /root/.bashrc
|
||||
RUN conda create --name fastvideo-dev python=3.10.0 -y
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
# Copy just the pyproject.toml first to leverage Docker cache
|
||||
COPY pyproject.toml ./
|
||||
@@ -37,36 +27,22 @@ COPY pyproject.toml ./
|
||||
# Create a dummy README to satisfy the installation
|
||||
RUN echo "# Placeholder" > README.md
|
||||
|
||||
# Create and activate virtual environment with specific Python version and seed
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
uv venv --python 3.10 --seed /opt/venv && \
|
||||
source /opt/venv/bin/activate && \
|
||||
uv pip install --no-cache-dir --upgrade pip && \
|
||||
uv pip install --no-cache-dir .[dev] && \
|
||||
uv pip install --no-cache-dir flash-attn==2.8.3 --no-build-isolation
|
||||
RUN conda run -n fastvideo-dev pip install --no-cache-dir --upgrade pip && \
|
||||
conda run -n fastvideo-dev pip install --no-cache-dir .[dev] && \
|
||||
conda run -n fastvideo-dev pip install --no-cache-dir flash-attn==2.7.4.post1 --no-build-isolation && \
|
||||
conda clean -afy
|
||||
|
||||
COPY . .
|
||||
|
||||
# Install dependencies using uv and set up shell configuration
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
uv pip install --no-cache-dir -e .[dev] && \
|
||||
git config --unset-all http.https://github.com/.extraheader || true && \
|
||||
echo 'source /opt/venv/bin/activate' >> /root/.bashrc && \
|
||||
echo 'if [ -n "$ZSH_VERSION" ] && [ -f ~/.zshrc ]; then . ~/.zshrc; elif [ -f ~/.bashrc ]; then . ~/.bashrc; fi' > /root/.profile
|
||||
RUN conda run -n fastvideo-dev pip install --no-cache-dir -e .[dev]
|
||||
|
||||
# Install STA (Sliding Tile Attention)
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
cd csrc/attn && \
|
||||
git submodule update --init --recursive && \
|
||||
python setup_sta.py install
|
||||
# Remove authentication headers
|
||||
RUN git config --unset-all http.https://github.com/.extraheader || true
|
||||
|
||||
# Install VSA
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
cd csrc/attn && \
|
||||
git submodule update --init --recursive && \
|
||||
python setup_vsa.py install
|
||||
# Set up automatic conda environment activation for all shells
|
||||
RUN echo 'source /opt/conda/etc/profile.d/conda.sh' >> /root/.bashrc && \
|
||||
echo 'conda activate fastvideo-dev' >> /root/.bashrc && \
|
||||
# Ensure .bashrc is sourced for SSH login shells
|
||||
echo 'if [ -f ~/.bashrc ]; then . ~/.bashrc; fi' > /root/.profile
|
||||
|
||||
EXPOSE 22
|
||||
@@ -1,9 +1,7 @@
|
||||
FROM nvidia/cuda:12.8.0-devel-ubuntu22.04
|
||||
FROM nvidia/cuda:12.4.1-devel-ubuntu20.04
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
WORKDIR /FastVideo
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
@@ -11,25 +9,17 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
git \
|
||||
ca-certificates \
|
||||
openssh-server \
|
||||
zsh \
|
||||
vim \
|
||||
curl \
|
||||
gcc-11 \
|
||||
g++-11 \
|
||||
clang-11 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Set up C++20 compilers for ThunderKittens
|
||||
RUN update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
RUN wget https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh && \
|
||||
bash Miniconda3-latest-Linux-x86_64.sh -b -p /opt/conda && \
|
||||
rm Miniconda3-latest-Linux-x86_64.sh
|
||||
|
||||
# Set CUDA environment variables
|
||||
ENV CUDA_HOME=/usr/local/cuda-12.8
|
||||
ENV PATH=${CUDA_HOME}/bin:${PATH}
|
||||
ENV LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
ENV PATH=/opt/conda/bin:$PATH
|
||||
|
||||
# Install uv and source its environment
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh && \
|
||||
echo 'source $HOME/.local/bin/env' >> /root/.bashrc
|
||||
RUN conda create --name fastvideo-dev python=3.11.11 -y
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
# Copy just the pyproject.toml first to leverage Docker cache
|
||||
COPY pyproject.toml ./
|
||||
@@ -37,36 +27,22 @@ COPY pyproject.toml ./
|
||||
# Create a dummy README to satisfy the installation
|
||||
RUN echo "# Placeholder" > README.md
|
||||
|
||||
# Create and activate virtual environment with specific Python version and seed
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
uv venv --python 3.11 --seed /opt/venv && \
|
||||
source /opt/venv/bin/activate && \
|
||||
uv pip install --no-cache-dir --upgrade pip && \
|
||||
uv pip install --no-cache-dir .[dev] && \
|
||||
uv pip install --no-cache-dir flash-attn==2.8.3 --no-build-isolation
|
||||
RUN conda run -n fastvideo-dev pip install --no-cache-dir --upgrade pip && \
|
||||
conda run -n fastvideo-dev pip install --no-cache-dir .[dev] && \
|
||||
conda run -n fastvideo-dev pip install --no-cache-dir flash-attn==2.7.4.post1 --no-build-isolation && \
|
||||
conda clean -afy
|
||||
|
||||
COPY . .
|
||||
|
||||
# Install dependencies using uv and set up shell configuration
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
uv pip install --no-cache-dir -e .[dev] && \
|
||||
git config --unset-all http.https://github.com/.extraheader || true && \
|
||||
echo 'source /opt/venv/bin/activate' >> /root/.bashrc && \
|
||||
echo 'if [ -n "$ZSH_VERSION" ] && [ -f ~/.zshrc ]; then . ~/.zshrc; elif [ -f ~/.bashrc ]; then . ~/.bashrc; fi' > /root/.profile
|
||||
RUN conda run -n fastvideo-dev pip install --no-cache-dir -e .[dev]
|
||||
|
||||
# Install STA (Sliding Tile Attention)
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
cd csrc/attn && \
|
||||
git submodule update --init --recursive && \
|
||||
python setup_sta.py install
|
||||
# Remove authentication headers
|
||||
RUN git config --unset-all http.https://github.com/.extraheader || true
|
||||
|
||||
# Install VSA
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
cd csrc/attn && \
|
||||
git submodule update --init --recursive && \
|
||||
python setup_vsa.py install
|
||||
# Set up automatic conda environment activation for all shells
|
||||
RUN echo 'source /opt/conda/etc/profile.d/conda.sh' >> /root/.bashrc && \
|
||||
echo 'conda activate fastvideo-dev' >> /root/.bashrc && \
|
||||
# Ensure .bashrc is sourced for SSH login shells
|
||||
echo 'if [ -f ~/.bashrc ]; then . ~/.bashrc; fi' > /root/.profile
|
||||
|
||||
EXPOSE 22
|
||||
@@ -43,7 +43,7 @@ RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
uv pip install --no-cache-dir --upgrade pip && \
|
||||
uv pip install --no-cache-dir .[dev] && \
|
||||
uv pip install --no-cache-dir flash-attn==2.8.3 --no-build-isolation
|
||||
uv pip install --no-cache-dir flash-attn==2.8.0.post2 --no-build-isolation
|
||||
|
||||
COPY . .
|
||||
|
||||
|
||||
@@ -1,72 +0,0 @@
|
||||
FROM nvidia/cuda:12.9.1-cudnn-devel-ubuntu22.04
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
WORKDIR /FastVideo
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
wget \
|
||||
git \
|
||||
ca-certificates \
|
||||
openssh-server \
|
||||
zsh \
|
||||
vim \
|
||||
curl \
|
||||
gcc-11 \
|
||||
g++-11 \
|
||||
clang-11 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Set up C++20 compilers for ThunderKittens
|
||||
RUN update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
# Set CUDA environment variables
|
||||
ENV CUDA_HOME=/usr/local/cuda-12.9
|
||||
ENV PATH=${CUDA_HOME}/bin:${PATH}
|
||||
ENV LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
|
||||
# Install uv and source its environment
|
||||
RUN curl -LsSf https://astral.sh/uv/install.sh | sh && \
|
||||
echo 'source $HOME/.local/bin/env' >> /root/.bashrc
|
||||
|
||||
# Copy just the pyproject.toml first to leverage Docker cache
|
||||
COPY pyproject.toml ./
|
||||
|
||||
# Create a dummy README to satisfy the installation
|
||||
RUN echo "# Placeholder" > README.md
|
||||
|
||||
# Create and activate virtual environment with specific Python version and seed
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
uv venv --python 3.12 --seed /opt/venv && \
|
||||
source /opt/venv/bin/activate && \
|
||||
uv pip install --no-cache-dir --upgrade pip && \
|
||||
uv pip install --no-cache-dir .[dev] && \
|
||||
uv pip install --no-cache-dir flash-attn==2.8.3 --no-build-isolation
|
||||
|
||||
COPY . .
|
||||
|
||||
# Install dependencies using uv and set up shell configuration
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
uv pip install --no-cache-dir -e .[dev] && \
|
||||
git config --unset-all http.https://github.com/.extraheader || true && \
|
||||
echo 'source /opt/venv/bin/activate' >> /root/.bashrc && \
|
||||
echo 'if [ -n "$ZSH_VERSION" ] && [ -f ~/.zshrc ]; then . ~/.zshrc; elif [ -f ~/.bashrc ]; then . ~/.bashrc; fi' > /root/.profile
|
||||
|
||||
# Install STA (Sliding Tile Attention)
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
cd csrc/attn && \
|
||||
git submodule update --init --recursive && \
|
||||
python setup_sta.py install
|
||||
|
||||
# Install VSA
|
||||
RUN source $HOME/.local/bin/env && \
|
||||
source /opt/venv/bin/activate && \
|
||||
cd csrc/attn && \
|
||||
git submodule update --init --recursive && \
|
||||
python setup_vsa.py install
|
||||
|
||||
EXPOSE 22
|
||||
|
Before Width: | Height: | Size: 194 KiB |
@@ -96,7 +96,8 @@ copybutton_prompt_is_regexp = True
|
||||
#
|
||||
html_title = project
|
||||
html_theme = 'sphinx_book_theme'
|
||||
html_logo = '../../assets/logos/icon_simple.svg'
|
||||
html_logo = '../../assets/logo.jpg'
|
||||
#html_favicon = 'assets/logos/vllm-logo-only-light.ico'
|
||||
html_theme_options = {
|
||||
'path_to_docs': 'docs/source',
|
||||
'repository_url': 'https://github.com/hao-ai-lab/FastVideo/',
|
||||
|
||||
@@ -3,12 +3,12 @@
|
||||
|
||||
If you prefer a containerized development environment or want to avoid managing dependencies manually, you can use our prebuilt Docker image:
|
||||
|
||||
**Images:** [`ghcr.io/hao-ai-lab/fastvideo/fastvideo-dev:py3.12-latest`](https://ghcr.io/hao-ai-lab/fastvideo/fastvideo-dev)
|
||||
**Image:** [`ghcr.io/hao-ai-lab/fastvideo/fastvideo-dev:latest`](https://ghcr.io/hao-ai-lab/fastvideo/fastvideo-dev)
|
||||
|
||||
## Starting the container
|
||||
|
||||
```bash
|
||||
docker run --gpus all -it ghcr.io/hao-ai-lab/fastvideo/fastvideo-dev:py3.12-latest
|
||||
docker run --gpus all -it ghcr.io/hao-ai-lab/fastvideo/fastvideo-dev:latest
|
||||
```
|
||||
|
||||
This will:
|
||||
|
||||
@@ -6,7 +6,7 @@ You can easily use the FastVideo Docker image as a custom container on [RunPod](
|
||||
|
||||
## Creating a new pod
|
||||
|
||||
Choose a GPU that supports CUDA 12.8
|
||||
Choose a GPU that supports CUDA 12.4
|
||||
|
||||
Pick 1 or 2 L40S GPU(s)
|
||||
|
||||
|
||||
@@ -22,20 +22,10 @@ source ~/.bashrc
|
||||
Create and activate a Conda environment for FastVideo:
|
||||
|
||||
```
|
||||
conda create -n fastvideo python=3.12 -y
|
||||
conda create -n fastvideo python=3.10 -y
|
||||
conda activate fastvideo
|
||||
```
|
||||
|
||||
Install `uv` (optional, but recommended):
|
||||
|
||||
From instructions on [uv](https://astral.sh/uv/):
|
||||
|
||||
```
|
||||
curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
# or
|
||||
wget -qO- https://astral.sh/uv/install.sh | sh
|
||||
```
|
||||
|
||||
Clone the FastVideo repository and go to the FastVideo directory:
|
||||
|
||||
```
|
||||
@@ -46,10 +36,10 @@ git clone https://github.com/hao-ai-lab/FastVideo.git && cd FastVideo
|
||||
Now you can install FastVideo and setup git hooks for running linting. By using `pre-commit`, the linters will run and have to pass before you'll be able to make a commit.
|
||||
|
||||
```bash
|
||||
uv pip install -e .[dev]
|
||||
pip install -e .[dev]
|
||||
|
||||
# Can also install flash-attn (optional)
|
||||
uv pip install flash-attn --no-build-isolation
|
||||
pip install flash-attn==2.7.4.post1 --no-build-isolation
|
||||
|
||||
# Linting, formatting and static type checking
|
||||
pre-commit install --hook-type pre-commit --hook-type commit-msg
|
||||
@@ -60,14 +50,3 @@ pre-commit run --all-files
|
||||
# Unit tests
|
||||
pytest tests/
|
||||
```
|
||||
|
||||
If you are on a Hopper GPU, you should also install [FA3](https://github.com/Dao-AILab/flash-attention) for much better performance:
|
||||
|
||||
```
|
||||
git clone https://github.com/Dao-AILab/flash-attention.git && cd flash-attention/hopper
|
||||
|
||||
# make sure you have ninja installed
|
||||
uv pip install ninja
|
||||
|
||||
python setup.py install
|
||||
```
|
||||
|
||||
@@ -1,43 +0,0 @@
|
||||
(v0-data-preprocess)=
|
||||
|
||||
# 🧱 Data Preprocess for Distillation
|
||||
|
||||
For distillation, we use the same data preprocessing pipeline as training. Please refer to the [Training Data Preprocess](../training/data_preprocess.md) for general preprocessing steps.
|
||||
|
||||
## Distillation-Specific Datasets
|
||||
|
||||
### FastVideo 480P Synthetic Wan Dataset
|
||||
|
||||
For Wan2.1 T2V distillation, we use the **FastVideo 480P Synthetic Wan dataset** ([FastVideo/Wan-Syn_77x448x832_600k](https://huggingface.co/datasets/FastVideo/Wan-Syn_77x448x832_600k)) which contains 600k synthetic latents.
|
||||
|
||||
```bash
|
||||
# Download the preprocessed dataset
|
||||
python scripts/huggingface/download_hf.py \
|
||||
--repo_id "FastVideo/Wan-Syn_77x448x832_600k" \
|
||||
--local_dir "FastVideo/Wan-Syn_77x448x832_600k" \
|
||||
--repo_type "dataset"
|
||||
```
|
||||
|
||||
### Crush Smol Dataset
|
||||
|
||||
For Wan2.2 TI2V distillation, we use the crush_smol dataset which includes both raw videos and preprocessed latents.
|
||||
|
||||
```bash
|
||||
# Download dataset
|
||||
python scripts/huggingface/download_hf.py \
|
||||
--repo_id=FastVideo/mini_i2v_dataset \
|
||||
--local_dir=data/mini_i2v_dataset \
|
||||
--repo_type=dataset
|
||||
```
|
||||
|
||||
## Preprocessing for Distillation
|
||||
|
||||
The preprocessing steps are identical to training. Run the appropriate preprocessing script based on your model:
|
||||
|
||||
```bash
|
||||
# For Wan2.1 T2V
|
||||
bash scripts/preprocess/v1_preprocess_wan_data_t2v
|
||||
|
||||
# For Wan2.2 TI2V
|
||||
bash examples/distill/Wan2.2-TI2V-5B-Diffusers/crush_smol/preprocess_wan_data_ti2v_5b.sh
|
||||
```
|
||||
@@ -1,87 +0,0 @@
|
||||
# 🎯 Distillation
|
||||
|
||||
We introduce a new finetuning strategy - **Sparse-distill**, which jointly integrates **[DMD](https://arxiv.org/abs/2405.14867)** and **[VSA](https://arxiv.org/abs/2505.13389)** in a single training process. This approach combines the benefits of both distillation to shorten diffusion steps and sparse attention to reduce attention computations, enabling much faster video generation.
|
||||
|
||||
## 📊 Model Overview
|
||||
|
||||
We provide two distilled models:
|
||||
|
||||
- **[FastWan2.1-T2V-1.3B-Diffusers](https://huggingface.co/FastVideo/FastWan2.1-T2V-1.3B-Diffusers)**: 3-step inference, up to **16 FPS** on H100 GPU
|
||||
- **[FastWan2.1-T2V-14B-480P-Diffusers](https://huggingface.co/FastVideo/FastWan2.1-T2V-14B-480P-Diffusers)**: 3-step inference, up to **60x speed up** at 480P, **90x speed up** at 720P for denoising loop
|
||||
- **[FastWan2.2-TI2V-5B-FullAttn-Diffusers](https://huggingface.co/FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers)**: 3-step inference, up to **50x speed up** at 720P for denoising loop
|
||||
|
||||
Both models are trained on **61×448×832** resolution but support generating videos with **any resolution** (1.3B model mainly support 480P, 14B model support 480P and 720P, quality may degrade for different resolutions).
|
||||
|
||||
## ⚙️ Inference
|
||||
First install [VSA](https://hao-ai-lab.github.io/FastVideo/video_sparse_attention/installation.html). Set `MODEL_BASE` to your own model path and run:
|
||||
|
||||
```bash
|
||||
bash scripts/inference/v1_inference_wan_dmd.sh
|
||||
```
|
||||
|
||||
## 🗂️ Dataset
|
||||
|
||||
We use the **FastVideo 480P Synthetic Wan dataset** ([FastVideo/Wan-Syn_77x448x832_600k](https://huggingface.co/datasets/FastVideo/Wan-Syn_77x448x832_600k)) for distillation, which contains 600k synthetic latents.
|
||||
|
||||
### Download Dataset
|
||||
|
||||
```bash
|
||||
# Download the preprocessed dataset
|
||||
python scripts/huggingface/download_hf.py \
|
||||
--repo_id "FastVideo/Wan-Syn_77x448x832_600k" \
|
||||
--local_dir "FastVideo/Wan-Syn_77x448x832_600k" \
|
||||
--repo_type "dataset"
|
||||
```
|
||||
|
||||
## 🚀 Training Scripts
|
||||
|
||||
### Wan2.1 1.3B Model Sparse-Distill
|
||||
|
||||
For the 1.3B model, we use **4 nodes with 32 H200 GPUs** (8 GPUs per node):
|
||||
|
||||
```bash
|
||||
# Multi-node training (8 nodes, 64 GPUs total)
|
||||
sbatch examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/distill_dmd_VSA_t2v_1.3B.slurm
|
||||
```
|
||||
|
||||
**Key Configuration:**
|
||||
- Global batch size: 64
|
||||
- Gradient accumulation steps: 2
|
||||
- Learning rate: 1e-5
|
||||
- VSA attention sparsity: 0.8
|
||||
- Training steps: 4000 (~12 hours)
|
||||
|
||||
### Wan2.1 14B Model Sparse-Distill
|
||||
|
||||
For the 14B model, we use **8 nodes with 64 H200 GPUs** (8 GPUs per node):
|
||||
|
||||
```bash
|
||||
# Multi-node training (8 nodes, 64 GPUs total)
|
||||
sbatch examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/distill_dmd_VSA_t2v_14B.slurm
|
||||
```
|
||||
|
||||
**Key Configuration:**
|
||||
- Global batch size: 64
|
||||
- Sequence parallel size: 4
|
||||
- Gradient accumulation steps: 4
|
||||
- Learning rate: 1e-5
|
||||
- VSA attention sparsity: 0.9
|
||||
- Training steps: 3000 (~52 hours)
|
||||
- HSDP shard dim: 8
|
||||
|
||||
### Wan2.2 5B Model Sparse-Distill
|
||||
|
||||
For the 5B model, we use **8 nodes with 64 H200 GPUs** (8 GPUs per node):
|
||||
|
||||
```bash
|
||||
# Multi-node training (8 nodes, 64 GPUs total)
|
||||
sbatch examples/distill/Wan2.2-TI2V-5B-Diffusers/Data-free/distill_dmd_t2v_5B.sh
|
||||
```
|
||||
|
||||
**Key Configuration:**
|
||||
- Global batch size: 64
|
||||
- Sequence parallel size: 1
|
||||
- Gradient accumulation steps: 1
|
||||
- Learning rate: 2e-5
|
||||
- Training steps: 3000 (~12 hours)
|
||||
- HSDP shard dim: 1
|
||||
@@ -287,35 +287,11 @@ def create_nested_structures(
|
||||
relative_path = example.path.relative_to(category_dir)
|
||||
path_parts = relative_path.parts
|
||||
|
||||
if example.category == "training":
|
||||
# For training examples like finetune/wan_i2v_14b_480p/crush_smol
|
||||
if len(path_parts) >= 3:
|
||||
method = path_parts[0] # e.g., "finetune"
|
||||
model = path_parts[1] # e.g., "wan_i2v_14b_480p"
|
||||
dataset = path_parts[2] # e.g., "crush_smol"
|
||||
|
||||
# Initialize nested structure
|
||||
if example.category not in nested_structures:
|
||||
nested_structures[example.category] = {}
|
||||
if method not in nested_structures[example.category]:
|
||||
nested_structures[example.category][method] = {}
|
||||
if model not in nested_structures[example.category][method]:
|
||||
nested_structures[example.category][method][model] = {}
|
||||
|
||||
# Store the nested structure
|
||||
nested_structures[
|
||||
example.category][method][model][dataset] = NestedStructure(
|
||||
category=example.category,
|
||||
method=method,
|
||||
model=model,
|
||||
dataset=dataset,
|
||||
example=example)
|
||||
|
||||
elif example.category == "distillation" and len(path_parts) >= 2:
|
||||
# For distillation examples like Wan2.1-T2V/Wan-Syn-Data-480P
|
||||
model = path_parts[0] # e.g., "Wan2.1-T2V"
|
||||
dataset = path_parts[1] # e.g., "Wan-Syn-Data-480P"
|
||||
method = "DMD" # Default method for distillation
|
||||
# For nested examples like finetune/wan_i2v_14b_480p/crush_smol
|
||||
if len(path_parts) >= 3:
|
||||
method = path_parts[0] # e.g., "finetune"
|
||||
model = path_parts[1] # e.g., "wan_i2v_14b_480p"
|
||||
dataset = path_parts[2] # e.g., "crush_smol"
|
||||
|
||||
# Initialize nested structure
|
||||
if example.category not in nested_structures:
|
||||
|
||||
@@ -6,7 +6,7 @@ Instructions to install FastVideo for NVIDIA CUDA GPUs.
|
||||
|
||||
- **OS: Linux or Windows WSL**
|
||||
- **Python: 3.10-3.12**
|
||||
- **CUDA 12.8**
|
||||
- **CUDA 12.4**
|
||||
- **At least 1 NVIDIA GPU**
|
||||
|
||||
## Set up using Python
|
||||
@@ -38,7 +38,6 @@ conda activate fastvideo
|
||||
|
||||
:::{tip}
|
||||
We highly recommend using `uv` to install FastVideo. In our experience, `uv` speeds up installation by at least 3x.
|
||||
Note that you can also use `uv` to install FastVideo in a Conda environment.
|
||||
:::
|
||||
|
||||
Or you can create a new Python environment using [uv](https://docs.astral.sh/uv/), a very fast Python environment manager. Please follow the [documentation](https://docs.astral.sh/uv/#getting-started) to install `uv`. After installing `uv`, you can create a new Python environment using the following command:
|
||||
@@ -61,7 +60,7 @@ uv pip install fastvideo
|
||||
Also optionally install flash-attn:
|
||||
|
||||
```bash
|
||||
pip install flash-attn --no-build-isolation
|
||||
pip install flash-attn==2.7.4.post1 --no-build-isolation
|
||||
```
|
||||
|
||||
### Installation from Source
|
||||
@@ -88,7 +87,7 @@ uv pip install -e .
|
||||
#### Flash Attention
|
||||
|
||||
```bash
|
||||
pip install flash-attn --no-build-isolation
|
||||
pip install flash-attn==2.7.4.post1 --no-build-isolation
|
||||
```
|
||||
|
||||
## Set up using Docker
|
||||
@@ -103,7 +102,7 @@ If you're planning to contribute to FastVideo please see the following page:
|
||||
## Hardware Requirements
|
||||
|
||||
### For Basic Inference
|
||||
- NVIDIA GPU with CUDA 12.8 support
|
||||
- NVIDIA GPU with CUDA 12.4 support
|
||||
|
||||
### For Lora Finetuning
|
||||
- 40GB GPU memory each for 2 GPUs with lora
|
||||
|
||||
@@ -39,7 +39,6 @@ conda activate fastvideo
|
||||
|
||||
:::{tip}
|
||||
We highly recommend using `uv` to install FastVideo. In our experience, `uv` speeds up installation by at least 3x.
|
||||
Note that you can also use `uv` to install FastVideo in a Conda environment.
|
||||
:::
|
||||
|
||||
Or you can create a new Python environment using [uv](https://docs.astral.sh/uv/), a very fast Python environment manager. Please follow the [documentation](https://docs.astral.sh/uv/#getting-started) to install `uv`. After installing `uv`, you can create a new Python environment using the following command:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Welcome to FastVideo
|
||||
|
||||
:::{figure} ../../assets/logos/logo.svg
|
||||
:::{figure} ../../assets/logo.jpg
|
||||
:align: center
|
||||
:alt: FastVideo
|
||||
:class: no-scaled-link
|
||||
@@ -9,7 +9,7 @@
|
||||
|
||||
:::{raw} html
|
||||
<p style="text-align:center">
|
||||
<strong>FastVideo is a unified inference and post-training framework for accelerated video generation.
|
||||
<strong>FastVideo is a unified framework for accelerated video generation.
|
||||
</strong>
|
||||
</p>
|
||||
|
||||
@@ -21,10 +21,11 @@
|
||||
</p>
|
||||
:::
|
||||
|
||||
FastVideo is an inference and post-training framework for diffusion models. It features an end-to-end unified pipeline for accelerating diffusion models, starting from data preprocessing to model training, finetuning, distillation, and inference. FastVideo is designed to be modular and extensible, allowing users to easily add new optimizations and techniques. Whether it is training-free optimizations or post-training optimizations, FastVideo has you covered.
|
||||
It features a clean, consistent API that works across popular video models, making it easier for developers to author new models and incorporate system- or kernel-level optimizations.
|
||||
With FastVideo's optimizations, you can achieve more than 3x inference improvement compared to other systems.
|
||||
|
||||
<div style="text-align: center;">
|
||||
<img src=_static/images/fastwan.png width="100%"/>
|
||||
<img src=_static/images/perf.png width="100%"/>
|
||||
</div>
|
||||
|
||||
## Key Features
|
||||
@@ -34,11 +35,16 @@ FastVideo has the following features:
|
||||
- [Sliding Tile Attention](https://arxiv.org/pdf/2502.04507)
|
||||
- [TeaCache](https://arxiv.org/pdf/2411.19108)
|
||||
- [Sage Attention](https://arxiv.org/abs/2410.02367)
|
||||
- E2E post-training support
|
||||
- Data preprocessing pipeline for video data.
|
||||
- [Sparse distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/) for Wan2.1 and Wan2.2 using [Video Sparse Attention](https://arxiv.org/pdf/2505.13389) and [Distribution Matching Distillation](https://tianweiy.github.io/dmd2/)
|
||||
- Support full finetuning and LoRA finetuning for state-of-the-art open video DiTs.
|
||||
- Scalable training with FSDP2, sequence parallelism, and selective activation checkpointing, with near linear scaling to 64 GPUs.
|
||||
- Cutting edge models
|
||||
- Wan2.1 T2V, I2V
|
||||
- HunyuanVideo
|
||||
- FastHunyuan: consistency distilled video diffusion models for 8x inference speedup.
|
||||
- StepVideo T2V
|
||||
- Distillation support
|
||||
- Recipes for video DiT, based on [PCM](https://github.com/G-U-N/Phased-Consistency-Model).
|
||||
- Support distilling/finetuning/inferencing state-of-the-art open video DiTs: 1. Mochi 2. Hunyuan.
|
||||
- Scalable training with FSDP, sequence parallelism, and selective activation checkpointing, with near linear scaling to 64 GPUs.
|
||||
- Memory efficient finetuning with LoRA, precomputed latent, and precomputed text embeddings.
|
||||
|
||||
## Documentation
|
||||
|
||||
@@ -72,16 +78,18 @@ inference/add_pipeline
|
||||
|
||||
training/examples/examples_training_index
|
||||
training/data_preprocess
|
||||
training/distillation
|
||||
<!-- training/finetune -->
|
||||
:::
|
||||
|
||||
:::{toctree}
|
||||
<!-- :::{toctree}
|
||||
:caption: Distillation
|
||||
:maxdepth: 1
|
||||
|
||||
distillation/examples/examples_distillation_index
|
||||
distillation/data_preprocess
|
||||
distillation/dmd
|
||||
distillation/dmd -->
|
||||
<!-- training/finetune -->
|
||||
:::
|
||||
|
||||
% What is STA Kernel?
|
||||
@@ -94,15 +102,6 @@ sliding_tile_attention/installation
|
||||
sliding_tile_attention/demo
|
||||
:::
|
||||
|
||||
% What is VSA Kernel?
|
||||
|
||||
:::{toctree}
|
||||
:caption: Video Sparse Attention
|
||||
:maxdepth: 1
|
||||
|
||||
video_sparse_attention/installation
|
||||
:::
|
||||
|
||||
:::{toctree}
|
||||
:caption: Design
|
||||
:maxdepth: 1
|
||||
|
||||
@@ -5,7 +5,7 @@ This page contains step-by-step instructions to get you quickly started with vid
|
||||
## Requirements
|
||||
- **OS**: Linux (Tested on Ubuntu 22.04+)
|
||||
- **Python**: 3.10-3.12
|
||||
- **CUDA**: 12.8
|
||||
- **CUDA**: 12.4
|
||||
- **GPU**: At least one NVIDIA GPU
|
||||
|
||||
## Installation
|
||||
|
||||
@@ -19,7 +19,6 @@ This page describes the various options for speeding up generation times in Fast
|
||||
- Torch SDPA: `FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA`
|
||||
- Flash Attention 2 and 3: `FASTVIDEO_ATTENTION_BACKEND=FLASH_ATTN`
|
||||
- Sliding Tile Attention: `FASTVIDEO_ATTENTION_BACKEND=SLIDING_TILE_ATTN`
|
||||
- Video Sparse Attention: `FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN`
|
||||
- Sage Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN`
|
||||
|
||||
### Configuring Backends
|
||||
@@ -75,17 +74,6 @@ pip install st_attn==0.0.4
|
||||
|
||||
Please see [this page](#sta-installation) for more installation instructions.
|
||||
|
||||
(optimizations-vsa)=
|
||||
### Video Sparse Attention
|
||||
**`VIDEO_SPARSE_ATTN`**
|
||||
|
||||
```bash
|
||||
git submodule update --init --recursive
|
||||
python setup_vsa.py install
|
||||
```
|
||||
|
||||
Please see [this page](#vsa-installation) for more installation instructions.
|
||||
|
||||
(optimizations-sage)=
|
||||
### Sage Attention
|
||||
**`SAGE_ATTN`**
|
||||
|
||||
@@ -6,7 +6,6 @@ The symbols used have the following meanings:
|
||||
|
||||
- ✅ = Full compatibility
|
||||
- ❌ = No compatibility
|
||||
- ⭕ = Does not apply to this model
|
||||
|
||||
## Models x Optimization
|
||||
The `HuggingFace Model ID` can be directly pass to `from_pretrained()` methods and FastVideo will use the optimal default parameters when initializing and generating videos.
|
||||
@@ -38,94 +37,51 @@ The `HuggingFace Model ID` can be directly pass to `from_pretrained()` methods a
|
||||
* TeaCache
|
||||
* Sliding Tile Attn
|
||||
* Sage Attn
|
||||
* Video Sparse Attention (VSA)
|
||||
- * FastWan2.1 T2V 1.3B
|
||||
* `FastVideo/FastWan2.1-T2V-1.3B-Diffusers`
|
||||
* 480P
|
||||
* ⭕
|
||||
* ⭕
|
||||
* ⭕
|
||||
* ✅
|
||||
- * FastWan2.2 TI2V 5B Full Attn*
|
||||
* `FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers`
|
||||
* 720P
|
||||
* ⭕
|
||||
* ⭕
|
||||
* ⭕
|
||||
* ✅
|
||||
- * Wan2.2 TI2V 5B
|
||||
* `Wan-AI/Wan2.2-TI2V-5B-Diffusers`
|
||||
* 720P
|
||||
* ⭕
|
||||
* ⭕
|
||||
* ✅
|
||||
* ⭕
|
||||
- * Wan2.2 T2V A14B
|
||||
* `Wan-AI/Wan2.2-T2V-A14B-Diffusers`
|
||||
* 480P<br>720P
|
||||
* ❌
|
||||
* ❌
|
||||
* ✅
|
||||
* ⭕
|
||||
- * Wan2.2 I2V A14B
|
||||
* `Wan-AI/Wan2.2-I2V-A14B-Diffusers`
|
||||
* 480P<br>720P
|
||||
* ❌
|
||||
* ❌
|
||||
* ✅
|
||||
* ⭕
|
||||
- * HunyuanVideo
|
||||
* `hunyuanvideo-community/HunyuanVideo`
|
||||
* 720px1280p<br>544px960p
|
||||
* ❌
|
||||
* ✅
|
||||
* ✅
|
||||
* ⭕
|
||||
- * FastHunyuan
|
||||
* `FastVideo/FastHunyuan-diffusers`
|
||||
* 720px1280p<br>544px960p
|
||||
* ❌
|
||||
* ✅
|
||||
* ✅
|
||||
* ⭕
|
||||
- * Wan2.1 T2V 1.3B
|
||||
- * Wan T2V 1.3B
|
||||
* `Wan-AI/Wan2.1-T2V-1.3B-Diffusers`
|
||||
* 480P
|
||||
* ✅
|
||||
* ✅*
|
||||
* ✅
|
||||
* ⭕
|
||||
- * Wan2.1 T2V 14B
|
||||
- * Wan T2V 14B
|
||||
* `Wan-AI/Wan2.1-T2V-14B-Diffusers`
|
||||
* 480P, 720P
|
||||
* ✅
|
||||
* ✅*
|
||||
* ✅
|
||||
* ⭕
|
||||
- * Wan2.1 I2V 480P
|
||||
- * Wan I2V 480P
|
||||
* `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers`
|
||||
* 480P
|
||||
* ✅
|
||||
* ✅*
|
||||
* ✅
|
||||
* ⭕
|
||||
- * Wan2.1 I2V 720P
|
||||
- * Wan I2V 720P
|
||||
* `Wan-AI/Wan2.1-I2V-14B-720P-Diffusers`
|
||||
* 720P
|
||||
* ✅
|
||||
* ✅*
|
||||
* ✅
|
||||
* ✅
|
||||
* ⭕
|
||||
- * StepVideo T2V
|
||||
* `FastVideo/stepvideo-t2v-diffusers`
|
||||
* 768px768px204f<br>544px992px204f<br>544px992px136f
|
||||
* ❌
|
||||
* ❌
|
||||
* ✅
|
||||
* ⭕
|
||||
:::
|
||||
|
||||
**Note**: Wan2.2 TI2V 5B has some quality issues when performing I2V generation. We are working on fixing this issue.
|
||||
**Note**: there are some known quality issues with Wan2.1 + Sliding Tile Attn. We are working on fixing this issue.
|
||||
|
||||
## Special requirements
|
||||
|
||||
|
||||
@@ -29,13 +29,13 @@ export CUDA_HOME=/usr/local/cuda-12.4
|
||||
export PATH=${CUDA_HOME}/bin:${PATH}
|
||||
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
git submodule update --init --recursive
|
||||
python setup_sta.py install
|
||||
python setup.py install
|
||||
```
|
||||
|
||||
# 🧪 Test
|
||||
|
||||
```bash
|
||||
python csrc/attn/tests/test_sta.py
|
||||
python test/test_sta.py
|
||||
```
|
||||
|
||||
# 📋 Usage
|
||||
@@ -53,9 +53,3 @@ out = sliding_tile_attention(q, k, v, window_size, text_length)
|
||||
out = sliding_tile_attention(q, k, v, window_size, 0, False)
|
||||
|
||||
```
|
||||
|
||||
# 🚀Inference
|
||||
|
||||
```bash
|
||||
bash scripts/inference/v1_inference_wan_STA.sh
|
||||
```
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
(v0-distill)=
|
||||
# 🎯 Distill
|
||||
Our distillation recipe is based on [Phased Consistency Model](https://github.com/G-U-N/Phased-Consistency-Model). We did not find significant improvement using multi-phase distillation, so we keep the one phase setup similar to the original latent consistency model's recipe.
|
||||
We use the [MixKit](https://huggingface.co/datasets/LanguageBind/Open-Sora-Plan-v1.1.0/tree/main/all_mixkit) dataset for distillation. To avoid running the text encoder and VAE during training, we prprocess all data to generate text embeddings and VAE latents.
|
||||
Preprocessing instructions can be found [data_preprocess.md](#v0-data-preprocess). For convenience, we also provide preprocessed data that can be downloaded directly using the following command:
|
||||
|
||||
```bash
|
||||
python scripts/huggingface/download_hf.py --repo_id=FastVideo/HD-Mixkit-Finetune-Hunyuan --local_dir=data/HD-Mixkit-Finetune-Hunyuan --repo_type=dataset
|
||||
```
|
||||
|
||||
Next, download the original model weights with:
|
||||
|
||||
```bash
|
||||
python scripts/huggingface/download_hf.py --repo_id=FastVideo/hunyuan --local_dir=data/hunyuan --repo_type=model # original hunyuan
|
||||
python scripts/huggingface/download_hf.py --repo_id=genmo/mochi-1-preview --local_dir=data/mochi --repo_type=model # original mochi
|
||||
```
|
||||
|
||||
To launch the distillation process, use the following commands:
|
||||
|
||||
```
|
||||
bash scripts/distill/distill_hunyuan.sh # for hunyuan
|
||||
bash scripts/distill/distill_mochi.sh # for mochi
|
||||
```
|
||||
|
||||
We also provide an optional script for distillation with adversarial loss, located at `fastvideo/distill_adv.py`. Although we tried adversarial loss, we did not observe significant improvements.
|
||||
@@ -7,7 +7,7 @@ Ensure your data is prepared and preprocessed in the format specified in [data_p
|
||||
python scripts/huggingface/download_hf.py --repo_id=FastVideo/Mochi-Black-Myth --local_dir=data/Mochi-Black-Myth --repo_type=dataset
|
||||
```
|
||||
|
||||
Download the original model weights as specified in the [Distillation Section](../distillation/dmd.md):
|
||||
Download the original model weights as specified in [Distill Section](#v0-distill):
|
||||
|
||||
Then you can run the finetune with:
|
||||
|
||||
|
||||
@@ -1,67 +0,0 @@
|
||||
(vsa-installation)=
|
||||
|
||||
# 🔧 Installation
|
||||
You can install the Video Sparse Attention package using
|
||||
|
||||
```bash
|
||||
git submodule update --init --recursive
|
||||
python setup_vsa.py install
|
||||
```
|
||||
|
||||
# Building from Source
|
||||
We support H100 (via ThunderKittens) and any other GPU (via Triton) for VSA.
|
||||
|
||||
First, install C++20 for ThunderKittens (if using H100):
|
||||
|
||||
```bash
|
||||
sudo apt update
|
||||
sudo apt install gcc-11 g++-11
|
||||
|
||||
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
sudo apt update
|
||||
sudo apt install clang-11
|
||||
```
|
||||
|
||||
Set up CUDA environment (if using CUDA 12.8):
|
||||
|
||||
```bash
|
||||
export CUDA_HOME=/usr/local/cuda-12.8
|
||||
export PATH=${CUDA_HOME}/bin:${PATH}
|
||||
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
```
|
||||
|
||||
Install VSA:
|
||||
|
||||
```bash
|
||||
cd csrc/attn/
|
||||
git submodule update --init --recursive
|
||||
python setup_vsa.py install
|
||||
```
|
||||
|
||||
# 🧪 Test
|
||||
|
||||
```bash
|
||||
python csrc/attn/tests/test_vsa.py
|
||||
```
|
||||
|
||||
# 📋 Usage
|
||||
|
||||
```python
|
||||
from vsa import video_sparse_attn
|
||||
|
||||
# q, k, v: [batch_size, num_heads, seq_len, head_dim]
|
||||
# variable_block_sizes: [num_blocks] - number of valid tokens in each block
|
||||
# topk: int - number of top-k blocks to attend to
|
||||
# block_size: int or tuple of 3 ints - size of each block (default: 64 tokens)
|
||||
# compress_attn_weight: optional weight for compressed attention branch
|
||||
|
||||
output = video_sparse_attn(q, k, v, variable_block_sizes, topk, block_size, compress_attn_weight)
|
||||
|
||||
```
|
||||
|
||||
# 🚀Inference
|
||||
|
||||
```bash
|
||||
bash scripts/inference/v1_inference_wan_VSA.sh
|
||||
```
|
||||
@@ -1,139 +0,0 @@
|
||||
# Basic Info
|
||||
export NCCL_P2P_DISABLE=1
|
||||
export TORCH_NCCL_ENABLE_MONITORING=0
|
||||
# different cache dir for different processes
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
|
||||
export MASTER_PORT=29500
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=offline
|
||||
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
|
||||
export WANDB_API_KEY='8d9f4b39abd68eb4e29f6fc010b7ee71a2207cde'
|
||||
|
||||
# Configs
|
||||
NUM_GPUS=8
|
||||
|
||||
# Model paths for DMD distillation:
|
||||
GENERATOR_MODEL_PATH="wlsaidhi/SFWan2.1-T2V-1.3B-Diffusers"
|
||||
REAL_SCORE_MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # Teacher model
|
||||
FAKE_SCORE_MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # Critic model
|
||||
|
||||
DATA_DIR="data/crush-smol_processed_t2v/combined_parquet_dataset/"
|
||||
VALIDATION_DATASET_FILE="/mnt/weka/home/hao.zhang/wl/FastVideo/examples/distill/SFWan2.1-I2V/validation_better.json"
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name SFwan_t2v_distill_self_forcing_dmd
|
||||
--output_dir "/mnt/sharefs/users/hao.zhang/wl/sf_checkpoints/ode0_SFwan_t2v_finetune_sf_${lr}_c${critic_lr}"
|
||||
--wandb_run_name "DEBUG${lr}_c${critic_lr}"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--num_frames 81
|
||||
--warp_denoising_step
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
--log_visualization
|
||||
--simulate_generator_forward
|
||||
--num_frame_per_block 3
|
||||
--enable_gradient_masking
|
||||
--gradient_mask_last_n_frames 21
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 8 # 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1 # 64
|
||||
--hsdp_shard_dim 8
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $GENERATOR_MODEL_PATH # TODO: check if you can remove this in this script
|
||||
--pretrained_model_name_or_path $GENERATOR_MODEL_PATH
|
||||
--generator_model_path $GENERATOR_MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
# --log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 50
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-5
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 50
|
||||
--weight_only_checkpointing_steps 50
|
||||
--weight_decay 0.01
|
||||
--betas '0.0,0.999'
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--use_ema True
|
||||
--ema_decay 0.99
|
||||
--ema_start_step 100
|
||||
--init_weights_from_safetensors "/mnt/weka/home/hao.zhang/wl/Self-Forcing/diffusers_ode_init/model.safetensors"
|
||||
)
|
||||
|
||||
# Self-forcing DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,750,500,250'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--dfake_gen_update_ratio 5
|
||||
--real_score_guidance_scale 3.0
|
||||
--fake_score_learning_rate 8e-6
|
||||
--fake_score_betas '0.0,0.999'
|
||||
)
|
||||
|
||||
# Self-forcing specific arguments
|
||||
self_forcing_args=(
|
||||
--independent_first_frame False # Whether to treat first frame independently
|
||||
--same_step_across_blocks False # Whether to use same denoising step across all blocks
|
||||
--last_step_only False # Whether to only use the last denoising step
|
||||
--context_noise 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
--validate_cache_structure False # Set to True for debugging KV cache issues
|
||||
)
|
||||
|
||||
torchrun \
|
||||
--nnodes 1 \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
--master_port $MASTER_PORT \
|
||||
fastvideo/training/wan_self_forcing_distillation_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}" \
|
||||
"${self_forcing_args[@]}"
|
||||
@@ -1,165 +0,0 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=t2v
|
||||
#SBATCH --partition=main
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks=1
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --gres=gpu:1
|
||||
#SBATCH --cpus-per-task=128
|
||||
#SBATCH --mem=1440G
|
||||
#SBATCH --output=dmd_t2v_output/t2v_%j.out
|
||||
#SBATCH --error=dmd_t2v_output/t2v_%j.err
|
||||
#SBATCH --exclusive
|
||||
set -e -x
|
||||
|
||||
# Environment Setup
|
||||
source ~/conda/miniconda/bin/activate
|
||||
conda activate wei-fv
|
||||
|
||||
# Basic Info
|
||||
export WANDB_MODE="online"
|
||||
export NCCL_P2P_DISABLE=1
|
||||
export TORCH_NCCL_ENABLE_MONITORING=0
|
||||
# different cache dir for different processes
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
|
||||
export MASTER_PORT=29500
|
||||
export NODE_RANK=$SLURM_PROCID
|
||||
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
|
||||
export MASTER_ADDR=${nodes[0]}
|
||||
export CUDA_VISIBLE_DEVICES=$SLURM_LOCALID
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export FASTVIDEO_ATTENTION_BACKEND=FLASH_ATTN
|
||||
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
|
||||
|
||||
echo "MASTER_ADDR: $MASTER_ADDR"
|
||||
echo "NODE_RANK: $NODE_RANK"
|
||||
|
||||
# Configs
|
||||
NUM_GPUS=1
|
||||
|
||||
# Model paths for Self-Forcing DMD distillation:
|
||||
GENERATOR_MODEL_PATH="wlsaidhi/SFWan2.1-T2V-1.3B-Diffusers"
|
||||
REAL_SCORE_MODEL_PATH="Wan-AI/Wan2.1-T2V-14B-Diffusers" # Teacher model
|
||||
FAKE_SCORE_MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # Critic model
|
||||
|
||||
DATA_DIR="data/crush-smol-single_processed_t2v/combined_parquet_dataset/"
|
||||
VALIDATION_DATASET_FILE="data/crush-smol-single_processed_t2v/validation.json"
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name SFwan_t2v_distill_self_forcing_dmd # Updated for self-forcing DMD
|
||||
--output_dir "checkpoints/SFwan_t2v_finetune"
|
||||
--max_train_steps 500
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--num_frames 81 # Must be divisible by num_frame_per_block (81 % 3 = 0 ✓)
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
--log_visualization
|
||||
--simulate_generator_forward
|
||||
--num_frame_per_block 3 # Frame generation block size for self-forcing
|
||||
--enable_gradient_masking
|
||||
--gradient_mask_last_n_frames 21
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 1 # 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1 # 64
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $GENERATOR_MODEL_PATH # TODO: check if you can remove this in this script
|
||||
--pretrained_model_name_or_path $GENERATOR_MODEL_PATH
|
||||
--generator_model_path $GENERATOR_MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 10
|
||||
--validation_sampling_steps "4"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-5
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 50
|
||||
--weight_only_checkpointing_steps 50
|
||||
--weight_decay 0.01
|
||||
--betas '0.0,0.999'
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--use_ema True
|
||||
--ema_decay 0.99
|
||||
--ema_start_step 100
|
||||
--init_weights_from_safetensors "/mnt/weka/home/hao.zhang/wl/Self-Forcing/diffusers_ode_init/model.safetensors"
|
||||
)
|
||||
|
||||
# Self-forcing DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,750,500,250'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--dfake_gen_update_ratio 5
|
||||
--real_score_guidance_scale 3.0
|
||||
--fake_score_learning_rate 8e-6
|
||||
--fake_score_betas '0.0,0.999'
|
||||
)
|
||||
|
||||
# Self-forcing specific arguments
|
||||
self_forcing_args=(
|
||||
--independent_first_frame False # Whether to treat first frame independently
|
||||
--same_step_across_blocks False # Whether to use same denoising step across all blocks
|
||||
--last_step_only False # Whether to only use the last denoising step
|
||||
--context_noise 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
--validate_cache_structure False # Set to True for debugging KV cache issues
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
--node_rank $SLURM_PROCID \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_self_forcing_distillation_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}" \
|
||||
"${self_forcing_args[@]}"
|
||||
@@ -1,3 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
python scripts/huggingface/download_hf.py --repo_id "wlsaidhi/crush-smol-merged" --local_dir "data/crush-smol" --repo_type "dataset"
|
||||
@@ -1,43 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Download the full dataset first
|
||||
python scripts/huggingface/download_hf.py --repo_id "wlsaidhi/crush-smol-merged" --local_dir "data/crush-smol" --repo_type "dataset"
|
||||
|
||||
# Create a single-example dataset for debugging
|
||||
SINGLE_EXAMPLE_DIR="data/crush-smol-single"
|
||||
mkdir -p "$SINGLE_EXAMPLE_DIR/videos"
|
||||
|
||||
# Copy the specific video that matches the validation.json style (macaron crushing)
|
||||
cp "data/crush-smol/videos/7P02AihYkCU-Scene-005.mp4" "$SINGLE_EXAMPLE_DIR/videos/"
|
||||
|
||||
# Create a single-line videos.txt
|
||||
echo "videos/7P02AihYkCU-Scene-005.mp4" > "$SINGLE_EXAMPLE_DIR/videos.txt"
|
||||
|
||||
# Create a single-line prompt.txt with the macaron crushing prompt
|
||||
echo "PIKA_CRUSH A large metal press is shown compressing a pile of colorful macarons, flattening them as if they were under a hydraulic press. The press moves down, crushing the macarons into a pile of crumbs and squishing the colorful filling out." > "$SINGLE_EXAMPLE_DIR/prompt.txt"
|
||||
|
||||
# Generate the JSON file and merge.txt for the single example
|
||||
python scripts/dataset_preparation/prepare_json_file.py --data_folder "$SINGLE_EXAMPLE_DIR" --output "videos2caption.json"
|
||||
|
||||
# Create a validation.json that uses the same example for consistency
|
||||
cat > "$SINGLE_EXAMPLE_DIR/validation.json" << 'EOF'
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"caption": "A large metal press is shown compressing a pile of colorful macarons, flattening them as if they were under a hydraulic press. The press moves down, crushing the macarons into a pile of crumbs and squishing the colorful filling out.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 81
|
||||
}
|
||||
]
|
||||
}
|
||||
EOF
|
||||
|
||||
echo "Single example dataset created at $SINGLE_EXAMPLE_DIR"
|
||||
echo "Contains:"
|
||||
echo "- 1 video: $(cat $SINGLE_EXAMPLE_DIR/videos.txt)"
|
||||
echo "- 1 prompt: $(cat $SINGLE_EXAMPLE_DIR/prompt.txt)"
|
||||
echo "- Validation file created with the same example for consistency"
|
||||
@@ -1,24 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
GPU_NUM=1 # 2,4,8
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
MODEL_TYPE="wan"
|
||||
DATA_MERGE_PATH="data/crush-smol/merge.txt"
|
||||
OUTPUT_DIR="data/crush-smol_processed_t2v/"
|
||||
|
||||
torchrun --nproc_per_node=$GPU_NUM \
|
||||
fastvideo/pipelines/preprocess/v1_preprocess.py \
|
||||
--model_path $MODEL_PATH \
|
||||
--data_merge_path $DATA_MERGE_PATH \
|
||||
--preprocess_video_batch_size 8 \
|
||||
--seed 42 \
|
||||
--max_height 480 \
|
||||
--max_width 832 \
|
||||
--num_frames 81 \
|
||||
--dataloader_num_workers 0 \
|
||||
--output_dir=$OUTPUT_DIR \
|
||||
--train_fps 16 \
|
||||
--samples_per_file 8 \
|
||||
--flush_frequency 8 \
|
||||
--video_length_tolerance_range 5 \
|
||||
--preprocess_task "t2v"
|
||||
@@ -1,29 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
GPU_NUM=1 # 2,4,8
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
MODEL_TYPE="wan"
|
||||
DATA_MERGE_PATH="data/crush-smol-single/merge.txt"
|
||||
OUTPUT_DIR="data/crush-smol-single_processed_t2v/"
|
||||
|
||||
torchrun --nproc_per_node=$GPU_NUM \
|
||||
fastvideo/pipelines/preprocess/v1_preprocess.py \
|
||||
--model_path $MODEL_PATH \
|
||||
--data_merge_path $DATA_MERGE_PATH \
|
||||
--preprocess_video_batch_size 1 \
|
||||
--seed 42 \
|
||||
--max_height 480 \
|
||||
--max_width 832 \
|
||||
--num_frames 81 \
|
||||
--dataloader_num_workers 0 \
|
||||
--output_dir=$OUTPUT_DIR \
|
||||
--train_fps 16 \
|
||||
--samples_per_file 1 \
|
||||
--flush_frequency 1 \
|
||||
--video_length_tolerance_range 5 \
|
||||
--preprocess_task "t2v"
|
||||
|
||||
# Copy the validation.json to the output directory for consistency
|
||||
cp "data/crush-smol-single/validation.json" "$OUTPUT_DIR/"
|
||||
|
||||
echo "Preprocessing completed. Validation file copied to $OUTPUT_DIR/"
|
||||
@@ -1,31 +0,0 @@
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"caption": "A large metal cylinder is seen pressing down on a pile of Oreo cookies, flattening them as if they were under a hydraulic press.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A large metal cylinder is seen compressing colorful clay into a compact shape, demonstrating the power of a hydraulic press.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A large metal cylinder is seen pressing down on a pile of colorful candies, flattening them as if they were under a hydraulic press. The candies are crushed and broken into small pieces, creating a mess on the table.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -86,7 +86,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
--validation_guidance_scale "1.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -102,6 +102,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
|
||||
@@ -86,7 +86,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
--validation_guidance_scale "1.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -102,6 +102,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
|
||||
@@ -86,7 +86,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
--validation_guidance_scale "1.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -102,6 +102,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
|
||||
@@ -1,13 +0,0 @@
|
||||
# Wan2.2-5B Distill Example
|
||||
These are end-to-end example scripts for distilling Wan2.2 TI2V 5B model DMD+VSA methods.
|
||||
|
||||
### 0. Make sure you have installed VSA
|
||||
|
||||
```bash
|
||||
cd csrc/attn
|
||||
git submodule update --init --recursive
|
||||
python setup_vsa.py install
|
||||
```
|
||||
|
||||
### Data-free Distillation
|
||||
When `--simulate_generator_forward` is enabled, distillation becomes data-free by simulating intermediate steps through forward inference of the generator. This helps avoid training–inference mismatch. See Section 4.5 of [DMD2](https://arxiv.org/pdf/2405.14867) for details.
|
||||
@@ -1,144 +0,0 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=t2v
|
||||
#SBATCH --partition=main
|
||||
#SBATCH --nodes=8
|
||||
#SBATCH --ntasks=8
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=128
|
||||
#SBATCH --mem=1440G
|
||||
#SBATCH --output=dmd_Wan2.2/t2v_g2e5_f1e5_%j.out
|
||||
#SBATCH --error=dmd_Wan2.2/t2v_g2e5_f1e5_%j.err
|
||||
#SBATCH --exclusive
|
||||
set -e -x
|
||||
|
||||
# Environment Setup
|
||||
source ~/conda/miniconda/bin/activate
|
||||
conda activate your_env
|
||||
|
||||
# Basic Info
|
||||
export WANDB_MODE="online"
|
||||
export NCCL_P2P_DISABLE=1
|
||||
export TORCH_NCCL_ENABLE_MONITORING=0
|
||||
# different cache dir for different processes
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
|
||||
export MASTER_PORT=29500
|
||||
export NODE_RANK=$SLURM_PROCID
|
||||
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
|
||||
export MASTER_ADDR=${nodes[0]}
|
||||
export CUDA_VISIBLE_DEVICES=$SLURM_LOCALID
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export FASTVIDEO_ATTENTION_BACKEND=FLASH_ATTN
|
||||
export WANDB_API_KEY=your_wandb_api_key
|
||||
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
|
||||
|
||||
echo "MASTER_ADDR: $MASTER_ADDR"
|
||||
echo "NODE_RANK: $NODE_RANK"
|
||||
|
||||
# Configs
|
||||
NUM_GPUS=8
|
||||
MODEL_PATH="Wan-AI/Wan2.2-TI2V-5B-Diffusers"
|
||||
DATA_DIR=your_data_dir
|
||||
VALIDATION_DIR=your_validation_path #(example:validation_64.json)
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name Wan_distillation
|
||||
--output_dir "your_output_dir"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 31
|
||||
--num_height 704
|
||||
--num_width 1280
|
||||
--num_frames 121
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 64
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DIR"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 2e-5
|
||||
--lr_scheduler "cosine_with_min_lr"
|
||||
--min_lr_ratio 0.5
|
||||
--lr_warmup_steps 100
|
||||
--fake_score_learning_rate 1e-5
|
||||
--fake_score_lr_scheduler "cosine_with_min_lr"
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3
|
||||
--simulate_generator_forward
|
||||
--log_visualization # disable if oom
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
--node_rank $SLURM_PROCID \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
@@ -1,145 +0,0 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=t2v
|
||||
#SBATCH --partition=main
|
||||
#SBATCH --nodes=8
|
||||
#SBATCH --ntasks=8
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=128
|
||||
#SBATCH --mem=1440G
|
||||
#SBATCH --output=dmd_Wan2.2/t2v_g2e5_f1e5_%j.out
|
||||
#SBATCH --error=dmd_Wan2.2/t2v_g2e5_f1e5_%j.err
|
||||
#SBATCH --exclusive
|
||||
set -e -x
|
||||
|
||||
# Environment Setup
|
||||
source ~/conda/miniconda/bin/activate
|
||||
conda activate your_env
|
||||
|
||||
# Basic Info
|
||||
export WANDB_MODE="online"
|
||||
export NCCL_P2P_DISABLE=1
|
||||
export TORCH_NCCL_ENABLE_MONITORING=0
|
||||
# different cache dir for different processes
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
|
||||
export MASTER_PORT=29500
|
||||
export NODE_RANK=$SLURM_PROCID
|
||||
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
|
||||
export MASTER_ADDR=${nodes[0]}
|
||||
export CUDA_VISIBLE_DEVICES=$SLURM_LOCALID
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN
|
||||
export WANDB_API_KEY=your_wandb_api_key
|
||||
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
|
||||
|
||||
echo "MASTER_ADDR: $MASTER_ADDR"
|
||||
echo "NODE_RANK: $NODE_RANK"
|
||||
|
||||
# Configs
|
||||
NUM_GPUS=8
|
||||
MODEL_PATH="Wan-AI/Wan2.2-TI2V-5B-Diffusers"
|
||||
DATA_DIR=your_data_dir
|
||||
VALIDATION_DIR=your_validation_path #(example:validation_64.json)
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name Wan_distillation
|
||||
--output_dir "your_output_dir"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 31
|
||||
--num_height 704
|
||||
--num_width 1280
|
||||
--num_frames 121
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 64
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DIR"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 2e-5
|
||||
--lr_scheduler "cosine_with_min_lr"
|
||||
--min_lr_ratio 0.5
|
||||
--lr_warmup_steps 100
|
||||
--fake_score_learning_rate 1e-5
|
||||
--fake_score_lr_scheduler "cosine_with_min_lr"
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3
|
||||
--simulate_generator_forward
|
||||
--log_visualization # disable if oom
|
||||
--VSA_sparsity 0.8
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
--node_rank $SLURM_PROCID \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
@@ -1,516 +0,0 @@
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
|
||||
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
|
||||
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
|
||||
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
|
||||
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
|
||||
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
|
||||
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
|
||||
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
|
||||
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
|
||||
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
|
||||
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
|
||||
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
|
||||
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
|
||||
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
|
||||
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
|
||||
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
|
||||
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "An expansive view of a calm bay reveals a fleet of sailboats, each anchored in a regimented line stretching toward the horizon. The water is a serene blue, reflecting the soft hues of the early morning sky. A gentle breeze is indicated by the subtle ripples trailing behind the boats, while a single, larger vessel cuts a distinct path, leaving a graceful wake in its journey to the open sea. On one side, a cluster of modern high-rise buildings stands, contrasting against the natural simplicity of the water, suggesting a blend of urban and marine life. The distant shoreline is barely visible, softened by the atmospheric perspective, giving a sense of endless waters meeting the sky. The overall mood is peaceful and orderly, with the boats appearing almost as sentinels guarding the expanse of the tranquil bay.",
|
||||
"video_path": "beach/mixkit-flying-backwards-over-the-sea-near-a-coast-50187_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a person is standing in the center of a dark, featureless space, illuminated by a spotlight that emphasizes their presence. The individual is dressed in a traditional martial arts uniform, known as a gi, which is predominantly white with a black belt tied around the waist, indicating a high level of expertise. The background remains pitch black, creating a stark contrast with the brightly lit figure, ensuring complete focus on them. The person's expression is serious and focused, reflecting a deep sense of discipline and concentration. Their hands move gracefully, transitioning through various martial arts stances, demonstrating practiced skill and fluidity. The uniform's crisp fabric folds and subtly reflects the light, further highlighting each precise movement. Despite the simplicity of the environment, the scene is dynamic, with each motion capturing the essence of martial arts practice. The video effectively conveys a sense of calm strength and mastery, making it ideal for an AI to recreate with attention to posture, lighting, and attire.",
|
||||
"video_path": "Sport/mixkit-karate-fighter-bowing-to-the-front-49706_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a dimly lit room bathed in a mix of neon purple and blue lights, a focused individual is seated in a gaming chair. She wears a white hoodie and large headphones with cat ears that glow softly, creating a striking silhouette. Her hands rest on a keyboard, typing swiftly as she concentrates intently on the screen in front of her. The atmosphere exudes a sense of intensity and immersion, with the soft-colored lighting enhancing the futuristic vibe. Her long hair cascades down her shoulders, adding a touch of elegance to the otherwise tech-centric setting. The overall scene captures the essence of a dedicated gamer deeply engaged in her virtual world.",
|
||||
"video_path": "earth/mixkit-a-young-woman-wearing-headphones-with-rgb-lights-suddenly-gets-51621_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "Inside a dimly-lit bus, five individuals are seated along the rows of worn seats, each subtly illuminated by the colorful lights emanating from overhead. On the left, a woman sits with a relaxed posture, her curly hair accented by a patterned scarf, wearing a plaid outfit paired with bright neon socks. Next to her, a person clad in a denim jacket appears deep in thought, resting their head on a hand. Further back, another figure in a bucket hat and oversized yellow attire gazes across the aisle, evoking a sense of introspection. The atmosphere is enriched by the soft glow of red and green lights, bathing the bus interior in an almost surreal ambiance, creating a compelling tableau of urban life.",
|
||||
"video_path": "Music/mixkit-conceptual-urban-fashion-42581_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "An aerial view captures two tennis players on a court, with one dressed in white on the left and another in red on the right. They are mid-game, each poised for action with rackets in hand, accentuated by their strategic positioning at opposite baselines. The court itself is a stark, deep blue, bordered by the vibrant green of the surrounding area, with a dark central net dividing the space. Long shadows stretch dramatically across the ground, suggesting a late afternoon setting. The subtly textured surface of the court contrasts with the crisp, white lines marking its boundaries and sections. This scene creates a vivid, balanced composition, highlighting both the competitive tension and serene atmosphere of the game.",
|
||||
"video_path": "People/mixkit-two-people-playing-tennis-aerial-view-880_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a vibrant, dreamlike setting, a lone figure moves energetically against a backdrop of deep blue and purple hues, casting emotive shadows that ripple with dynamic motion. The figure, almost obscured by a smeared effect, suggests a rhythmic dance or a passionate performance, arms blurred as they sweep through colorful, streaked lighting. A neon glow accentuates their form, particularly highlighting the face which is abstractly illuminated in bursts of orange and red, suggesting intense emotional expression. The scene is dominated by two primary elements \u2013 the figure\u2019s motion and the dramatic lighting, creating a synergy of human emotion and visual spectacle. Swirling trails of light seem to intertwine with the figure, like a visual symphony of movement and color that floods the space. The lighting changes, casting intricate patterns on the figure and the surrounding space, giving the impression of a kaleidoscope in motion. Despite the blurred and abstract portrayal, there is a sense of focus conveyed through the figure\u2019s intent movements, akin to a conductor orchestrating a visual and auditory performance. The environment resonates with an electric energy, suggesting a seamless fusion of art and technology. As the visual drama unfolds, the scene invites viewers to lose themselves in the abstract dance and the play of vivid luminance.",
|
||||
"video_path": "Music/mixkit-dancer-dancing-with-a-light-bar-in-his-hands-42221_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a brightly lit studio, a photographer wearing a denim jacket focuses intently, capturing shots with a professional camera. Facing him, a model stands gracefully, adjusting her long, flowing hair with delicate movements. The scene is characterized by strong contrasts; the model's soft pink attire and gentle gestures complement the rugged, precise demeanor of the photographer. Positioned against a minimalist backdrop, the pair work seamlessly, with the camera\u2019s lens pointed directly at the model, capturing her elegance. The soft, diffused lighting casts a gentle glow on both subjects, creating an airy and ethereal atmosphere perfect for a high-fashion photo shoot.",
|
||||
"video_path": "Fashion/mixkit-professional-photo-session-with-a-young-female-model-41621_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The video showcases a serene, expansive landscape covered with a variety of trees dotting the hills. The hills gently slope across the frame, with patches of dry grass contrasting against the lush green foliage. Tall trees with dense canopies stand elegantly, casting soft shadows on the ground below. The sunlight bathes the entire scene, highlighting the varied textures of the leaves and terrain. Gaps between the trees reveal a narrow dirt path meandering through the hills, suggesting a sense of quiet solitude. The undulating hills extend into the distance, creating depth and a calming sense of vast space. The verdant hues of the leaves contrast with the earthy tones of the hills, enhancing the visual richness. In the background, a faint outline of distant hills can be seen, blurred softly by the atmospheric perspective. This tranquil setting could be efficiently recreated in a virtual environment by focusing on its layered composition, color palette, and natural textures.",
|
||||
"video_path": "forest/mixkit-aerial-panorama-of-a-sunny-mountain-landscape-40846_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A bustling ski slope comes alive with skiers descending a pristine, snow-covered hill, surrounded by towering, snow-draped evergreens. Several figures stand atop the slope, silhouetted against a clear blue sky, preparing to embark on their ski run. The chair lift on the right continuously drops off eager adventurers, adding to the excitement at the hilltop. Each skier, clad in colorful winter gear, carves distinct paths into the textured snow as they weave their way down. The interplay of sunlight and shadows accentuates the myriad tracks etched into the slope, creating a dynamic visual rhythm. The scene captures a vibrant winter wonderland, full of action and the thrill of a perfect ski day.",
|
||||
"video_path": "Car/mixkit-skiers-on-a-snowy-slope-3327_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The scene unfolds within a dimly lit bus, where three young individuals are seated, each absorbed in their unique world. To the left, a person with tied-back hair rests their head on their hand, dressed casually in a jacket and jeans, projecting a relaxed demeanor. Central to the frame is another individual, sitting upright with intense focus, donning a plaid blazer and oversize hoops, enhancing their confident presence. The muted green and red lighting casts an atmospheric glow, adding depth and intrigue to the setting. On the right, a person in a bucket hat and striped shirt leans back, appearing contemplative as they adjust their hat with a nonchalant gesture. The interplay of light and shadow highlights their expressions, creating an intimate and cinematic ambiance. Together, these figures form a cohesive tableau, capturing a moment of introspection amid a bustling yet serene urban environment.",
|
||||
"video_path": "City/mixkit-three-models-posing-to-the-lens-while-on-board-a-42575_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A determined climber is scaling a massive rock face, showcasing exceptional strength and skill. The person, clad in a teal shirt and dark pants, climbs with precision, their movements measured and deliberate. They are secured by climbing gear, which includes ropes and a harness, emphasizing their commitment to safety. The rugged texture of the sandy-colored rock provides an imposing backdrop, adding drama and scale to the climb. In the distance, other large rock formations and sparse vegetation can be seen under a bright, overcast sky, contributing to the natural and adventurous atmosphere. The scene captures a moment of focus and challenge, highlighting the climber's tenacity and the breathtaking environment.",
|
||||
"video_path": "Sport/mixkit-alpinist-climbing-a-huge-rock-in-a-desert-43306_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A woman stands confidently in front of a large array of solar panels, her navy blue jumpsuit contrasting against the lush green grass beneath her feet. Her expression is calm and focused, eyes facing directly ahead, suggesting a deep connection to the subject matter\u2014renewable energy. The sunlight bathes the scene in warm hues, casting gentle shadows and highlighting the geometric precision of the solar panels' grid-like structure. The background reveals a blend of nature and technology, as the panels are anchored on a grassy slope with foliage on the left side of the frame. This composition captures a harmonious blend of human innovation and environmental consciousness, accentuated by the serene outdoor setting.",
|
||||
"video_path": "Business/mixkit-woman-standing-in-front-of-a-solar-panel-4880_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, two people are working at a wooden desk, using an iMac computer. One person, wearing a white knit sweater, is using the apple wireless mouse with their right hand, while their left hand rests on the sleek white keyboard. Their movements are smooth yet intentional, suggesting they are focused on a task on the computer screen. The monitor displays a well-organized array of files and folders, hinting at a task that involves detailed organization or detailed data navigation. The second person, only subtly visible, sits closely by and appears to observe or assist, creating a collaborative atmosphere. Their presence adds a quiet dynamic to the scene, as if they are ready to provide input or guidance. Sticky notes with handwritten notes are attached to the monitor\u2019s stand, adding a touch of personal organization amidst the digital workspace. The focus on the keyboard and mouse emphasizes a streamlined workflow, indicative of a productive work environment. The overall ambiance is calm and focuses on teamwork, technology, and efficient workspace management.",
|
||||
"video_path": "People/mixkit-person-with-glasses-working-on-a-desktop-computer-3248_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A man stands in front of a modern glass facade, taking off a dark hoodie to reveal his gray tank top underneath. His arms are lifted high as he maneuvers the hoodie over his head, showcasing a fluid motion that conveys a sense of calm and routine. The lighting highlights the contours of his muscles, emphasizing a combination of strength and quiet determination. Behind him, the reflective surface of the glass panels provides a subtle backdrop, enhancing the focus on his focused and serene demeanor.",
|
||||
"video_path": "Sport/mixkit-man-puts-on-sleeveless-hoodie-603_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The video displays a captivating dance of fiery orange flames against a stark black background, creating an intense visual contrast. The flames twist and intertwine, forming symmetrical, swirling patterns that expand and contract rhythmically across the frame. Each fiery tendril seems to be alive, moving with an almost hypnotic fluidity that captures the viewer's attention. The illumination from the flames casts subtle shadows, enhancing the depth and texture of the scene. Overall, the dynamic movement and vibrant color palette create an atmosphere of both beauty and power.",
|
||||
"video_path": "fire/mixkit-two-orange-flames-on-black-background-685_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In this scene, a person is seated in a dimly lit room, possibly a recording studio, holding several drumsticks in their hands. The individual's face is partially obscured by sunglasses, adding a touch of mystery to their demeanor. They are wearing a colorful, patterned shirt with a mix of orange and blue tones that stands out against the darker background. The person appears focused and engaged with the drumsticks, their hands prominently displayed. The ambient light casts warm, soft shadows, emphasizing the texture and colors of their shirt and the wooden drumsticks. The room features wooden paneling, which complements the overall cozy, music-centric setting of the scene. The use of perspective centers on the drumsticks, highlighting the importance of rhythm and music in the captured moment.",
|
||||
"video_path": "Music/mixkit-drummer-stretching-before-playing-42783_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A man is casually sitting on a sofa, engrossed in his meal and entertainment. He is holding a TV remote in one hand while reaching for food with the other, indicating a laid-back, comfortable evening. The table before him is filled with takeout containers, revealing a variety of appetizers and dishes, suggestive of a casual dining experience at home. The background is defined by colorful patterned cushions, adding a cozy, homey feel to the scene. Warm, ambient lighting highlights the relaxed atmosphere, casting soft shadows that contribute to the intimate setting. In this moment, he takes a bite of a sandwich, comfortably balancing his attention between food and whatever is playing on the screen.",
|
||||
"video_path": "Man/mixkit-man-watching-tv-and-eating-fast-food-26089_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The scene opens to a breathtaking view of a tranquil ocean horizon at dusk, displaying a vibrant tapestry of oranges, pinks, and purples as the sun sets. In the foreground, tall, swaying palm trees frame the scene, their silhouettes stark against the colorful sky. The ocean itself shimmers with reflections of the sunset, creating a peaceful, almost ethereal atmosphere. A small boat can be seen in the distance, centered on the horizon, adding a sense of scale and solitude to the scene. The waves gently lap the shore, creating faint patterns on the sandy beach, which stretches across the foreground. Above, the sky is dotted with scattered clouds that catch the last light of the day, enhancing the drama and beauty of the scene. The overall mood is serene and contemplative, capturing a perfect moment of nature\u2019s grandeur.",
|
||||
"video_path": "beach/mixkit-sunset-with-sailing-boats-2166_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A man sits hunched on a couch, the weight of emotions clearly visible on his posture. He wears a simple, gray t-shirt, and his head is bowed, resting in his hands, which cover most of his face, obscuring his features. The gentle light filtering through sheer curtains in the background casts a soft glow upon him, emphasizing the contrast between his static form and the hazy brightness behind. His elbows rest upon his knees, suggesting a posture of deep contemplation or distress. The simplicity of the room, with its muted colors, highlights the focus on the man's internal struggle. Delicate detailing on the fabric of his shirt adds texture, enhancing the scene's realism. Subtle changes in the natural light indicate the passage of time, as the man remains unmoving, absorbed in thought. This intimate moment captures a profound vulnerability, making the scene universally relatable and poignant.",
|
||||
"video_path": "Man/mixkit-worried-and-sad-man-with-his-head-down-4701_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A pair of hands, belonging to an unseen figure, carefully unrolls a large sheet of crisp, white paper on a dark wooden table. The lighting is warm, casting a gentle glow that highlights the textures of the paper and the wood grain of the table. As the paper unfurls, the edges reveal the faint beginnings of a colorful map printed on its surface. The arms, clad in a casual gray T-shirt, suggest a relaxed and focused task at hand. Each motion is deliberate, with fingers deftly guiding the paper, ensuring it lays flat without creases. In the background, a hint of a red curtain can be seen, adding a touch of color and depth to the setting. The composition of the scene emphasizes the contrast between the bright paper and the rich tones of the surroundings. This serene and methodical action evokes a sense of exploration and preparation.",
|
||||
"video_path": "Man/mixkit-unrolling-a-world-map-on-a-table-21626_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A young woman sits on a vibrant green seat inside a bus, illuminated by the soft glow of pink and blue lights. Her outfit is a striking mix of colors: a neon pink top paired with a jacket featuring dark sleeves, and jeans that provide a neutral contrast. She wears large, hoop earrings that catch the light as she moves slightly, exuding an air of cool confidence. Her gaze is directed thoughtfully to the side, suggesting contemplation or daydreaming during her commute. The metallic pole beside her adds a geometric element to the composition, reflecting the kaleidoscope of neon hues. The background is a clean, futuristic white, serving as a blank canvas that amplifies the neon atmosphere. Her relaxed posture and the modern bus setting create a scene that captures a blend of urban life and personal introspection.",
|
||||
"video_path": "City/mixkit-fashion-model-posing-on-a-bus-42578_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A silver SUV drives along a winding, snow-covered mountain road, with dense pine trees blanketed in snow lining both sides. The scene is serene, with the vehicle moving smoothly, possibly on a winter journey or vacation. As the SUV disappears around the bend, another, darker SUV follows, creating a sense of motion and perspective on the snow-dusted asphalt. The towering, snow-laden rock formation to the right contrasts with the dark green of the pines, highlighting the peacefulness of the wintry landscape.",
|
||||
"video_path": "Car/mixkit-curve-on-a-snowy-forest-road-3317_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The video showcases a vibrant urban skyline during twilight, with towering buildings reflecting the warm hues of the setting sun. A series of tall, cylindrical structures dominate the foreground, adjacent to a complex of industrial equipment and grids. The scene includes modern high-rise buildings with glass exteriors, capturing the evolving architecture of a bustling cityscape. A prominent structure labeled \"CITY OF AUSTIN POWER PLANT\" stands out, highlighting the industrial theme amidst the urban backdrop. The soft glow of city lights begins to pierce the approaching dusk, creating an inviting yet dynamic atmosphere. Shadows cast by the buildings add depth and contrast, emphasizing their massive scale and intricate designs. The overall composition is balanced between the natural light of the sunset and the artificial illumination of the city, offering a compelling visual narrative.",
|
||||
"video_path": "Car/mixkit-slow-air-travel-in-reverse-over-a-big-city-49841_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the scene, a striking architectural structure dominates the view, bathed in a soft, ambient light. The enormous yellow arches serve as the centerpiece, drawing the eye upwards with their majestic curves and towering presence. The smooth, clean surfaces of the structure reflect the light, highlighting the texture and depth of the architecture. In the foreground, blurred streaks of headlights and taillights suggest the motion of vehicles passing by, adding dynamic energy to the otherwise still scene. The contrast between the fast-moving lights and the static arches creates a balanced composition. To the left, a lone streetlamp and a small tree provide a touch of nature and urban elements against the monumental backdrop. The night sky subtly peeks through the gaps in the structure, hinting at a clear, calm evening. Shadows from the arches create patterns on the ground, adding an intricate detail to the scene. Overall, the combination of light, shadow, and movement makes for a dramatic and visually captivating moment.",
|
||||
"video_path": "Car/mixkit-a-fast-timelapse-of-the-street-with-a-monumental-yellow-50993_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A tranquil marina comes into full view under the golden hues of a setting sun. A collection of gleaming yachts and boats are neatly moored, their reflections shimmering softly on the gentle water. The sun's low position casts elongated shadows over the bustling harbor scene, while rolling hillsides surround the distant cityscape. The skyline is interspersed with modern buildings and clusters of residences, adding layers to the vibrant community. At the center, a broad wooden pier juts confidently into the harbor, extending an invitation for leisurely strolls. To the left, various shops and colorful structures line the waterfront, indicating a vibrant coastal economy. The entire atmosphere exudes a serene yet lively charm, balancing the hustle of maritime activity with the peacefulness of the encroaching dusk. It's a scene of calm anticipation, as if the whole place holds its breath before the night's events unfold.",
|
||||
"video_path": "beach/mixkit-harbor-on-a-tourist-coast-with-many-boats-and-yachts-40077_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The video features a confident individual standing atop a structure against a clear blue sky, exuding a sense of freedom and style. The person is clad in a striking yellow button-up shirt tied at the waist, and beneath it, they wear a simple white top that adds to their relaxed yet stylish appearance. Completing the ensemble are high-waisted white jeans paired with a black belt, adding a touch of contrast. Around their neck is a bold red scarf, providing a splash of color and an air of vintage flair. The person's sunglasses, tinted in yellow, reflect the sunlight and contribute to the overall cool and composed demeanor. Their hair is styled elegantly, pulled back with headphones resting over the ears, suggesting they are immersed in music. One hand casually grazes the headphones, while the other rests gently on the railing, grounding the individual in the moment. The scene is an effortless blend of fashion and tranquility, capturing the spirit of sunny, carefree days.",
|
||||
"video_path": "Music/mixkit-standing-woman-listening-to-music-460_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A ballerina gracefully spins and moves across a pink-hued studio, her poised figure accentuated by a shimmering white tutu and bodice. The background, a continuous wash of soft pink, provides a serene and ethereal atmosphere, emphasizing her fluid movements. Her arms extend with elegance, highlighting the delicacy and precision of her ballet pose, while her focused expression adds intensity to the scene. The subtle details of her costume, combined with the pink monochromatic ambiance, create a dreamlike spectacle, ideal for an AI to envision a oneiric dance setting.",
|
||||
"video_path": "Dance/mixkit-portrait-of-a-ballerina-spinning-with-pink-background-40163_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "The scene unfolds with two human figures in the distance, making their way through a serene meadow, thick with tall golden grass swaying gently in the breeze. The sun hangs low in the sky, casting a soft, diffused glow that illuminates the landscape with a warm, ethereal light. These figures, clad in hiking gear, move deliberately, suggesting they're either embarking on or concluding a journey. Their silhouettes contrast against the lush greenery of the surrounding trees, whose branches reach out, framing the horizon. The play of light and shadow among the trees creates a quilt of textures, with each leaf catching a hint of the sun's dying rays. This tranquil setting evokes a sense of calm and adventure, capturing the quintessential beauty of nature\u2019s landscape.",
|
||||
"video_path": "People/mixkit-landscape-in-nature-while-two-people-are-jogging-44348_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A large cargo ship is docked at an industrial port, its white superstructure contrasting with the deep green and yellow of its deck. The foreground is dominated by the calm, deep blue waters of the harbor, which reflect the vessel\u2019s imposing presence. Surrounding the ship, a series of industrial buildings and storage facilities are visible, hinting at the bustling activity of the port. The deck is intricately detailed, featuring an array of pipes, equipment, and railings, showcasing the ship's functionality and purpose. In the background, a paved area with green patches and a few parked vehicles adds to the busy, industrious atmosphere of the scene.",
|
||||
"video_path": "sea/mixkit-empty-cargo-ship-waiting-at-the-port-4209_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A lone climber ascends a towering rock face, clad in a pink shirt and gray pants, displaying a determined and focused expression. The climber navigates the rugged surface, where the texture of the rock is peppered with natural pockets and crevices that offer handholds and footholds. Sunlight casts soft shadows across the cliff, highlighting the intricate patterns and the climber\u2019s strategic movements. The cliff looms high, with sparse vegetation breaking the monotony of the stone, while distant rocky formations form a dramatic backdrop against the clear blue sky. The climber\u2019s gear, including a harness and chalk bag, underscores the adventure and challenge woven into this majestic, vertical journey.",
|
||||
"video_path": "Sport/mixkit-mountaineer-girl-climbing-a-steep-rocky-mountain-41089_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A person is seen in a close-up shot, skillfully adjusting the tuning pegs of a guitar, showcasing a focused and practiced hand. The image is in black and white, highlighting the contrast between the textures of the instrument and the clothing. The individual's shirt, visible in the background, adds a soft, subtle texture, while the dark tones of the guitar neck create depth in the scene. This composition captures a moment of concentration and finesse, perfect for recreating an intimate musical setting.",
|
||||
"video_path": "Music/mixkit-guitarist-playing-so-inspired-black-and-white-shot-44178_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A musician is playing a large brass instrument with the words \"Brass Band\" clearly visible on its bell. The scene is set against a vibrant yellow backdrop, casting a warm glow on the subject. The musician wears a dark cap and a matching suit, adding a formal touch to his attire. He is deeply focused on his performance, with the instrument's intricate tubing adding complexity to the visual composition. The lighting creates dramatic shadows and highlights, emphasizing the musician's expression and the instrument's metallic sheen. This harmonious blend of color and form captures the essence of a live brass band performance.",
|
||||
"video_path": "Music/mixkit-musician-playing-the-trombone-while-dancing-43752_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a lone musician stands gracefully in front of a grand cathedral, playing an accordion while surrounded by the lively water display of a central fountain. Dressed in a casual ensemble, he wears a light-colored shirt, dark pants, and a flat cap that gives him a vintage charm. His posture is relaxed, yet engaged, as he sways gently in rhythm with the music, casting soft shadows on the cobblestone steps beneath him. The backdrop features the cathedral's towering twin spires, with intricate stonework that casts a rich, historical aura around the scene. Sunlight bathes the entire setting, enhancing the golden hues of the cathedral facade and creating a halo-like effect around the musician. The fountain's water jets splash playfully, catching glimmers of light and adding a dynamic element to the tranquil atmosphere. The scene captures a harmonious blend of architectural majesty and human creativity, framed by the clear, azure sky that extends infinitely above. It's a vivid depiction of solitude and artistry, set against a timeless urban landscape.",
|
||||
"video_path": "Music/mixkit-man-plays-an-accordion-in-front-of-a-fountain-630_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the tranquil video, a person sits in a meditative pose on a gentle hillside, silhouetted against the dawning sky. The person is facing the breathtaking sunrise, with their back slightly turned to the viewer, wearing a simple, light-colored shirt. Their right hand rests on their knee, fingers relaxed in a common meditation mudra, symbolizing calmness and peace. The sky, a stunning blend of soft oranges and deep purples, gradually brightens, casting a warm glow over the lush, green landscape. To the left, the outlines of distant urban buildings can be seen against the horizon, adding a contrast between nature and city life. A river reflecting the sky's colors meanders through the scene, lending a serene, flowing dynamic to the landscape. Trees rise and fall gently across the terrain, their leaves rustling only faintly in the morning breeze. The person remains still and focused, embodying a moment of mindfulness and connection with nature. This visual captures a harmonious balance, evoking a sense of tranquility and introspection.",
|
||||
"video_path": "City/mixkit-girl-meditating-in-yoga-pose-at-sunset-4803_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A serene landscape video captures a breathtaking panoramic view of a vast valley covered in a gentle mist. The undulating hills are lush with dense greenery, their rich foliage creating a vibrant border on the left side of the frame. The mist weaves through the landscape like a soft, ethereal blanket, lending a dream-like quality to the scene. In the distance, several mountain peaks emerge, their dark outlines contrasting against the pale blue sky. A few faint, wispy clouds drift lazily across the horizon, complementing the tranquil atmosphere. The sunlight filters through the haze, casting a warm glow and highlighting different textures of the flora. The overall mood is calm and contemplative, inviting the viewer to pause and appreciate nature's untouched beauty. The composition emphasizes depth and expansiveness, drawing attention to the harmony between earth and sky. This captivating scene embodies tranquility, offering a perfect backdrop for meditation or relaxation.",
|
||||
"video_path": "forest/mixkit-flying-over-a-hill-with-a-view-of-the-surrounding-49743_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In this scene, a bearded individual is intently focused on their smartphone, with the sun setting in the background, casting a warm glow across the cityscape. The person, partially visible, is wearing a dark, buttoned shirt that contrasts with the golden hue of the sunset. Their hands are holding the smartphone delicately but purposefully, reflecting a sense of engagement and focus on the screen. The sunlight creates a striking lens flare effect, enhancing the dramatic atmosphere of the moment as it glimmers off the phone\u2019s surface. The surrounding environment hints at an elevated vantage point, providing a panoramic view of the urban landscape below.",
|
||||
"video_path": "City/mixkit-guy-texting-at-sunset-265_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In an expansive, industrial space defined by towering columns and high ceilings, a solitary figure takes center stage. The person, dressed in dark, fitted clothing, assumes a powerful, dynamic stance with one leg bent forward and both arms outstretched in a horizontal arc. Framing this pose are intense flames that engulf their arms, creating a striking visual contrast against the muted tones of the room. The fire forms a brilliant halo of orange and yellow, casting flickering shadows on the weathered walls and worn, tiled floor. This interplay between light and dark showcases the dancer's poise and agility, as they maintain balance amidst the intense heat. Windows line the background, their panes dimly illuminated by the daylight filtering in, adding depth and perspective to the scene. The entire performance evokes a sense of raw energy and elemental mastery, as the figure continues to manipulate the fire in a seamless, mesmerizing display.",
|
||||
"video_path": "fire/mixkit-expert-juggler-doing-tricks-with-a-stick-with-fire-43663_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A man is playing the violin, focused intently on his music. His fingers gracefully dance along the strings, flawlessly executing each note. He holds the violin close to his chin with a sense of familiarity and expertise. The rich, warm tones of the violin reflect in the soft lighting of the room. He wears a dark shirt, and a subtle necklace rests against his chest, adding a personal touch to his attire. The bow moves smoothly across the strings, producing a melody that seems to fill the space with emotion. His expression is one of concentration and passion, immersing himself fully in the performance. The background is softly blurred, bringing the violin's intricate craftsmanship and his precise movements into sharp focus. This serene and intimate moment captures the essence of his musical artistry.",
|
||||
"video_path": "Music/mixkit-fiddler-playing-a-song-639_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the dimly lit parking garage, two figures engage in an impromptu game of soccer. The first person, wearing a light grey shirt and black pants with three white stripes, skillfully maneuvers the ball with precise footwork. The ground is slick with patches of water, reflecting the vibrant neon lights above. A second figure, clad in dark clothing, stands poised in the background, ready to intercept. The space is defined by stark yellow lines and orange safety bollards, adding structure to the chaotic energy of the scene. The soccer ball glides smoothly across the wet floor, kicking up droplets as it passes. Despite the muted colors of the environment, the players' movements are dynamic and full of life. Their shadowy silhouettes dance with the reflecting light, creating a mesmerizing visual interplay. The atmosphere is charged with focus and camaraderie, encapsulating the essence of a late-night urban soccer experience.",
|
||||
"video_path": "Sport/mixkit-player-making-skillful-play-in-a-street-soccer-game-43504_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A lone climber is seen scaling a towering vertical rock face, demonstrating remarkable strength and focus. Dressed in a light-colored shirt and jeans, the climber grips the stone tightly, navigating the rough textures and crevices with precision. The sheer cliff is massive, exhibiting a range of natural hues from light tan to deep gray, accentuating the climber's figure against the vast rocky backdrop. Surrounding the cliff, scattered greenery and rugged terrain provide a sense of wilderness and isolation. The scene portrays a daring ascension requiring concentration and skill, capturing the essence of human endeavor against nature's formidable beauty.",
|
||||
"video_path": "Sport/mixkit-skilled-mountaineer-climbing-a-gigantic-mountain-41083_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In this serene landscape, a lush meadow stretches across the foreground, dotted with vibrant yellow wildflowers swaying gently in the breeze. A towering tree stands majestically on the right side, its branches reaching wide under the bright blue sky filled with fluffy white clouds. On the left, dense trees form a natural corridor leading to the horizon, suggesting a sense of journey and possibility. The richness of the green grass contrasts beautifully with the golden hue of the distant fields, creating a harmonious palette of nature\u2019s colors. The play of light and shadow adds depth and dimension, evoking a tranquil, inviting atmosphere. It's a scene where nature\u2019s beauty simply commands attention, offering a perfect escape into tranquility.",
|
||||
"video_path": "sky/mixkit-countryside-meadow-4075_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A solitary boat glides across the expansive, tranquil expanse of a serene lake. The vessel leaves a gentle wake behind, creating delicate ripples across the mirror-like surface. The water appears a rich shade of teal, seamlessly blending with the sky at the horizon. Silhouettes of distant trees are faintly visible, creating a picturesque backdrop that enhances the solitary journey of the boat. The sky is a calm gradient, shifting from soft oranges near the shore to the pale blues above. In the distance, a few slender poles emerge from the water, remnants of an old structure or natural formation. The mood of the scene is one of peace and solitude, with the boat journeying steadily through the quiet landscape. There is a sense of endless possibilities as the boat moves toward the unseen beyond the frame. The simplicity and stillness of the scene invite contemplation and reflection, encapsulating a perfect moment of quietude on the water.",
|
||||
"video_path": "mountain/mixkit-motorboat-on-a-large-lake-with-turquoise-blue-waters-4996_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a cozy, dimly lit caf\u00e9, a woman sits alone at a rustic wooden table, fully engrossed in her reading. Her dark, wavy hair frames her face as she leans forward over an open book, suggesting deep focus and contemplation. The caf\u00e9\u2019s ambiance is warm, with hanging pendant lights casting a soft glow over the wooden shelves lined with jars and coffee paraphernalia in the background. A small cup of coffee rests just within her reach, alongside a glass dome encasing a solitary pastry, adding a touch of tranquility to the scene. Her casual attire, a denim jacket over a simple shirt, complements the laid-back, comfortable setting of the caf\u00e9. The contrast between her concentrated expression and the bustling, yet subdued caf\u00e9 atmosphere creates a harmonious, serene visual. The overall composition captures a quiet moment of introspection amidst the gentle hum of caf\u00e9 life.",
|
||||
"video_path": "Woman/mixkit-woman-drinking-coffee-in-a-cafe-223_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a vast, deserted landscape under the night sky, a solitary figure stands at a small music setup, illuminated by strategically placed lights. The person is engrossed in playing a keyboard, with various electronic equipment surrounding them, casting soft glows of orange and blue hues across the scene. To the left, a large circular light adds a dramatic focal point, highlighting the intense contrast between the darkness and the lit performance area. This setup, with its minimalistic design and strategic lighting, creates a captivating and easily recognizable scene that merges the serene, expansive backdrop with an intimate, focused music performance.",
|
||||
"video_path": "Music/mixkit-talented-dj-playing-in-a-lonely-desert-42414_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In a bustling urban scene, cars zoom past a weathered building, their blurred motion a testament to the city\u2019s lively pace. The building, with its faded yellow and brown facade, boasts graffiti that speaks of both art and decay, framing the scene with an air of urban grit. A solitary figure stands slightly to the side, clad casually in a gray top and mustard trousers, gazing into the street, seemingly detached from the surrounding flurry. The motion of the traffic creates a dynamic contrast against the static backdrop, emphasizing the relentless movement of the city. As the video progresses, a bright yellow taxi appears, slowing down as it approaches the figure, adding a pop of color to the desaturated hues of the environment. The interaction suggests a routine, a possibly daily exchange between the driver and the pedestrian, hinting at the rhythms of city life. Overhead, a soft, overcast sky casts a diffused light, lending the scene a subdued, timeless quality. Small elements, like the vertical pole cutting through the frame and the distant chatter of urban sounds, complete this vivid tableau of urban existence.",
|
||||
"video_path": "Car/mixkit-morning-in-the-street-time-lapse-1648_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "A young woman sits on a curb in a tranquil park, basking in the golden hue of the setting sun. Beside her, a collie dog rests calmly, its fur illuminated by the warm sunlight, creating a serene glow. The woman's hand gently strokes the dog's back, highlighting the bond and affection between them. Tall trees surround the pair, casting elongated shadows on the leaf-laden ground, adding to the peaceful and intimate ambiance of the scene.",
|
||||
"video_path": "Pets/mixkit-a-woman-pets-a-dog-in-a-park-1562_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a grand, majestic elephant stands in an open, sunlit field, its massive form dominating the scene. The elephant's skin is a tapestry of earthy tones, with rough, textured wrinkles that add character to its already imposing presence. Its trunk, a powerful and flexible appendage, moves gently, swaying as the elephant possibly enjoys the warmth of the day. The background is a blur of greenery, suggesting a lively environment filled with trees and shrubs that provide a natural habitat. Light plays on the elephant's skin, highlighting patches of dust and dirt that give it an authentic wilderness look. The scene captures the tranquility and majesty of this gentle giant in its natural surroundings.",
|
||||
"video_path": "Zoo/mixkit-wet-elephant-in-the-savanna-3663_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a fluffy dog with brown patches is intently engaged with a bright red toy shaped like a fire hydrant, which has a yellow and orange rope attached. The dog's body is relaxed as it lies on a plain white background, concentrating on nudging and playfully biting the toy. Its ears perk up slightly with curiosity, and its eyes are fixated on the toy, suggesting a scene of focused playfulness. The neutral tones of the dog's fur contrast starkly against the vivid red of the toy, creating a visually striking moment.",
|
||||
"video_path": "Pets/mixkit-a-cute-border-collie-dog-play-with-a-fire-street-50662_clip_1.mp4",
|
||||
"num_inference_steps": 3,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 61
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -9,14 +9,4 @@ git submodule update --init --recursive
|
||||
python setup_vsa.py install
|
||||
```
|
||||
|
||||
### 1. Download dataset:
|
||||
```bash
|
||||
bash examples/distill/Wan2.2-TI2V-5B-Diffusers/crush_smol/download_dataset.sh
|
||||
```
|
||||
|
||||
### 2. Configure and run distillation:
|
||||
|
||||
#### For DMD-only distillation:
|
||||
```bash
|
||||
bash examples/distill/Wan2.2-TI2V-5B-Diffusers/crush_smol/examples/distill/Wan2.2-TI2V-5B-Diffusers/crush_smol/
|
||||
```
|
||||
### TODO
|
||||
@@ -63,7 +63,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
--validation_guidance_scale "1.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -77,6 +77,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
def main():
|
||||
@@ -8,29 +8,35 @@ def main():
|
||||
# model.
|
||||
# If a local path is provided, FastVideo will make a best effort
|
||||
# attempt to identify the optimal arguments.
|
||||
model_name = "Wan-AI/Wan2.2-TI2V-5B-Diffusers"
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
model_name,
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
dit_cpu_offload=True,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
# Set pin_cpu_memory to false if CPU RAM is limited and there're no frequent CPU-GPU transfer
|
||||
pin_cpu_memory=True,
|
||||
# image_encoder_cpu_offload=False,
|
||||
)
|
||||
|
||||
# sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
# sampling_param.num_frames = 45
|
||||
# sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
|
||||
sampling_param = SamplingParam.from_pretrained(model_name)
|
||||
sampling_param.image_path = "test.jpg"
|
||||
# sampling_param.num_inference_steps = 0
|
||||
# Generate videos with the same simple API, regardless of GPU count
|
||||
prompt = (
|
||||
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
)
|
||||
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True)
|
||||
i2v_prompt = "An astronaut hatching from an egg, on the surface of the moon, the darkness and depth of space realised in the background. High quality, ultrarealistic detail and breath-taking movie-like camera shot."
|
||||
i2v_prompt = "A little girl is packing a suitcase and the contents starts flying out of the suitcase everywhere."
|
||||
prompt = i2v_prompt
|
||||
# prompt = (
|
||||
# "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
# "wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
# "natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
# )
|
||||
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True, sampling_param=sampling_param)
|
||||
# video = generator.generate_video(prompt, sampling_param=sampling_param, output_path="wan_t2v_videos/")
|
||||
return
|
||||
|
||||
# Generate another video with a different prompt, without reloading the
|
||||
# model!
|
||||
|
||||
@@ -17,7 +17,7 @@ def main():
|
||||
use_fsdp_inference=True,
|
||||
# Adjust these offload parameters if you have < 32GB of VRAM
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
pin_cpu_memory=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
VSA_sparsity=0.8,
|
||||
@@ -31,6 +31,7 @@ def main():
|
||||
prompt = (
|
||||
"A neon-lit alley in futuristic Tokyo during a heavy rainstorm at night. The puddles reflect glowing signs in kanji, advertising ramen, karaoke, and VR arcades. A woman in a translucent raincoat walks briskly with an LED umbrella. Steam rises from a street food cart, and a cat darts across the screen. Raindrops are visible on the camera lens, creating a cinematic bokeh effect."
|
||||
)
|
||||
prompt = "A vintage train snakes through the mountains, its plume of white steam rising dramatically against the jagged peaks. The cars glint in the late afternoon sun, their deep crimson and gold accents lending a touch of elegance. The tracks carve a precarious path along the cliffside, revealing glimpses of a roaring river far below. Inside, passengers peer out the large windows, their faces lit with awe as the landscape unfolds."
|
||||
start_time = time.perf_counter()
|
||||
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True, sampling_param=sampling_param)
|
||||
end_time = time.perf_counter()
|
||||
@@ -45,7 +46,7 @@ def main():
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic.")
|
||||
start_time = time.perf_counter()
|
||||
video2 = generator.generate_video(prompt2, output_path=OUTPUT_PATH, save_video=True)
|
||||
video2 = generator.generate_video(prompt2, output_path=OUTPUT_PATH, save_video=False)
|
||||
end_time = time.perf_counter()
|
||||
gen_time2 = end_time - start_time
|
||||
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
import os
|
||||
import time
|
||||
from fastvideo import VideoGenerator, SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_causal"
|
||||
def main():
|
||||
# FastVideo will automatically use the optimal default arguments for the
|
||||
# model.
|
||||
# If a local path is provided, FastVideo will make a best effort
|
||||
# attempt to identify the optimal arguments.
|
||||
model_name = "wlsaidhi/SFWan2.1-T2V-1.3B-Diffusers"
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_name,
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
text_encoder_cpu_offload=False,
|
||||
dit_cpu_offload=False,
|
||||
)
|
||||
|
||||
sampling_param = SamplingParam.from_pretrained(model_name)
|
||||
|
||||
prompt = (
|
||||
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
)
|
||||
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True, sampling_param=sampling_param)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,48 +0,0 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_wan2_2_14B_t2v"
|
||||
def main():
|
||||
# FastVideo will automatically use the optimal default arguments for the
|
||||
# model.
|
||||
# If a local path is provided, FastVideo will make a best effort
|
||||
# attempt to identify the optimal arguments.
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.2-T2V-A14B-Diffusers",
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=True, # DiT need to be offloaded for MoE
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
# Set pin_cpu_memory to false if CPU RAM is limited and there're no frequent CPU-GPU transfer
|
||||
pin_cpu_memory=True,
|
||||
# image_encoder_cpu_offload=False,
|
||||
)
|
||||
|
||||
# sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
# sampling_param.num_frames = 45
|
||||
# sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
|
||||
# Generate videos with the same simple API, regardless of GPU count
|
||||
prompt = (
|
||||
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
)
|
||||
_ = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True, height=720, width=1280, num_frames=81)
|
||||
# video = generator.generate_video(prompt, sampling_param=sampling_param, output_path="wan_t2v_videos/")
|
||||
|
||||
# Generate another video with a different prompt, without reloading the
|
||||
# model!
|
||||
prompt2 = (
|
||||
"A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently in "
|
||||
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic.")
|
||||
_ = generator.generate_video(prompt2, output_path=OUTPUT_PATH, save_video=True, height=720, width=1280, num_frames=81)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,30 +0,0 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_wan2_2_14B_i2v"
|
||||
def main():
|
||||
# FastVideo will automatically use the optimal default arguments for the
|
||||
# model.
|
||||
# If a local path is provided, FastVideo will make a best effort
|
||||
# attempt to identify the optimal arguments.
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.2-I2V-A14B-Diffusers",
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=True, # DiT need to be offloaded for MoE
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
# Set pin_cpu_memory to false if CPU RAM is limited and there're no frequent CPU-GPU transfer
|
||||
pin_cpu_memory=True,
|
||||
# image_encoder_cpu_offload=False,
|
||||
)
|
||||
|
||||
prompt = "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline's intricate details and the refreshing atmosphere of the seaside."
|
||||
image_path = "https://huggingface.co/datasets/YiYiXu/testing-images/resolve/main/wan_i2v_input.JPG"
|
||||
|
||||
video = generator.generate_video(prompt, image_path=image_path, output_path=OUTPUT_PATH, save_video=True, height=832, width=480, num_frames=81)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,38 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
def main():
|
||||
# FastVideo will automatically use the optimal default arguments for the
|
||||
# model.
|
||||
# If a local path is provided, FastVideo will make a best effort
|
||||
# attempt to identify the optimal arguments.
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-I2V-14B-480P-Diffusers",
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=2,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=False,
|
||||
image_encoder_cpu_offload=False,
|
||||
)
|
||||
|
||||
sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-I2V-14B-480P-Diffusers")
|
||||
sampling_param.num_frames = 61
|
||||
sampling_param.num_inference_steps = 40
|
||||
sampling_param.guidance_scale = 5.0
|
||||
sampling_param.height = 448
|
||||
sampling_param.width = 832
|
||||
sampling_param.seed = 1024
|
||||
sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
|
||||
# Generate videos with the same simple API, regardless of GPU count
|
||||
prompt = (
|
||||
"An astronaut hatching from an egg, on the surface of the moon, the darkness and depth of space realised in the background. High quality, ultrarealistic detail and breath-taking movie-like camera shot."
|
||||
)
|
||||
video = generator.generate_video(prompt, sampling_param=sampling_param, output_path=OUTPUT_PATH, save_video=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,59 @@
|
||||
# FastVideo Gradio Demo
|
||||
|
||||
This is a Gradio-based web interface for generating videos using the FastVideo framework. The demo allows users to create videos from text prompts with various customization options.
|
||||
|
||||
## Overview
|
||||
|
||||
The demo uses the FastVideo framework to generate videos based on text prompts. It provides a simple web interface built with Gradio that allows users to:
|
||||
|
||||
- Enter text prompts to generate videos
|
||||
- Customize video parameters (dimensions, number of frames, etc.)
|
||||
- Use negative prompts to guide the generation process
|
||||
- Set or randomize seeds for reproducibility
|
||||
|
||||
---
|
||||
|
||||
## Usage
|
||||
|
||||
Run the demo with:
|
||||
|
||||
```bash
|
||||
python examples/inference/gradio/gradio_demo.py
|
||||
```
|
||||
|
||||
This will start a web server at `http://0.0.0.0:7860` where you can access the interface.
|
||||
|
||||
---
|
||||
|
||||
## Model Initialization
|
||||
|
||||
This demo initializes a `VideoGenerator` with the minimum required arguments for inference. Users can seamlessly adjust inference options between generations, including prompts, resolution, video length, or even the number of inference steps, *without ever needing to reload the model*.
|
||||
|
||||
## Video Generation
|
||||
|
||||
The core functionality is in the `generate_video` function, which:
|
||||
1. Processes user inputs
|
||||
2. Uses the FastVideo VideoGenerator from earlier to run inference (`generator.generate_video()`)
|
||||
3. Returns an output path that Gradio uses to display the generated video
|
||||
|
||||
## Gradio Interface
|
||||
|
||||
The interface is built with several components:
|
||||
- A text input for the prompt
|
||||
- A video display for the result
|
||||
- Inference options in a collapsible accordion:
|
||||
- Height and width sliders
|
||||
- Number of frames slider
|
||||
- Guidance scale slider
|
||||
- Inference steps slider
|
||||
- Negative prompt options
|
||||
- Seed controls
|
||||
|
||||
### Inference Options
|
||||
|
||||
- **Height/Width**: Control the resolution of the generated video
|
||||
- **Number of Frames**: Set how many frames to generate
|
||||
- **Guidance Scale**: Control how closely the generation follows the prompt
|
||||
- **Inference Steps**: More steps can improve quality but take longer
|
||||
- **Negative Prompt**: Specify what you don't want to see in the video
|
||||
- **Seed**: Control randomness for reproducible results
|
||||
@@ -0,0 +1,182 @@
|
||||
# FastVideo Dual Model Setup (T2V + I2V)
|
||||
|
||||
This document describes the dual model functionality that has been added to the FastVideo Gradio app, supporting both Text-to-Video (T2V) and Image-to-Video (I2V) generation modes with specialized models.
|
||||
|
||||
## Overview
|
||||
|
||||
The Gradio app now supports both Text-to-Video (T2V) and Image-to-Video (I2V) generation modes using specialized models:
|
||||
- **T2V Model**: `FastVideo/FastWan2.1-T2V-1.3B-Diffusers` for text-to-video generation
|
||||
- **I2V Model**: `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers` for image-to-video generation
|
||||
|
||||
Users can switch between modes using the tabbed interface and upload images for I2V generation.
|
||||
|
||||
## Model Configuration
|
||||
|
||||
### T2V Model
|
||||
- **Model**: `FastVideo/FastWan2.1-T2V-1.3B-Diffusers`
|
||||
- **Purpose**: Text-to-video generation
|
||||
- **Default Parameters**: Optimized for text prompts
|
||||
|
||||
### I2V Model
|
||||
- **Model**: `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers`
|
||||
- **Purpose**: Image-to-video generation
|
||||
- **Default Parameters**: Optimized for image animation
|
||||
|
||||
## Changes Made
|
||||
|
||||
### Backend Changes (`ray_serve_backend.py`)
|
||||
|
||||
1. **Added dual model support**:
|
||||
- Separate model paths for T2V and I2V
|
||||
- Automatic model selection based on request type
|
||||
- Independent model initialization
|
||||
|
||||
2. **Updated `VideoGenerationRequest`**:
|
||||
- Added `model_type` field ("t2v" or "i2v")
|
||||
- Added `image_path` field for I2V input
|
||||
|
||||
3. **Enhanced `FastVideoAPI` class**:
|
||||
- Dual model initialization (`t2v_generator` and `i2v_generator`)
|
||||
- Separate default parameters for each model
|
||||
- Automatic model selection in `generate_video` method
|
||||
|
||||
### Frontend Changes (`gradio_frontend.py`)
|
||||
|
||||
1. **Tabbed interface**:
|
||||
- "Text-to-Video" tab for T2V generation
|
||||
- "Image-to-Video" tab for I2V generation
|
||||
|
||||
2. **Automatic model selection**:
|
||||
- T2V tab uses T2V model automatically
|
||||
- I2V tab uses I2V model automatically
|
||||
- Model type sent in API requests
|
||||
|
||||
3. **Separate event handlers**:
|
||||
- `handle_t2v_generation` for text-to-video
|
||||
- `handle_i2v_generation` for image-to-video
|
||||
|
||||
## Usage
|
||||
|
||||
### Starting the Application
|
||||
|
||||
1. **Using the combined startup script (recommended)**:
|
||||
```bash
|
||||
python start_ray_serve_app.py
|
||||
```
|
||||
|
||||
2. **Manual startup**:
|
||||
```bash
|
||||
# Start backend
|
||||
python ray_serve_backend.py \
|
||||
--t2v_model_path "FastVideo/FastWan2.1-T2V-1.3B-Diffusers" \
|
||||
--i2v_model_path "Wan-AI/Wan2.1-I2V-14B-480P-Diffusers"
|
||||
|
||||
# Start frontend
|
||||
python gradio_frontend.py --backend_url "http://localhost:8000"
|
||||
```
|
||||
|
||||
### Using T2V Mode
|
||||
|
||||
1. Navigate to the "Text-to-Video" tab
|
||||
2. Enter a text prompt describing the video you want to generate
|
||||
3. Adjust advanced parameters if needed
|
||||
4. Click "Run" to generate the video
|
||||
|
||||
### Using I2V Mode
|
||||
|
||||
1. Navigate to the "Image-to-Video" tab
|
||||
2. Upload an image using the image upload component
|
||||
3. Enter a prompt describing how the image should animate
|
||||
4. Adjust advanced parameters if needed
|
||||
5. Click "Run" to generate the video
|
||||
|
||||
### Example Prompts
|
||||
|
||||
**T2V Examples**:
|
||||
- "A hand enters the frame, pulling a sheet of plastic wrap over three balls of dough placed on a wooden surface."
|
||||
- "A vintage train snakes through the mountains, its plume of white steam rising dramatically against the jagged peaks."
|
||||
|
||||
**I2V Examples**:
|
||||
- "The image comes to life with subtle movement, the scene gently animating while maintaining the original composition and mood."
|
||||
- "The static image transforms into a dynamic scene with natural motion, preserving the original lighting and atmosphere."
|
||||
|
||||
## Testing
|
||||
|
||||
A comprehensive test script is provided to verify both T2V and I2V functionality:
|
||||
|
||||
```bash
|
||||
python test_i2v.py
|
||||
```
|
||||
|
||||
This script:
|
||||
- Tests backend health
|
||||
- Tests T2V functionality with text prompts
|
||||
- Tests I2V functionality with image uploads
|
||||
- Verifies response formats for both modes
|
||||
- Cleans up test files
|
||||
|
||||
## Technical Details
|
||||
|
||||
### Backend API Changes
|
||||
|
||||
The `/generate_video` endpoint now accepts:
|
||||
|
||||
```json
|
||||
{
|
||||
"prompt": "Animation description",
|
||||
"model_type": "t2v", // or "i2v"
|
||||
"image_path": "/path/to/input/image.png", // for I2V
|
||||
// ... other parameters
|
||||
}
|
||||
```
|
||||
|
||||
### Model Selection Logic
|
||||
|
||||
- **T2V Mode**: Uses `FastVideo/FastWan2.1-T2V-1.3B-Diffusers`
|
||||
- **I2V Mode**: Uses `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers`
|
||||
- **Automatic Selection**: Based on presence of `image_path` and `model_type`
|
||||
|
||||
### Memory Management
|
||||
|
||||
- Both models are loaded independently
|
||||
- Automatic cleanup prevents memory leaks
|
||||
- Temporary files are cleaned up after processing
|
||||
|
||||
## Configuration Options
|
||||
|
||||
### Command Line Arguments
|
||||
|
||||
**Backend**:
|
||||
- `--t2v_model_path`: Path to T2V model
|
||||
- `--i2v_model_path`: Path to I2V model
|
||||
- `--output_path`: Output directory
|
||||
- `--host`, `--port`: Server configuration
|
||||
|
||||
**Frontend**:
|
||||
- `--backend_url`: Backend API URL
|
||||
- `--t2v_model_path`, `--i2v_model_path`: Model paths (for reference)
|
||||
- `--host`, `--port`: Server configuration
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
1. **Model Loading Issues**: Ensure both models are accessible
|
||||
2. **Memory Issues**: The backend includes automatic cleanup
|
||||
3. **Image Upload Failures**: Check image format and size
|
||||
4. **Generation Failures**: Check backend logs for detailed errors
|
||||
|
||||
## Performance Considerations
|
||||
|
||||
- **Model Loading**: Both models are loaded at startup
|
||||
- **Memory Usage**: Higher memory requirements due to dual models
|
||||
- **Generation Time**: I2V may take longer due to larger model size
|
||||
- **GPU Requirements**: Ensure sufficient VRAM for both models
|
||||
|
||||
## Future Enhancements
|
||||
|
||||
Potential improvements:
|
||||
- Model switching without restart
|
||||
- Batch processing for both modes
|
||||
- Advanced image preprocessing
|
||||
- Progress indicators
|
||||
- Result caching and history
|
||||
- Model-specific parameter optimization
|
||||
@@ -0,0 +1,222 @@
|
||||
# FastVideo with Ray Serve Backend and Gradio Frontend
|
||||
|
||||
This setup provides a scalable web application for FastVideo inference using Ray Serve as the backend and Gradio as the frontend.
|
||||
|
||||
## Architecture
|
||||
|
||||
- **Backend**: Ray Serve handles video generation requests with GPU acceleration
|
||||
- **Frontend**: Gradio provides a user-friendly web interface
|
||||
- **Communication**: HTTP REST API between frontend and backend
|
||||
|
||||
## Features
|
||||
|
||||
- ✅ Scalable backend with Ray Serve
|
||||
- ✅ GPU-accelerated video generation
|
||||
- ✅ User-friendly Gradio interface
|
||||
- ✅ Health monitoring and error handling
|
||||
- ✅ All original functionality preserved
|
||||
- ✅ Easy deployment and management
|
||||
|
||||
## Installation
|
||||
|
||||
1. Install the additional dependencies:
|
||||
|
||||
```bash
|
||||
pip install -r requirements_ray_serve.txt
|
||||
```
|
||||
|
||||
2. Ensure you have the FastVideo model available (the default is `FastVideo/FastHunyuan-diffusers`)
|
||||
|
||||
## Usage
|
||||
|
||||
### Option 1: Start Both Services Together (Recommended)
|
||||
|
||||
Use the startup script to launch both backend and frontend:
|
||||
|
||||
```bash
|
||||
python start_ray_serve_app.py
|
||||
```
|
||||
|
||||
This will:
|
||||
- Start the Ray Serve backend on port 8000
|
||||
- Start the Gradio frontend on port 7860
|
||||
- Monitor both services and provide unified logging
|
||||
- Handle graceful shutdown with Ctrl+C
|
||||
|
||||
### Option 2: Start Services Separately
|
||||
|
||||
#### Start Backend Only
|
||||
|
||||
```bash
|
||||
python ray_serve_backend.py --model_path FastVideo/FastHunyuan-diffusers --output_path outputs
|
||||
```
|
||||
|
||||
#### Start Frontend Only
|
||||
|
||||
```bash
|
||||
python gradio_frontend.py --backend_url http://localhost:8000 --model_path FastVideo/FastHunyuan-diffusers
|
||||
```
|
||||
|
||||
## Configuration
|
||||
|
||||
### Command Line Arguments
|
||||
|
||||
#### Startup Script (`start_ray_serve_app.py`)
|
||||
|
||||
- `--model_path`: Path to the FastVideo model (default: `FastVideo/FastHunyuan-diffusers`)
|
||||
- `--output_path`: Directory to save generated videos (default: `outputs`)
|
||||
- `--backend_host`: Backend host to bind to (default: `0.0.0.0`)
|
||||
- `--backend_port`: Backend port (default: `8000`)
|
||||
- `--frontend_host`: Frontend host to bind to (default: `0.0.0.0`)
|
||||
- `--frontend_port`: Frontend port (default: `7860`)
|
||||
- `--skip_backend_check`: Skip backend health check
|
||||
|
||||
#### Backend (`ray_serve_backend.py`)
|
||||
|
||||
- `--model_path`: Path to the FastVideo model
|
||||
- `--output_path`: Directory to save generated videos
|
||||
- `--host`: Host to bind to
|
||||
- `--port`: Port to bind to
|
||||
|
||||
#### Frontend (`gradio_frontend.py`)
|
||||
|
||||
- `--backend_url`: URL of the Ray Serve backend
|
||||
- `--model_path`: Path to the model (for default parameters)
|
||||
- `--host`: Host to bind to
|
||||
- `--port`: Port to bind to
|
||||
|
||||
### Environment Variables
|
||||
|
||||
You can also set these environment variables:
|
||||
|
||||
- `FASTVIDEO_MODEL_PATH`: Path to the FastVideo model
|
||||
- `FASTVIDEO_OUTPUT_PATH`: Directory to save generated videos
|
||||
- `RAY_SERVE_HOST`: Backend host
|
||||
- `RAY_SERVE_PORT`: Backend port
|
||||
- `GRADIO_HOST`: Frontend host
|
||||
- `GRADIO_PORT`: Frontend port
|
||||
|
||||
## API Endpoints
|
||||
|
||||
### Backend API (Ray Serve)
|
||||
|
||||
- `GET /health`: Health check endpoint
|
||||
- `POST /generate_video`: Video generation endpoint
|
||||
|
||||
#### Video Generation Request
|
||||
|
||||
```json
|
||||
{
|
||||
"prompt": "A beautiful sunset over the ocean",
|
||||
"negative_prompt": "blurry, low quality",
|
||||
"use_negative_prompt": true,
|
||||
"seed": 42,
|
||||
"guidance_scale": 7.5,
|
||||
"num_frames": 21,
|
||||
"height": 512,
|
||||
"width": 512,
|
||||
"num_inference_steps": 20,
|
||||
"randomize_seed": false
|
||||
}
|
||||
```
|
||||
|
||||
#### Video Generation Response
|
||||
|
||||
```json
|
||||
{
|
||||
"output_path": "/path/to/generated/video.mp4",
|
||||
"seed": 42,
|
||||
"success": true,
|
||||
"error_message": null
|
||||
}
|
||||
```
|
||||
|
||||
## Deployment
|
||||
|
||||
### Local Development
|
||||
|
||||
1. Start the application:
|
||||
```bash
|
||||
python start_ray_serve_app.py
|
||||
```
|
||||
|
||||
2. Access the frontend at: `http://localhost:7860`
|
||||
3. Access the backend API at: `http://localhost:8000`
|
||||
|
||||
### Production Deployment
|
||||
|
||||
For production deployment, consider:
|
||||
|
||||
1. **Load Balancing**: Use a reverse proxy (nginx, traefik) in front of the services
|
||||
2. **Monitoring**: Add monitoring and logging (Prometheus, Grafana)
|
||||
3. **Scaling**: Configure Ray Serve for horizontal scaling
|
||||
4. **Security**: Add authentication and rate limiting
|
||||
5. **Storage**: Use shared storage for video outputs
|
||||
|
||||
### Docker Deployment
|
||||
|
||||
Create a Dockerfile for containerized deployment:
|
||||
|
||||
```dockerfile
|
||||
FROM python:3.9-slim
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Install dependencies
|
||||
COPY requirements_ray_serve.txt .
|
||||
RUN pip install -r requirements_ray_serve.txt
|
||||
|
||||
# Copy application files
|
||||
COPY . .
|
||||
|
||||
# Expose ports
|
||||
EXPOSE 8000 7860
|
||||
|
||||
# Start the application
|
||||
CMD ["python", "start_ray_serve_app.py"]
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Common Issues
|
||||
|
||||
1. **Backend not starting**: Check GPU availability and Ray installation
|
||||
2. **Frontend can't connect**: Verify backend URL and network connectivity
|
||||
3. **Video generation fails**: Check model path and GPU memory
|
||||
4. **Port conflicts**: Change ports using command line arguments
|
||||
|
||||
### Logs
|
||||
|
||||
- Backend logs are prefixed with `[BACKEND]`
|
||||
- Frontend logs are prefixed with `[FRONTEND]`
|
||||
- Use `--skip_backend_check` if you need to debug startup issues
|
||||
|
||||
### Performance Tuning
|
||||
|
||||
- Adjust `num_replicas` in the Ray Serve deployment for scaling
|
||||
- Configure `max_concurrent_queries` based on GPU memory
|
||||
- Use multiple GPUs by modifying `ray_actor_options`
|
||||
|
||||
## Migration from Original Gradio Demo
|
||||
|
||||
The new setup maintains full compatibility with the original functionality:
|
||||
|
||||
1. All parameters and options are preserved
|
||||
2. The same example prompts are included
|
||||
3. The UI layout and behavior are identical
|
||||
4. Video generation quality is the same
|
||||
|
||||
The main differences are:
|
||||
- Backend processing is now handled by Ray Serve
|
||||
- Better error handling and status monitoring
|
||||
- Scalable architecture for production use
|
||||
- Separation of concerns between frontend and backend
|
||||
|
||||
## Contributing
|
||||
|
||||
To extend this setup:
|
||||
|
||||
1. Add new endpoints to `ray_serve_backend.py`
|
||||
2. Update the frontend in `gradio_frontend.py`
|
||||
3. Modify the startup script if needed
|
||||
4. Update this README with new features
|
||||
@@ -0,0 +1,501 @@
|
||||
import argparse
|
||||
import os
|
||||
import time
|
||||
import json
|
||||
import statistics
|
||||
import asyncio
|
||||
import aiohttp
|
||||
from copy import deepcopy
|
||||
from typing import List, Dict, Any
|
||||
import threading
|
||||
|
||||
import torch
|
||||
|
||||
# All the prompts for stress testing
|
||||
STRESS_TEST_PROMPTS = [
|
||||
"A person reading a book with words that float off the pages and form pictures.",
|
||||
"A person diving into a pool of liquid crystal, creating ripples of light.",
|
||||
"A handheld shot chasing after a group of friends laughing and playing on the beach at sunset.",
|
||||
"A mysterious ancient temple hidden in the jungle.",
|
||||
"A high-speed train navigating a steep descent.",
|
||||
"a toy robot wearing blue jeans and a white t shirt taking a pleasant stroll in Antarctica during a winter storm",
|
||||
"A cheetah accelerating to full speed while chasing its prey.",
|
||||
"A serene orchard is in full bloom, with trees heavy with blossoms and bees buzzing around, darting from flower to flower in a display of natural harmony.",
|
||||
"A little child let out a big yawn",
|
||||
"Subtle reflections of a woman on the window of a train moving at hyper-speed in a Japanese city.",
|
||||
"A truck left along the edge of a cliff, revealing the stunning coastal landscape below with waves crashing against the rocks.",
|
||||
"A red bird transforms into a flag",
|
||||
"A zoom-out from a single leaf on a tree to reveal the entire forest, showcasing the vastness and diversity of the woodland.",
|
||||
"A slow-motion video of a liquid droplet bouncing on a water-repellent surface.",
|
||||
"Static camera shot. A dinasour running near some lions and chasing them away.",
|
||||
"an adorable kangaroo wearing purple overalls and cowboy boots taking a pleasant stroll in Mumbai India during a beautiful sunset",
|
||||
"A zoom-in on an artist's brush touching the canvas, highlighting the texture of the paint and the strokes being made.",
|
||||
"an old man wearing blue jeans and a white t shirt taking a pleasant stroll in Mumbai India during a colorful festival",
|
||||
"A woman is ascending to the sky from the ground",
|
||||
"View out a window of a giant strange creature walking in rundown city at night, one single street lamp dimly lighting the area.",
|
||||
"An arc shot around a lone tree in a vast, foggy field at dawn, revealing the changing light and shadows.",
|
||||
"A person sculpting a statue out of a waterfall, the water solidifying under their touch.",
|
||||
"The person's forehead creased with concentration as she worked on a challenging puzzle.",
|
||||
"The person's cheeks flushed with pleasure as she savored a delicious meal.",
|
||||
"Hand-drawn simple line art, a young kid looking up into space with a wondrous expression on his face.",
|
||||
"A crab made of different jewlery is walking on the beach. As it walks, it drops different jewelry pieces like diamonds, pearls, etc",
|
||||
"Gold coins are falling out when elevator door opens",
|
||||
"the scene transitions from huge waves into a snowy mountain at sunset",
|
||||
"a giant cathedral is completely filled with cats. there are cats everywhere you look. a man enters the cathedral and bows before the giant cat king sitting on a throne.",
|
||||
"A mother dog gently picks up a piece of meat and carefully places it in her puppy's bowl, her eyes filled with warmth and care as she watches her little one eat.",
|
||||
"A soap bubble floating in the air, displaying iridescent colors that shift and change as it moves through different angles of light.",
|
||||
"A truck left alongside a train moving through the countryside, matching its speed and revealing the changing landscape.",
|
||||
"An astronaut walking between stone buildings.",
|
||||
"A close-up shot of the person's face reveals his fear and desperation as he navigates the ship through the storm.",
|
||||
"A frozen lake slowly cracking and thawing as spring arrives, with sheets of ice breaking apart and drifting across the surface.",
|
||||
"A FPV shot zooming through a tunnel into a vibrant underwater space.",
|
||||
"a toy robot wearing blue jeans and a white t shirt taking a pleasant stroll in Mumbai India during a colorful festival",
|
||||
"A person sips on a smoothie, the cool and fruity flavors refreshing her mouth.",
|
||||
"In a vibrant theater, a magician in dazzling attire stands center stage, pulling a comically oversized rubber chicken from an ornate, old-fashioned box. His costume shimmers under the stage lights, adding to the spectacle. The crowd erupts in laughter and applause, their faces filled with joy and amazement. The magician's expression hints at mischievous delight as he holds up the rubber chicken, his performance bringing cheer to the audience.",
|
||||
"A hamster running on a spinning wheel.",
|
||||
"A quaint village nestled in a valley is surrounded by blooming cherry blossoms, with petals drifting through the air as villagers go about their daily activities, adding life to the scene.",
|
||||
"In a tranquil forest clearing, a sparkling waterfall cascades down into a clear pool, surrounded by lush greenery and flowers, with occasional birds fluttering by.",
|
||||
"A woman beamed with pride as she watched her child perform on stage.",
|
||||
"an adorable kangaroo wearing blue jeans and a white t shirt taking a pleasant stroll in Mumbai India during a winter storm",
|
||||
"A man is eating salad",
|
||||
"An Asian girl wearing a bright yellow T-shirt and white pants is Hip-Hop dancing",
|
||||
"nighttime footage of a hermit crab using an incandescent lightbulb as its shell",
|
||||
"a toy robot wearing a green dress and a sun hat taking a pleasant stroll in Antarctica during a beautiful sunset",
|
||||
"A goat operating a food truck, serving gourmet grilled cheese sandwiches to a line of animals.",
|
||||
"Macro shot. Man in an antique scuba helmet with dark glass walking out of a flower",
|
||||
"A bustling train station in the heart of a vibrant city.",
|
||||
"Light filtering through a canopy of autumn leaves, casting warm, dappled patterns of yellow, orange, and red onto the ground.",
|
||||
"Chimneys in the setting sun",
|
||||
"A longboarder accelerating downhill, carving through turns.",
|
||||
"A couple runs through a sudden downpour, laughing and splashing in puddles as they try to find shelter.",
|
||||
"A glass of iced coffee condensing water on the outside, with droplets forming and sliding down the glass in slow motion.",
|
||||
"macro shot of a leaf showing tiny trains moving through its veins",
|
||||
"A corgi wearing sunglasses walks on the beach of a tropical island",
|
||||
"Borneo wildlife on the Kinabatangan River",
|
||||
"A beautiful silhouette animation shows a wolf howling at the moon, feeling lonely, until it finds its pack.",
|
||||
"an adorable kangaroo wearing blue jeans and a white t shirt taking a pleasant stroll in Johannesburg South Africa during a colorful festival",
|
||||
"A green monster made of plants walks through an airport.",
|
||||
"A close up view of a glass sphere that has a zen garden within it. There is a small dwarf in the sphere who is raking the zen garden and creating patterns in the sand.",
|
||||
"A person on a scooter colliding with a park bench, the scooter tipping over.",
|
||||
"A tilt-up from a city street, ascending to show the skyline with its mix of modern and historic architecture.",
|
||||
"A chef tossing a pancake into the air and catching it.",
|
||||
"A woman whispering a secret into a friend's ear.",
|
||||
"A vulture circling high in the sky.",
|
||||
"A medieval castle overlooking a bustling renaissance fair.",
|
||||
"a toy robot wearing purple overalls and cowboy boots taking a pleasant stroll in Mumbai India during a beautiful sunset",
|
||||
"A man standing in front of a burning building giving the 'thumbs up' sign.",
|
||||
"The person's cheeks flushed with embarrassment as he told a funny story.",
|
||||
"Llamas and Emus are playing chess",
|
||||
"A woman sipping a steaming cup of tea.",
|
||||
"A tree root bursting through the seat of an ancient, weathered bench, intertwining with the wood.",
|
||||
"Smoke rises from the chimney of a cozy log cabin nestled in the woods, with soft light glowing from the windows, suggesting a warm and inviting atmosphere.",
|
||||
"A close-up of sparkling water being poured into a glass, capturing the detailed flow and bubbles.",
|
||||
"a woman wearing blue jeans and a white t shirt taking a pleasant stroll in Antarctica during a beautiful sunset",
|
||||
"The Glenfinnan Viaduct is a historic railway bridge in Scotland, UK, that crosses over the west highland line between the towns of Mallaig and Fort William. It is a stunning sight as a steam train leaves the bridge, traveling over the arch-covered viaduct. The landscape is dotted with lush greenery and rocky mountains, creating a picturesque backdrop for the train journey. The sky is blue and the sun is shining, making for a beautiful day to explore this majestic spot.",
|
||||
"A piece of elastic fabric being pulled and stretched, then returning to its original size when the tension is released.",
|
||||
"a woman wearing a green dress and a sun hat taking a pleasant stroll in Antarctica during a beautiful sunset",
|
||||
"A video of a water jet cutting through metal, showing the powerful and precise movement of water.",
|
||||
"Car mirrors and sunsets",
|
||||
"Giant Pandas are eating hot noodles in a Chinese restaurant",
|
||||
"A rally car taking a fast turn on a track",
|
||||
"a toy robot wearing purple overalls and cowboy boots taking a pleasant stroll in Mumbai India during a colorful festival",
|
||||
"A crystal-clear icicle slowly dripping as it melts in the warmth of the midday sun, each drop sparkling as it falls.",
|
||||
"A tilt-down from a chandelier in a grand hall, revealing the ornate decor and people mingling below.",
|
||||
"A man is playing the drums under the water",
|
||||
"A person playing an electric guitar made of lightning, with thunderous sound waves.",
|
||||
"A person floating in a bubble, drifting over a bustling cityscape.",
|
||||
"A tilt-down from a starry night sky, revealing a quiet forest clearing bathed in moonlight.",
|
||||
"A pan right through a dense jungle, moving past lush vegetation and exotic wildlife.",
|
||||
"Close-up of a man eating an apple.",
|
||||
"A low-angle shot of a dancer leaping gracefully into the air, making their movement appear even more dynamic and powerful.",
|
||||
"A woman is search her bag trying to find something.",
|
||||
"A bulldozer clears debris from a demolished building, making way for new construction.",
|
||||
"A man sighed in relief as the doctor delivered the good news.",
|
||||
"A tsunami coming through an alley in Bulgaria, dynamic movement.",
|
||||
"Blooming Flowers",
|
||||
"A push-in through a dense crowd at a festival, moving towards a performer on stage who is captivating the audience.",
|
||||
"A truck right through a tranquil garden, moving past blooming flowers, trees, and a small fountain.",
|
||||
"The person's eyes sparkled with excitement as he greeted a friend.",
|
||||
"A person playing chess with a robot on a floating platform above the ocean.",
|
||||
"A gentle breeze rustles the leaves as someone walks down a serene forest path, sunlight filtering through the trees and shifting patterns on the ground as branches sway.",
|
||||
"A rollercoaster ride from a city to a desert and then to an ice world",
|
||||
"A pan left across an ancient library, moving from shelf to shelf, showcasing rows of leather-bound books.",
|
||||
"A mother otter floating on her back in a river, cradling her pup on her stomach to keep it safe and warm in the gentle current.",
|
||||
"an adorable kangaroo wearing purple overalls and cowboy boots taking a pleasant stroll in Johannesburg South Africa during a colorful festival",
|
||||
"a woman wearing a green dress and a sun hat taking a pleasant stroll in Mumbai India during a colorful festival",
|
||||
"A delicate layer of morning frost melting off a flower petal, the tiny droplets glistening like diamonds in the light.",
|
||||
"A panda is cooking for her child, her child is next to her.",
|
||||
"Macro shot of a man wearing an antique diving helmet with dark glass and a jetpack walking on the veins of a leaf. Realistic style",
|
||||
"an old man wearing purple overalls and cowboy boots taking a pleasant stroll in Johannesburg South Africa during a beautiful sunset",
|
||||
"A girl is unfolding a birthday gift.",
|
||||
"A pencil drawing an architectural plan.",
|
||||
"A handheld camera following a dog running through a park, bouncing and tilting as it captures the dog's joyful exploration.",
|
||||
"A pan left across a serene beach at sunrise, moving from the darkened shore to the brightening horizon.",
|
||||
"A group of people are clapping to celebrate",
|
||||
"Vendors set up stalls at a bustling farmer's market, displaying fresh fruits and vegetables, while people stroll through, selecting produce and enjoying the lively atmosphere.",
|
||||
"A police helicopter hovers above a high-speed chase, guiding officers on the ground to apprehend a suspect.",
|
||||
"A paper origami dragon riding a boat in waves. Realistic style.",
|
||||
"A close-up of a droplet of dew forming on a leaf, capturing the detailed surface tension.",
|
||||
"a toy robot wearing blue jeans and a white t shirt taking a pleasant stroll in Mumbai India during a beautiful sunset",
|
||||
"A dry rainbow rose is coming back to life.",
|
||||
"A glass falling off a table and shattering on the floor.",
|
||||
"A marathon runner crossing the finish line after a grueling race.",
|
||||
"A zoom-in on a drop of morning dew on a leaf, showing the reflection of the surrounding world within it.",
|
||||
"A child blowing on hot cocoa to cool it down.",
|
||||
"A squad of futsal players showcasing their skills on an indoor court.",
|
||||
"A princess is brushing her long golden hair in the garden.",
|
||||
"A close-up of a pair of eyes, revealing the subtle emotions and reflections within them.",
|
||||
"A tracking shot of a group of cyclists racing through a forest trail, with trees and foliage rushing by.",
|
||||
"A woman yawning widely at the end of a long day.",
|
||||
"an old man wearing a green dress and a sun hat taking a pleasant stroll in Johannesburg South Africa during a colorful festival",
|
||||
"Hidden within a garden, an ancient fountain trickles with water, surrounded by vibrant flowers and lush greenery that seem to whisper secrets of the past.",
|
||||
"A Chinese man sits at a table and eats noodles with chopsticks",
|
||||
"A pink pig running fast toward the camera in an alley in Tokyo.",
|
||||
"Strange creatures move through a mysterious, foggy marsh, their silhouettes barely visible through the dense mist as they navigate the eerie, otherworldly landscape.",
|
||||
"Tour of an art gallery with many beautiful works of art in different styles.",
|
||||
"FPV flying through a colorful coral lined streets of an underwater suburban neighborhood.",
|
||||
"Aerial view of Santorini during the blue hour, showcasing the stunning architecture of white Cycladic buildings with blue domes. The caldera views are breathtaking, and the lighting creates a beautiful, serene atmosphere.",
|
||||
"Camera zoom out. A couple walking along the beach as the sun sets over the ocean.",
|
||||
"an extreme close up shot of a woman's eye, with her iris appearing as earth",
|
||||
"a woman wearing purple overalls and cowboy boots taking a pleasant stroll in Mumbai India during a colorful festival",
|
||||
"an old man wearing a green dress and a sun hat taking a pleasant stroll in Mumbai India during a winter storm",
|
||||
"an adorable kangaroo wearing blue jeans and a white t shirt taking a pleasant stroll in Antarctica during a winter storm",
|
||||
"A martial artist breaking a board with a powerful punch.",
|
||||
"People gather on a peaceful beach at sunset, a bonfire crackling as they sit around, enjoying the warmth and the sight of the sun dipping below the horizon.",
|
||||
"A close-up of a waterfall, showing the detailed movement of water as it crashes down.",
|
||||
"A child is blowing bubbles",
|
||||
"a woman wearing a green dress and a sun hat taking a pleasant stroll in Johannesburg South Africa during a winter storm",
|
||||
"A wide-angle perspective of a serene lake surrounded by mountains, reflecting the sky and creating a sense of infinite space.",
|
||||
"The person's eyebrows arched in skepticism as she listened to a dubious claim.",
|
||||
"an old man wearing blue jeans and a white t shirt taking a pleasant stroll in Mumbai India during a beautiful sunset",
|
||||
"a woman wearing a green dress and a sun hat taking a pleasant stroll in Johannesburg South Africa during a colorful festival",
|
||||
"a woman wearing purple overalls and cowboy boots taking a pleasant stroll in Antarctica during a colorful festival",
|
||||
"a toy robot wearing blue jeans and a white t shirt taking a pleasant stroll in Antarctica during a colorful festival",
|
||||
"A chef flips a pancake and puts cream on it.",
|
||||
"An astronaut runs on the surface of the moon, the low angle shot shows the vast background of the moon, the movement is smooth and appears lightweight",
|
||||
"A man's face lit up with happiness as he received a heartfelt compliment.",
|
||||
"A futuristic spaceport hums with activity as ships of various shapes and sizes take off and land on multiple platforms, their engines glowing with vibrant colors.",
|
||||
"A person knitting a scarf using beams of light instead of yarn.",
|
||||
"A pedestal up from the edge of a canyon, gradually revealing the expansive landscape and river below.",
|
||||
"a woman wearing purple overalls and cowboy boots taking a pleasant stroll in Johannesburg South Africa during a colorful festival",
|
||||
"an old man wearing blue jeans and a white t shirt taking a pleasant stroll in Johannesburg South Africa during a colorful festival",
|
||||
"A person walking up a staircase made of clouds leading to a floating castle.",
|
||||
"Monks meditate in a serene mountaintop temple, sitting in quiet reflection as the wind gently moves through the surrounding trees, creating a sense of peace and tranquility.",
|
||||
"An aerial shot of a bustling city intersection at rush hour, capturing the organized chaos of cars and pedestrians.",
|
||||
"A pair of hands skillfully knitting a colorful scarf, the yarn winding through their fingers with each stitch.",
|
||||
"Close-up, a Chinese child is eating dumplings",
|
||||
"A kite losing wind and falling to the ground.",
|
||||
"Bioluminescent waves gently wash ashore on a deserted beach, illuminating the sand with each cresting wave as a figure walks along the water's edge, leaving glowing footprints.",
|
||||
"A red panda taking a bite of a pizza",
|
||||
"A close-up shot of a young woman driving a car, looking thoughtful, blurred green forest visible through the rainy car window.",
|
||||
"A high-speed video of a splash created by a stone thrown into a pond.",
|
||||
"A metal rod being bent slightly by a force and then springing back to its original straight shape when the force is removed.",
|
||||
"A hedgehog in a knight's armor, riding a toy horse into a medieval castle.",
|
||||
"A bird made of fresh oranges rushes out of the orange",
|
||||
"A low altitude first person perspective camera tracking shot of a soccer player's feet dribbling the ball on the groud in a soccer field, Sports Videography, Motion Tracking camera shot",
|
||||
"A tranquil island retreat features swaying palm trees and hammocks strung between them, inviting guests to relax and enjoy the serene beauty of the surroundings.",
|
||||
"a spooky haunted mansion, with friendly jack o lanterns and ghost characters welcoming trick or treaters to the entrance, tilt shift photography",
|
||||
"A coconut tree made of dollar bills at sunset, with bills falling off like leaves.",
|
||||
"A motocross bike accelerating out of a tight turn on a dirt track.",
|
||||
"A tranquil Zen garden with a gently flowing stream and koi fish.",
|
||||
"A green monster made of leaves walks through the airport, carrying a suitcase.",
|
||||
"A time-lapse of a frost-covered leaf gradually thawing in the morning sunlight, with tiny water droplets forming and trickling down.",
|
||||
"A woman practicing her archery skills at a range.",
|
||||
"A slow-motion video of ink being injected into a tank of water, creating intricate and beautiful patterns.",
|
||||
"a woman wearing blue jeans and a white t shirt taking a pleasant stroll in Johannesburg South Africa during a winter storm",
|
||||
"The person's forehead creased with worry as he listened to bad news.",
|
||||
"An arc shot around a grand piano being played in an empty concert hall, the motion revealing the intricate details of the instrument.",
|
||||
"A person conducting a symphony of animals in a forest clearing.",
|
||||
"A truck right alongside a flowing river, capturing the movement of the water and the surrounding forest.",
|
||||
"A rocket blasting off from the launch pad, accelerating rapidly into the sky.",
|
||||
"Workers move through a picturesque vineyard during the harvest season, carefully picking grapes and placing them into baskets as the sun bathes the vines in a warm glow.",
|
||||
"A person is eating an ice cream.",
|
||||
"An over-the-shoulder perspective of a chef meticulously plating a dish in a bustling kitchen.",
|
||||
"A man looked away in shame when confronted with his wrongdoing.",
|
||||
"A person is savoring a slice of pizza at a pizzeria."
|
||||
]
|
||||
|
||||
class BackendStressTest:
|
||||
def __init__(self, output_path: str,
|
||||
server_url: str = "http://localhost:8000", max_concurrent: int = 50):
|
||||
self.output_path = output_path
|
||||
self.server_url = server_url
|
||||
self.max_concurrent = max_concurrent
|
||||
|
||||
# Results storage
|
||||
self.results = []
|
||||
self.lock = threading.Lock()
|
||||
|
||||
async def check_health(self) -> bool:
|
||||
"""Check if the Ray Serve backend is healthy"""
|
||||
try:
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.get(f"{self.server_url}/health", timeout=aiohttp.ClientTimeout(total=5)) as response:
|
||||
return response.status == 200
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def build_request_params(self, prompt: str, **kwargs) -> Dict[str, Any]:
|
||||
"""Build request parameters for Ray Serve backend"""
|
||||
# Default parameters matching the Ray Serve backend
|
||||
default_params = {
|
||||
'prompt': prompt,
|
||||
'negative_prompt': None,
|
||||
'use_negative_prompt': False,
|
||||
'seed': 42,
|
||||
'guidance_scale': 7.5,
|
||||
'num_frames': 21,
|
||||
'height': 448,
|
||||
'width': 832,
|
||||
'num_inference_steps': 20,
|
||||
'randomize_seed': True,
|
||||
'return_frames': False # Don't return frames for stress testing to reduce overhead
|
||||
}
|
||||
|
||||
# Override with any provided kwargs
|
||||
for key, value in kwargs.items():
|
||||
if key in default_params:
|
||||
default_params[key] = value
|
||||
|
||||
# Randomize seed if requested
|
||||
if default_params.get('randomize_seed', True):
|
||||
default_params['seed'] = torch.randint(0, 1000000, (1,)).item()
|
||||
|
||||
# Handle negative prompt
|
||||
if not default_params.get('use_negative_prompt', False):
|
||||
default_params['negative_prompt'] = None
|
||||
|
||||
# NEW: Remove keys with None values to avoid sending nulls that may break validation
|
||||
clean_params = {k: v for k, v in default_params.items() if v is not None}
|
||||
return clean_params
|
||||
|
||||
async def test_single_request(self, session: aiohttp.ClientSession, prompt: str, request_id: int) -> Dict[str, Any]:
|
||||
"""Test a single request and measure latency"""
|
||||
start_time = time.time()
|
||||
|
||||
try:
|
||||
# Build request parameters
|
||||
request_params = self.build_request_params(prompt)
|
||||
|
||||
# Make request to Ray Serve backend
|
||||
async with session.post(
|
||||
f"{self.server_url}/generate_video",
|
||||
json=request_params,
|
||||
timeout=aiohttp.ClientTimeout(total=900) # 15 minute timeout for video generation
|
||||
) as response:
|
||||
|
||||
end_time = time.time()
|
||||
latency = end_time - start_time
|
||||
|
||||
if response.status == 200:
|
||||
response_data = await response.json()
|
||||
if response_data.get('success', False):
|
||||
result = {
|
||||
'request_id': request_id,
|
||||
'prompt': prompt,
|
||||
'latency': latency,
|
||||
'status': 'success',
|
||||
'response_time': latency, # Use our own timing
|
||||
'timestamp': start_time,
|
||||
'output_path': response_data.get('output_path', ''),
|
||||
'used_seed': response_data.get('seed', request_params['seed'])
|
||||
}
|
||||
else:
|
||||
result = {
|
||||
'request_id': request_id,
|
||||
'prompt': prompt,
|
||||
'latency': latency,
|
||||
'status': 'error',
|
||||
'error': response_data.get('error_message', 'Unknown backend error'),
|
||||
'timestamp': start_time
|
||||
}
|
||||
else:
|
||||
response_text = await response.text()
|
||||
result = {
|
||||
'request_id': request_id,
|
||||
'prompt': prompt,
|
||||
'latency': latency,
|
||||
'status': 'error',
|
||||
'error': f"HTTP {response.status}: {response_text}",
|
||||
'timestamp': start_time
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
end_time = time.time()
|
||||
latency = end_time - start_time
|
||||
result = {
|
||||
'request_id': request_id,
|
||||
'prompt': prompt,
|
||||
'latency': latency,
|
||||
'status': 'error',
|
||||
'error': str(e),
|
||||
'timestamp': start_time
|
||||
}
|
||||
|
||||
# Thread-safe result storage
|
||||
with self.lock:
|
||||
self.results.append(result)
|
||||
|
||||
return result
|
||||
|
||||
async def run_stress_test(self, num_iterations: int = 1, concurrent_requests: int = None):
|
||||
"""Run the stress test with multiple iterations and concurrent requests"""
|
||||
if concurrent_requests is None:
|
||||
concurrent_requests = self.max_concurrent
|
||||
|
||||
# Check backend health before starting
|
||||
print(f"Testing Ray Serve backend at {self.server_url}...")
|
||||
if not await self.check_health():
|
||||
print(f"❌ Backend is not healthy at {self.server_url}")
|
||||
print("Make sure the Ray Serve backend is running with:")
|
||||
print("python ray_serve_backend.py")
|
||||
return
|
||||
print("✅ Backend is healthy and ready for stress testing")
|
||||
|
||||
print(f"\nStarting stress test with {len(STRESS_TEST_PROMPTS)} prompts")
|
||||
print(f"Running {num_iterations} iteration(s) with {concurrent_requests} concurrent requests")
|
||||
print(f"Total requests: {len(STRESS_TEST_PROMPTS) * num_iterations}")
|
||||
print(f"Backend URL: {self.server_url}")
|
||||
print("-" * 80)
|
||||
|
||||
all_prompts = STRESS_TEST_PROMPTS * num_iterations
|
||||
request_id = 0
|
||||
|
||||
# Create semaphore to limit concurrent requests
|
||||
semaphore = asyncio.Semaphore(concurrent_requests)
|
||||
|
||||
async def limited_request(session: aiohttp.ClientSession, prompt: str, req_id: int):
|
||||
async with semaphore:
|
||||
return await self.test_single_request(session, prompt, req_id)
|
||||
|
||||
# Run concurrent requests using asyncio
|
||||
async with aiohttp.ClientSession() as session:
|
||||
# Create all tasks
|
||||
tasks = [
|
||||
limited_request(session, prompt, request_id + i)
|
||||
for i, prompt in enumerate(all_prompts)
|
||||
]
|
||||
|
||||
# Process completed requests as they finish
|
||||
completed = 0
|
||||
for coro in asyncio.as_completed(tasks):
|
||||
try:
|
||||
result = await coro
|
||||
completed += 1
|
||||
prompt = result['prompt']
|
||||
status_icon = "✅" if result['status'] == 'success' else "❌"
|
||||
output_info = f" -> {result.get('output_path', 'N/A')}" if result['status'] == 'success' else ""
|
||||
print(f"{status_icon} [{completed}/{len(all_prompts)}] {result['latency']:.2f}s - {prompt[:50]}...{output_info}")
|
||||
except Exception as e:
|
||||
completed += 1
|
||||
print(f"❌ [{completed}/{len(all_prompts)}] Exception: {e}")
|
||||
|
||||
self.analyze_results()
|
||||
|
||||
def analyze_results(self):
|
||||
"""Analyze and print test results"""
|
||||
print("\n" + "=" * 80)
|
||||
print("STRESS TEST RESULTS")
|
||||
print("=" * 80)
|
||||
|
||||
successful_requests = [r for r in self.results if r['status'] == 'success']
|
||||
failed_requests = [r for r in self.results if r['status'] == 'error']
|
||||
|
||||
print(f"Total Requests: {len(self.results)}")
|
||||
print(f"Successful: {len(successful_requests)}")
|
||||
print(f"Failed: {len(failed_requests)}")
|
||||
print(f"Success Rate: {len(successful_requests)/len(self.results)*100:.1f}%")
|
||||
|
||||
if successful_requests:
|
||||
latencies = [r['latency'] for r in successful_requests]
|
||||
print(f"\nLatency Statistics (seconds):")
|
||||
print(f" Min: {min(latencies):.2f}")
|
||||
print(f" Max: {max(latencies):.2f}")
|
||||
print(f" Mean: {statistics.mean(latencies):.2f}")
|
||||
print(f" Median: {statistics.median(latencies):.2f}")
|
||||
print(f" Std Dev: {statistics.stdev(latencies):.2f}")
|
||||
|
||||
# Percentiles
|
||||
sorted_latencies = sorted(latencies)
|
||||
p50 = sorted_latencies[int(len(sorted_latencies) * 0.5)]
|
||||
p90 = sorted_latencies[int(len(sorted_latencies) * 0.9)]
|
||||
p95 = sorted_latencies[int(len(sorted_latencies) * 0.95)]
|
||||
p99 = sorted_latencies[int(len(sorted_latencies) * 0.99)]
|
||||
|
||||
print(f" P50: {p50:.2f}")
|
||||
print(f" P90: {p90:.2f}")
|
||||
print(f" P95: {p95:.2f}")
|
||||
print(f" P99: {p99:.2f}")
|
||||
|
||||
if failed_requests:
|
||||
print(f"\nFailed Requests ({len(failed_requests)}):")
|
||||
for req in failed_requests[:5]: # Show first 5 failures
|
||||
print(f" - {req['error']}")
|
||||
if len(failed_requests) > 5:
|
||||
print(f" ... and {len(failed_requests) - 5} more")
|
||||
|
||||
# Save detailed results
|
||||
results_file = os.path.join(self.output_path, "stress_test_results.json")
|
||||
os.makedirs(self.output_path, exist_ok=True)
|
||||
|
||||
with open(results_file, 'w') as f:
|
||||
json.dump({
|
||||
'summary': {
|
||||
'total_requests': len(self.results),
|
||||
'successful_requests': len(successful_requests),
|
||||
'failed_requests': len(failed_requests),
|
||||
'success_rate': len(successful_requests)/len(self.results)*100 if self.results else 0
|
||||
},
|
||||
'latency_stats': {
|
||||
'min': min(latencies) if successful_requests else 0,
|
||||
'max': max(latencies) if successful_requests else 0,
|
||||
'mean': statistics.mean(latencies) if successful_requests else 0,
|
||||
'median': statistics.median(latencies) if successful_requests else 0,
|
||||
'std_dev': statistics.stdev(latencies) if len(successful_requests) > 1 else 0
|
||||
},
|
||||
'detailed_results': self.results
|
||||
}, f, indent=2)
|
||||
|
||||
print(f"\nDetailed results saved to: {results_file}")
|
||||
|
||||
async def main():
|
||||
parser = argparse.ArgumentParser(description="FastVideo Ray Serve Backend Stress Test")
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default="outputs",
|
||||
help="Path to save test results")
|
||||
parser.add_argument("--server_url",
|
||||
type=str,
|
||||
default="http://localhost:8000",
|
||||
help="Ray Serve backend URL")
|
||||
parser.add_argument("--max_concurrent",
|
||||
type=int,
|
||||
default=50,
|
||||
help="Maximum concurrent requests")
|
||||
parser.add_argument("--iterations",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Number of iterations through all prompts")
|
||||
parser.add_argument("--concurrent_requests",
|
||||
type=int,
|
||||
default=None,
|
||||
help="Number of concurrent requests (overrides max_concurrent)")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Create stress test instance
|
||||
stress_test = BackendStressTest(
|
||||
output_path=args.output_path,
|
||||
server_url=args.server_url,
|
||||
max_concurrent=args.max_concurrent
|
||||
)
|
||||
|
||||
# Run the stress test
|
||||
await stress_test.run_stress_test(
|
||||
num_iterations=args.iterations,
|
||||
concurrent_requests=args.concurrent_requests
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,169 @@
|
||||
import argparse
|
||||
import os
|
||||
from copy import deepcopy
|
||||
|
||||
import gradio as gr
|
||||
import torch
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample.base import SamplingParam
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="FastVideo Gradio Demo")
|
||||
parser.add_argument("--model_path",
|
||||
type=str,
|
||||
default="FastVideo/FastHunyuan-diffusers",
|
||||
help="Path to the model")
|
||||
parser.add_argument("--num_gpus",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Number of GPUs to use")
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default="outputs",
|
||||
help="Path to save generated videos")
|
||||
parsed_args = parser.parse_args()
|
||||
|
||||
# args = FastVideoArgs(model_path="FastVideo/FastHunyuan-Diffusers", num_gpus=2)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_path=parsed_args.model_path, num_gpus=parsed_args.num_gpus)
|
||||
|
||||
default_params = SamplingParam.from_pretrained(parsed_args.model_path)
|
||||
|
||||
def generate_video(
|
||||
prompt,
|
||||
negative_prompt,
|
||||
use_negative_prompt,
|
||||
seed,
|
||||
guidance_scale,
|
||||
num_frames,
|
||||
height,
|
||||
width,
|
||||
num_inference_steps,
|
||||
randomize_seed=False,
|
||||
):
|
||||
params = deepcopy(default_params)
|
||||
params.prompt = prompt
|
||||
params.negative_prompt = negative_prompt
|
||||
params.seed = seed
|
||||
params.guidance_scale = guidance_scale
|
||||
params.num_frames = num_frames
|
||||
params.height = height
|
||||
params.width = width
|
||||
params.num_inference_steps = num_inference_steps
|
||||
|
||||
if randomize_seed:
|
||||
params.seed = torch.randint(0, 1000000, (1, )).item()
|
||||
|
||||
if not use_negative_prompt:
|
||||
params.negative_prompt = None
|
||||
|
||||
generator.generate_video(prompt=prompt, sampling_param=params)
|
||||
|
||||
output_path = os.path.join(parsed_args.output_path,
|
||||
f"{params.prompt[:100]}.mp4")
|
||||
|
||||
return output_path, params.seed
|
||||
|
||||
examples = [
|
||||
"A hand enters the frame, pulling a sheet of plastic wrap over three balls of dough placed on a wooden surface. The plastic wrap is stretched to cover the dough more securely. The hand adjusts the wrap, ensuring that it is tight and smooth over the dough. The scene focuses on the hand’s movements as it secures the edges of the plastic wrap. No new objects appear, and the camera remains stationary, focusing on the action of covering the dough.",
|
||||
"A vintage train snakes through the mountains, its plume of white steam rising dramatically against the jagged peaks. The cars glint in the late afternoon sun, their deep crimson and gold accents lending a touch of elegance. The tracks carve a precarious path along the cliffside, revealing glimpses of a roaring river far below. Inside, passengers peer out the large windows, their faces lit with awe as the landscape unfolds.",
|
||||
"A crowded rooftop bar buzzes with energy, the city skyline twinkling like a field of stars in the background. Strings of fairy lights hang above, casting a warm, golden glow over the scene. Groups of people gather around high tables, their laughter blending with the soft rhythm of live jazz. The aroma of freshly mixed cocktails and charred appetizers wafts through the air, mingling with the cool night breeze.",
|
||||
]
|
||||
|
||||
with gr.Blocks() as demo:
|
||||
gr.Markdown("# FastVideo Inference Demo")
|
||||
|
||||
with gr.Group():
|
||||
with gr.Row():
|
||||
prompt = gr.Text(
|
||||
label="Prompt",
|
||||
show_label=False,
|
||||
max_lines=1,
|
||||
placeholder="Enter your prompt",
|
||||
container=False,
|
||||
)
|
||||
run_button = gr.Button("Run", scale=0)
|
||||
result = gr.Video(label="Result", show_label=False)
|
||||
|
||||
with gr.Accordion("Advanced options", open=False):
|
||||
with gr.Group():
|
||||
with gr.Row():
|
||||
height = gr.Slider(
|
||||
label="Height",
|
||||
minimum=256,
|
||||
maximum=1024,
|
||||
step=32,
|
||||
value=default_params.height,
|
||||
)
|
||||
width = gr.Slider(label="Width",
|
||||
minimum=256,
|
||||
maximum=1024,
|
||||
step=32,
|
||||
value=default_params.width)
|
||||
|
||||
with gr.Row():
|
||||
num_frames = gr.Slider(
|
||||
label="Number of Frames",
|
||||
minimum=21,
|
||||
maximum=163,
|
||||
value=default_params.num_frames,
|
||||
)
|
||||
guidance_scale = gr.Slider(
|
||||
label="Guidance Scale",
|
||||
minimum=1,
|
||||
maximum=12,
|
||||
value=default_params.guidance_scale,
|
||||
)
|
||||
num_inference_steps = gr.Slider(
|
||||
label="Inference Steps",
|
||||
minimum=4,
|
||||
maximum=100,
|
||||
value=default_params.num_inference_steps,
|
||||
)
|
||||
|
||||
with gr.Row():
|
||||
use_negative_prompt = gr.Checkbox(
|
||||
label="Use negative prompt", value=False)
|
||||
negative_prompt = gr.Text(
|
||||
label="Negative prompt",
|
||||
max_lines=1,
|
||||
placeholder="Enter a negative prompt",
|
||||
visible=False,
|
||||
)
|
||||
|
||||
seed = gr.Slider(label="Seed",
|
||||
minimum=0,
|
||||
maximum=1000000,
|
||||
step=1,
|
||||
value=default_params.seed)
|
||||
randomize_seed = gr.Checkbox(label="Randomize seed", value=True)
|
||||
seed_output = gr.Number(label="Used Seed")
|
||||
|
||||
gr.Examples(examples=examples, inputs=prompt)
|
||||
|
||||
use_negative_prompt.change(
|
||||
fn=lambda x: gr.update(visible=x),
|
||||
inputs=use_negative_prompt,
|
||||
outputs=default_params.negative_prompt,
|
||||
)
|
||||
|
||||
run_button.click(
|
||||
fn=generate_video,
|
||||
inputs=[
|
||||
prompt,
|
||||
negative_prompt,
|
||||
use_negative_prompt,
|
||||
seed,
|
||||
guidance_scale,
|
||||
num_frames,
|
||||
height,
|
||||
width,
|
||||
num_inference_steps,
|
||||
randomize_seed,
|
||||
],
|
||||
outputs=[result, seed_output],
|
||||
)
|
||||
|
||||
demo.queue(max_size=20).launch(server_name="0.0.0.0", server_port=7860)
|
||||
@@ -0,0 +1,182 @@
|
||||
import argparse
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import threading
|
||||
import signal
|
||||
import requests
|
||||
from pathlib import Path
|
||||
|
||||
# Add the project root to the Python path
|
||||
project_root = Path(__file__).parent.parent.parent.parent
|
||||
sys.path.insert(0, str(project_root))
|
||||
|
||||
|
||||
def check_frontend_health(frontend_url: str, max_retries: int = 30) -> bool:
|
||||
"""Check if the frontend is healthy"""
|
||||
for i in range(max_retries):
|
||||
try:
|
||||
response = requests.get(frontend_url, timeout=5)
|
||||
if response.status_code == 200:
|
||||
print(f"✅ Frontend is healthy at {frontend_url}")
|
||||
return True
|
||||
except requests.exceptions.RequestException:
|
||||
pass
|
||||
|
||||
if i < max_retries - 1:
|
||||
print(f"⏳ Waiting for frontend to start... ({i+1}/{max_retries})")
|
||||
time.sleep(2)
|
||||
|
||||
print(f"❌ Frontend failed to start within {max_retries * 2} seconds")
|
||||
return False
|
||||
|
||||
|
||||
def start_frontend_instance(args, instance_id: int, backend_url: str):
|
||||
"""Start a single frontend instance"""
|
||||
frontend_script = Path(__file__).parent / "gradio_frontend.py"
|
||||
frontend_port = args.frontend_base_port + instance_id
|
||||
|
||||
cmd = [
|
||||
sys.executable, str(frontend_script),
|
||||
"--backend_url", backend_url,
|
||||
"--t2v_model_path", args.t2v_model_path,
|
||||
"--i2v_model_path", args.i2v_model_path,
|
||||
"--host", args.frontend_host,
|
||||
"--port", str(frontend_port)
|
||||
]
|
||||
|
||||
print(f"🎨 Starting Frontend {instance_id + 1} on port {frontend_port}...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
# Start the frontend process
|
||||
frontend_process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
# Monitor frontend output
|
||||
def monitor_frontend():
|
||||
for line in frontend_process.stdout:
|
||||
print(f"[FRONTEND-{instance_id + 1}] {line.rstrip()}")
|
||||
|
||||
monitor_thread = threading.Thread(target=monitor_frontend, daemon=True)
|
||||
monitor_thread.start()
|
||||
|
||||
return frontend_process, frontend_port
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="FastVideo Multi-Frontend Launcher")
|
||||
|
||||
# Model and output settings
|
||||
parser.add_argument("--t2v_model_path",
|
||||
type=str,
|
||||
default="FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
help="Path to the T2V model")
|
||||
parser.add_argument("--i2v_model_path",
|
||||
type=str,
|
||||
default="Wan-AI/Wan2.1-I2V-14B-480P-Diffusers",
|
||||
help="Path to the I2V model")
|
||||
|
||||
# Frontend settings
|
||||
parser.add_argument("--frontend_host",
|
||||
type=str,
|
||||
default="0.0.0.0",
|
||||
help="Frontend host to bind to")
|
||||
parser.add_argument("--frontend_base_port",
|
||||
type=int,
|
||||
default=7860,
|
||||
help="Base port for frontend instances")
|
||||
parser.add_argument("--num_frontends",
|
||||
type=int,
|
||||
default=2,
|
||||
help="Number of frontend instances to start")
|
||||
|
||||
# Backend settings
|
||||
parser.add_argument("--backend_url",
|
||||
type=str,
|
||||
default="http://localhost:8000",
|
||||
help="Backend URL for frontends to connect to")
|
||||
|
||||
# Other settings
|
||||
parser.add_argument("--skip_health_check",
|
||||
action="store_true",
|
||||
help="Skip frontend health check")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
print("🎬 FastVideo Multi-Frontend Launcher")
|
||||
print("=" * 50)
|
||||
print(f"T2V Model: {args.t2v_model_path}")
|
||||
print(f"I2V Model: {args.i2v_model_path}")
|
||||
print(f"Backend URL: {args.backend_url}")
|
||||
print(f"Number of Frontends: {args.num_frontends}")
|
||||
print(f"Frontend Base Port: {args.frontend_base_port}")
|
||||
print("=" * 50)
|
||||
|
||||
# Start multiple frontend instances
|
||||
frontend_processes = []
|
||||
frontend_urls = []
|
||||
|
||||
for i in range(args.num_frontends):
|
||||
process, port = start_frontend_instance(args, i, args.backend_url)
|
||||
frontend_processes.append(process)
|
||||
frontend_urls.append(f"http://{args.frontend_host}:{port}")
|
||||
|
||||
# Wait for frontends to be ready
|
||||
if not args.skip_health_check:
|
||||
print("\n⏳ Waiting for frontends to start...")
|
||||
for i, url in enumerate(frontend_urls):
|
||||
if not check_frontend_health(url):
|
||||
print(f"❌ Frontend {i + 1} failed to start. Terminating...")
|
||||
for process in frontend_processes:
|
||||
process.terminate()
|
||||
sys.exit(1)
|
||||
|
||||
print("\n🎉 All frontend instances are starting up!")
|
||||
for i, url in enumerate(frontend_urls):
|
||||
print(f"📺 Frontend {i + 1}: {url}")
|
||||
print("\nPress Ctrl+C to stop all frontend instances...")
|
||||
|
||||
# Signal handler for graceful shutdown
|
||||
def signal_handler(signum, frame):
|
||||
print("\n🛑 Shutting down frontend instances...")
|
||||
for process in frontend_processes:
|
||||
process.terminate()
|
||||
|
||||
# Wait for processes to terminate
|
||||
try:
|
||||
for process in frontend_processes:
|
||||
process.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
print("⚠️ Force killing processes...")
|
||||
for process in frontend_processes:
|
||||
process.kill()
|
||||
|
||||
print("✅ Frontend instances stopped")
|
||||
sys.exit(0)
|
||||
|
||||
signal.signal(signal.SIGINT, signal_handler)
|
||||
signal.signal(signal.SIGTERM, signal_handler)
|
||||
|
||||
# Monitor processes
|
||||
try:
|
||||
while True:
|
||||
# Check if processes are still running
|
||||
for i, process in enumerate(frontend_processes):
|
||||
if process.poll() is not None:
|
||||
print(f"❌ Frontend {i + 1} process died unexpectedly")
|
||||
break
|
||||
|
||||
time.sleep(1)
|
||||
|
||||
except KeyboardInterrupt:
|
||||
signal_handler(signal.SIGINT, None)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,181 @@
|
||||
# Nginx configuration for FastVideo load balancing
|
||||
# This configuration implements the architecture:
|
||||
# ngrok -> nginx reverse proxy -> frontend1/frontend2 -> backend1×8/backend2×8
|
||||
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
# Basic settings
|
||||
sendfile on;
|
||||
tcp_nopush on;
|
||||
tcp_nodelay on;
|
||||
keepalive_timeout 65;
|
||||
types_hash_max_size 2048;
|
||||
client_max_body_size 100M; # Allow large video uploads
|
||||
|
||||
# Logging
|
||||
access_log /mnt/fast-disks/nfs/hao_lab/FastVideo/outputs/nginx_access.log;
|
||||
error_log /mnt/fast-disks/nfs/hao_lab/FastVideo/outputs/nginx_error.log;
|
||||
|
||||
# Gzip compression
|
||||
gzip on;
|
||||
gzip_vary on;
|
||||
gzip_min_length 1024;
|
||||
gzip_proxied any;
|
||||
gzip_comp_level 6;
|
||||
gzip_types
|
||||
text/plain
|
||||
text/css
|
||||
text/xml
|
||||
text/javascript
|
||||
application/json
|
||||
application/javascript
|
||||
application/xml+rss
|
||||
application/atom+xml
|
||||
image/svg+xml;
|
||||
|
||||
# Upstream for frontend load balancing
|
||||
upstream frontend_servers {
|
||||
# Round-robin load balancing between frontends
|
||||
server 127.0.0.1:7860 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:7861 weight=1 max_fails=3 fail_timeout=30s;
|
||||
upstream frontend_servers {
|
||||
# Round-robin load balancing between frontends
|
||||
server 127.0.0.1:7860 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:7861 weight=1 max_fails=3 fail_timeout=30s;
|
||||
|
||||
# Health check
|
||||
keepalive 32;
|
||||
}
|
||||
|
||||
# Upstream for backend1 load balancing
|
||||
upstream backend1_servers {
|
||||
server 127.0.0.1:8000 weight=1 max_fails=3 fail_timeout=30s; (8 replicas)
|
||||
upstream backend1_servers {
|
||||
# Round-robin load balancing for backend1 replicas
|
||||
server 127.0.0.1:8000 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8001 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8002 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8003 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8004 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8005 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8006 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8007 weight=1 max_fails=3 fail_timeout=30s;
|
||||
|
||||
keepalive 32;
|
||||
}
|
||||
|
||||
# Upstream for backend2 load balancing
|
||||
upstream backend2_servers {
|
||||
server 127.0.0.1:8000 weight=1 max_fails=3 fail_timeout=30s; (8 replicas)
|
||||
upstream backend2_servers {
|
||||
# Round-robin load balancing for backend2 replicas
|
||||
server 127.0.0.1:8010 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8011 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8012 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8013 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8014 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8015 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8016 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8017 weight=1 max_fails=3 fail_timeout=30s;
|
||||
|
||||
keepalive 32;
|
||||
}
|
||||
|
||||
# Main server block
|
||||
server {
|
||||
listen 80;
|
||||
server_name localhost;
|
||||
|
||||
# Security headers
|
||||
add_header X-Frame-Options "SAMEORIGIN" always;
|
||||
add_header X-Content-Type-Options "nosniff" always;
|
||||
add_header X-XSS-Protection "1; mode=block" always;
|
||||
add_header Referrer-Policy "no-referrer-when-downgrade" always;
|
||||
|
||||
# Frontend routes (Gradio interfaces)
|
||||
location / {
|
||||
proxy_pass http://frontend_servers;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# WebSocket support for Gradio
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
|
||||
# Timeouts
|
||||
proxy_connect_timeout 60s;
|
||||
proxy_send_timeout 60s;
|
||||
proxy_read_timeout 60s;
|
||||
|
||||
# Buffer settings
|
||||
proxy_buffering on;
|
||||
proxy_buffer_size 128k;
|
||||
proxy_buffers 4 256k;
|
||||
proxy_busy_buffers_size 256k;
|
||||
}
|
||||
|
||||
# Backend API routes for frontend1
|
||||
location /api/frontend1/ {
|
||||
# Strip the /api/frontend1/ prefix
|
||||
rewrite ^/api/frontend1/(.*) /$1 break;
|
||||
proxy_pass http://backend1_servers;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Timeouts for video generation
|
||||
proxy_connect_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
proxy_read_timeout 300s;
|
||||
|
||||
# Buffer settings for large responses
|
||||
proxy_buffering on;
|
||||
proxy_buffer_size 128k;
|
||||
proxy_buffers 4 256k;
|
||||
proxy_busy_buffers_size 256k;
|
||||
}
|
||||
|
||||
# Backend API routes for frontend2
|
||||
location /api/frontend2/ {
|
||||
# Strip the /api/frontend2/ prefix
|
||||
rewrite ^/api/frontend2/(.*) /$1 break;
|
||||
proxy_pass http://backend2_servers;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Timeouts for video generation
|
||||
proxy_connect_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
proxy_read_timeout 300s;
|
||||
|
||||
# Buffer settings for large responses
|
||||
proxy_buffering on;
|
||||
proxy_buffer_size 128k;
|
||||
proxy_buffers 4 256k;
|
||||
proxy_busy_buffers_size 256k;
|
||||
}
|
||||
|
||||
# Health check endpoint
|
||||
location /health {
|
||||
access_log off;
|
||||
return 200 "healthy\n";
|
||||
add_header Content-Type text/plain;
|
||||
}
|
||||
|
||||
# Static files (if needed)
|
||||
location /static/ {
|
||||
alias /var/www/static/;
|
||||
expires 1y;
|
||||
add_header Cache-Control "public, immutable";
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,173 @@
|
||||
# Nginx configuration for FastVideo load balancing
|
||||
# This configuration implements the architecture:
|
||||
# ngrok -> nginx reverse proxy -> frontend1/frontend2 -> backend1×8/backend2×8
|
||||
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
# Basic settings
|
||||
sendfile on;
|
||||
tcp_nopush on;
|
||||
tcp_nodelay on;
|
||||
keepalive_timeout 65;
|
||||
types_hash_max_size 2048;
|
||||
client_max_body_size 100M; # Allow large video uploads
|
||||
|
||||
# Logging
|
||||
access_log /var/log/nginx/access.log;
|
||||
error_log /var/log/nginx/error.log;
|
||||
|
||||
# Gzip compression
|
||||
gzip on;
|
||||
gzip_vary on;
|
||||
gzip_min_length 1024;
|
||||
gzip_proxied any;
|
||||
gzip_comp_level 6;
|
||||
gzip_types
|
||||
text/plain
|
||||
text/css
|
||||
text/xml
|
||||
text/javascript
|
||||
application/json
|
||||
application/javascript
|
||||
application/xml+rss
|
||||
application/atom+xml
|
||||
image/svg+xml;
|
||||
|
||||
# Upstream for frontend load balancing
|
||||
upstream frontend_servers {
|
||||
# Round-robin load balancing between frontends
|
||||
server 127.0.0.1:7860 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:7861 weight=1 max_fails=3 fail_timeout=30s;
|
||||
|
||||
# Health check
|
||||
keepalive 32;
|
||||
}
|
||||
|
||||
# Upstream for backend1 load balancing (8 replicas)
|
||||
upstream backend1_servers {
|
||||
# Round-robin load balancing for backend1 replicas
|
||||
server 127.0.0.1:8000 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8001 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8002 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8003 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8004 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8005 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8006 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8007 weight=1 max_fails=3 fail_timeout=30s;
|
||||
|
||||
keepalive 32;
|
||||
}
|
||||
|
||||
# Upstream for backend2 load balancing (8 replicas)
|
||||
upstream backend2_servers {
|
||||
# Round-robin load balancing for backend2 replicas
|
||||
server 127.0.0.1:8010 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8011 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8012 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8013 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8014 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8015 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8016 weight=1 max_fails=3 fail_timeout=30s;
|
||||
server 127.0.0.1:8017 weight=1 max_fails=3 fail_timeout=30s;
|
||||
|
||||
keepalive 32;
|
||||
}
|
||||
|
||||
# Main server block
|
||||
server {
|
||||
listen 80;
|
||||
server_name localhost;
|
||||
|
||||
# Security headers
|
||||
add_header X-Frame-Options "SAMEORIGIN" always;
|
||||
add_header X-Content-Type-Options "nosniff" always;
|
||||
add_header X-XSS-Protection "1; mode=block" always;
|
||||
add_header Referrer-Policy "no-referrer-when-downgrade" always;
|
||||
|
||||
# Frontend routes (Gradio interfaces)
|
||||
location / {
|
||||
proxy_pass http://frontend_servers;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# WebSocket support for Gradio
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
|
||||
# Timeouts
|
||||
proxy_connect_timeout 60s;
|
||||
proxy_send_timeout 60s;
|
||||
proxy_read_timeout 60s;
|
||||
|
||||
# Buffer settings
|
||||
proxy_buffering on;
|
||||
proxy_buffer_size 128k;
|
||||
proxy_buffers 4 256k;
|
||||
proxy_busy_buffers_size 256k;
|
||||
}
|
||||
|
||||
# Backend API routes for frontend1
|
||||
location /api/frontend1/ {
|
||||
# Strip the /api/frontend1/ prefix
|
||||
rewrite ^/api/frontend1/(.*) /$1 break;
|
||||
proxy_pass http://backend1_servers;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Timeouts for video generation
|
||||
proxy_connect_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
proxy_read_timeout 300s;
|
||||
|
||||
# Buffer settings for large responses
|
||||
proxy_buffering on;
|
||||
proxy_buffer_size 128k;
|
||||
proxy_buffers 4 256k;
|
||||
proxy_busy_buffers_size 256k;
|
||||
}
|
||||
|
||||
# Backend API routes for frontend2
|
||||
location /api/frontend2/ {
|
||||
# Strip the /api/frontend2/ prefix
|
||||
rewrite ^/api/frontend2/(.*) /$1 break;
|
||||
proxy_pass http://backend2_servers;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Timeouts for video generation
|
||||
proxy_connect_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
proxy_read_timeout 300s;
|
||||
|
||||
# Buffer settings for large responses
|
||||
proxy_buffering on;
|
||||
proxy_buffer_size 128k;
|
||||
proxy_buffers 4 256k;
|
||||
proxy_busy_buffers_size 256k;
|
||||
}
|
||||
|
||||
# Health check endpoint
|
||||
location /health {
|
||||
access_log off;
|
||||
return 200 "healthy\n";
|
||||
add_header Content-Type text/plain;
|
||||
}
|
||||
|
||||
# Static files (if needed)
|
||||
location /static/ {
|
||||
alias /var/www/static/;
|
||||
expires 1y;
|
||||
add_header Cache-Control "public, immutable";
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -5,45 +5,16 @@ import base64
|
||||
import io
|
||||
from copy import deepcopy
|
||||
from typing import Dict, Any, Optional, List
|
||||
import signal
|
||||
import sys
|
||||
|
||||
import ray
|
||||
from ray import serve
|
||||
from fastapi import FastAPI, Request, Response
|
||||
from fastapi import FastAPI, Request
|
||||
from pydantic import BaseModel
|
||||
from PIL import Image
|
||||
import numpy as np
|
||||
from slowapi import Limiter, _rate_limit_exceeded_handler
|
||||
from slowapi.util import get_remote_address
|
||||
from slowapi.errors import RateLimitExceeded
|
||||
import imageio
|
||||
from ray.serve.handle import DeploymentHandle
|
||||
from prometheus_client import Counter, Histogram, generate_latest
|
||||
|
||||
NUM_GPUS = 16
|
||||
DEFAULT_FPS = 16
|
||||
SEED_RANGE_MAX = 1_000_000
|
||||
SUPPORTED_MODELS = [
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
"FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers",
|
||||
]
|
||||
|
||||
MODEL_CONFIGS = {
|
||||
"1.3B": {
|
||||
"num_cpus": 2,
|
||||
"text_encoder_cpu_offload": False,
|
||||
"dit_cpu_offload": False,
|
||||
"vae_cpu_offload": False,
|
||||
"VSA_sparsity": 0.8,
|
||||
},
|
||||
"14B": {
|
||||
"num_cpus": 16,
|
||||
"text_encoder_cpu_offload": True,
|
||||
"dit_cpu_offload": True,
|
||||
"vae_cpu_offload": False,
|
||||
"VSA_sparsity": 0.9,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
class VideoGenerationRequest(BaseModel):
|
||||
@@ -55,297 +26,284 @@ class VideoGenerationRequest(BaseModel):
|
||||
num_frames: int = 21
|
||||
height: int = 448
|
||||
width: int = 832
|
||||
num_inference_steps: int = 20
|
||||
randomize_seed: bool = False
|
||||
return_frames: bool = False
|
||||
model_path: Optional[str] = None
|
||||
return_frames: bool = False # Whether to return base64 encoded frames
|
||||
image_path: Optional[str] = None # Path to input image for I2V
|
||||
model_type: str = "t2v" # "t2v" or "i2v" to specify which model to use
|
||||
|
||||
|
||||
class VideoGenerationResponse(BaseModel):
|
||||
video_data: Optional[str] = None
|
||||
output_path: str
|
||||
seed: int
|
||||
success: bool
|
||||
error_message: Optional[str] = None
|
||||
generation_time: Optional[float] = None
|
||||
model_load_time: Optional[float] = None
|
||||
inference_time: Optional[float] = None
|
||||
encoding_time: Optional[float] = None
|
||||
total_time: Optional[float] = None
|
||||
stage_names: Optional[List[str]] = None
|
||||
stage_execution_times: Optional[List[float]] = None
|
||||
frames: Optional[List[str]] = None # Base64 encoded frames
|
||||
|
||||
|
||||
def encode_video_to_base64(frames: List[np.ndarray], fps: int = DEFAULT_FPS) -> str:
|
||||
def encode_frames_to_base64(frames: List[np.ndarray]) -> List[str]:
|
||||
"""Convert numpy frames (0-255) to base64-encoded PNG images"""
|
||||
if not frames:
|
||||
return ""
|
||||
return []
|
||||
|
||||
try:
|
||||
buffer = io.BytesIO()
|
||||
imageio.mimsave(buffer, frames, fps=fps, format="mp4")
|
||||
buffer.seek(0)
|
||||
|
||||
video_base64 = base64.b64encode(buffer.getvalue()).decode('utf-8')
|
||||
return f"data:video/mp4;base64,{video_base64}"
|
||||
|
||||
except Exception as e:
|
||||
print(f"Warning: Failed to encode video: {e}")
|
||||
return ""
|
||||
|
||||
|
||||
def setup_model_environment(model_path: str) -> None:
|
||||
if "fullattn" in model_path.lower():
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "FLASH_ATTN"
|
||||
else:
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
|
||||
os.environ["FASTVIDEO_STAGE_LOGGING"] = "1"
|
||||
|
||||
|
||||
def process_generation_result(result: Any) -> tuple[List[np.ndarray], float, List[str], List[float]]:
|
||||
frames = result if isinstance(result, list) else result.get("frames", [])
|
||||
generation_time = result.get("generation_time", 0.0) if isinstance(result, dict) else 0.0
|
||||
encoded_frames = []
|
||||
|
||||
logging_info = result.get("logging_info", None)
|
||||
if logging_info:
|
||||
stage_names = logging_info.get_execution_order()
|
||||
stage_execution_times = [
|
||||
logging_info.get_stage_info(stage_name).get("execution_time", 0.0)
|
||||
for stage_name in stage_names
|
||||
]
|
||||
else:
|
||||
stage_names = []
|
||||
stage_execution_times = []
|
||||
for i, frame in enumerate(frames):
|
||||
try:
|
||||
# Ensure frame is numpy array
|
||||
if not isinstance(frame, np.ndarray):
|
||||
print(f"Warning: Frame {i} is not a numpy array, skipping")
|
||||
continue
|
||||
|
||||
# Ensure frame is uint8
|
||||
if frame.dtype != np.uint8:
|
||||
# Clip values to 0-255 range and convert to uint8
|
||||
frame = np.clip(frame, 0, 255).astype(np.uint8)
|
||||
|
||||
# Convert numpy array to PIL Image
|
||||
if len(frame.shape) == 3 and frame.shape[2] == 3:
|
||||
# RGB image
|
||||
pil_image = Image.fromarray(frame, mode='RGB')
|
||||
elif len(frame.shape) == 3 and frame.shape[2] == 4:
|
||||
# RGBA image
|
||||
pil_image = Image.fromarray(frame, mode='RGBA')
|
||||
elif len(frame.shape) == 2:
|
||||
# Grayscale image
|
||||
pil_image = Image.fromarray(frame, mode='L')
|
||||
else:
|
||||
print(f"Warning: Frame {i} has unsupported shape {frame.shape}, skipping")
|
||||
continue
|
||||
|
||||
# Save to bytes buffer as PNG
|
||||
buffer = io.BytesIO()
|
||||
pil_image.save(buffer, format='PNG')
|
||||
buffer.seek(0)
|
||||
|
||||
# Encode to base64
|
||||
img_base64 = base64.b64encode(buffer.getvalue()).decode('utf-8')
|
||||
encoded_frames.append(f"data:image/png;base64,{img_base64}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Warning: Failed to encode frame {i}: {e}")
|
||||
continue
|
||||
|
||||
return frames, generation_time, stage_names, stage_execution_times
|
||||
|
||||
|
||||
def prepare_sampling_params(video_request: VideoGenerationRequest, default_params: Any) -> Any:
|
||||
params = deepcopy(default_params)
|
||||
params.prompt = video_request.prompt
|
||||
|
||||
if video_request.use_negative_prompt:
|
||||
params.negative_prompt = video_request.negative_prompt
|
||||
|
||||
params.seed = (video_request.seed if not video_request.randomize_seed
|
||||
else torch.randint(0, SEED_RANGE_MAX, (1,)).item())
|
||||
params.randomize_seed = video_request.randomize_seed
|
||||
params.guidance_scale = video_request.guidance_scale
|
||||
params.num_frames = video_request.num_frames
|
||||
params.height = video_request.height
|
||||
params.width = video_request.width
|
||||
params.save_video = False
|
||||
params.return_frames = False
|
||||
|
||||
return params
|
||||
|
||||
|
||||
class BaseModelDeployment:
|
||||
def __init__(self, model_path: str, output_path: str = "outputs"):
|
||||
self.model_path = model_path
|
||||
self.output_path = output_path
|
||||
self.generator = None
|
||||
self.default_params = None
|
||||
|
||||
os.makedirs(self.output_path, exist_ok=True)
|
||||
setup_model_environment(self.model_path)
|
||||
|
||||
def _initialize_generator(self, config: Dict[str, Any]) -> None:
|
||||
from fastvideo.entrypoints.video_generator import VideoGenerator
|
||||
from fastvideo.configs.sample.base import SamplingParam
|
||||
|
||||
print(f"Initializing model: {self.model_path}")
|
||||
self.generator = VideoGenerator.from_pretrained(
|
||||
model_path=self.model_path,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
text_encoder_cpu_offload=config["text_encoder_cpu_offload"],
|
||||
dit_cpu_offload=config["dit_cpu_offload"],
|
||||
vae_cpu_offload=config["vae_cpu_offload"],
|
||||
VSA_sparsity=config["VSA_sparsity"],
|
||||
enable_stage_verification=False,
|
||||
)
|
||||
self.default_params = SamplingParam.from_pretrained(self.model_path)
|
||||
|
||||
def generate_video(self, video_request: VideoGenerationRequest) -> VideoGenerationResponse:
|
||||
total_start_time = time.time()
|
||||
|
||||
params = prepare_sampling_params(video_request, self.default_params)
|
||||
|
||||
inference_start_time = time.time()
|
||||
result = self.generator.generate_video(
|
||||
prompt=video_request.prompt,
|
||||
sampling_param=params,
|
||||
save_video=False,
|
||||
return_frames=False,
|
||||
)
|
||||
inference_time = time.time() - inference_start_time
|
||||
|
||||
frames, generation_time, stage_names, stage_execution_times = process_generation_result(result)
|
||||
|
||||
encoding_start_time = time.time()
|
||||
video_data = encode_video_to_base64(frames, fps=DEFAULT_FPS)
|
||||
encoding_time = time.time() - encoding_start_time
|
||||
|
||||
total_time = time.time() - total_start_time
|
||||
|
||||
return VideoGenerationResponse(
|
||||
video_data=video_data,
|
||||
seed=params.seed,
|
||||
success=True,
|
||||
generation_time=generation_time,
|
||||
inference_time=inference_time,
|
||||
encoding_time=encoding_time,
|
||||
total_time=total_time,
|
||||
stage_names=stage_names,
|
||||
stage_execution_times=stage_execution_times,
|
||||
)
|
||||
|
||||
|
||||
@serve.deployment(
|
||||
ray_actor_options={"num_cpus": 2, "num_gpus": 1, "runtime_env": {"conda": "fv"}},
|
||||
)
|
||||
class T2VModelDeployment(BaseModelDeployment):
|
||||
def __init__(self, t2v_model_path: str, output_path: str = "outputs"):
|
||||
super().__init__(t2v_model_path, output_path)
|
||||
self._initialize_generator(MODEL_CONFIGS["1.3B"])
|
||||
print("✅ T2V model initialized successfully")
|
||||
|
||||
|
||||
@serve.deployment(
|
||||
ray_actor_options={"num_cpus": 16, "num_gpus": 1, "runtime_env": {"conda": "fv"}},
|
||||
)
|
||||
class T2V14BModelDeployment(BaseModelDeployment):
|
||||
def __init__(self, t2v_14b_model_path: str, output_path: str = "outputs"):
|
||||
super().__init__(t2v_14b_model_path, output_path)
|
||||
# Override environment for 14B model
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
|
||||
self._initialize_generator(MODEL_CONFIGS["14B"])
|
||||
print("✅ T2V 14B model initialized successfully")
|
||||
return encoded_frames
|
||||
|
||||
|
||||
# Create FastAPI app with rate limiting
|
||||
app = FastAPI()
|
||||
|
||||
# Initialize rate limiter
|
||||
limiter = Limiter(key_func=get_remote_address)
|
||||
app.state.limiter = limiter
|
||||
app.add_exception_handler(RateLimitExceeded, _rate_limit_exceeded_handler)
|
||||
|
||||
|
||||
@serve.deployment(num_replicas=50, ray_actor_options={"num_cpus": 2})
|
||||
@serve.deployment(
|
||||
num_replicas=8,
|
||||
# ray_actor_options={"num_cpus": 10, "num_gpus": 1, "runtime_env": {"conda": "fv", "working_dir": "/mnt/fast-disks/nfs/hao_lab/FastVideo"}},
|
||||
ray_actor_options={"num_cpus": 10, "num_gpus": 1, "runtime_env": {"conda": "fv"}},
|
||||
)
|
||||
@serve.ingress(app)
|
||||
class FastVideoAPI:
|
||||
|
||||
def __init__(self, t2v_deployments: Dict[str, DeploymentHandle]):
|
||||
self.t2v_deployments = t2v_deployments
|
||||
def __init__(self, t2v_model_path: str, i2v_model_path: str, output_path: str):
|
||||
self.t2v_model_path = t2v_model_path
|
||||
self.i2v_model_path = i2v_model_path
|
||||
self.output_path = output_path
|
||||
|
||||
# Initialize Prometheus metrics
|
||||
self.request_count = Counter('fastvideo_requests_total', 'Total FastVideo requests', ['model_type', 'status'])
|
||||
self.request_duration = Histogram('fastvideo_request_duration_seconds', 'FastVideo request duration', ['model_type'])
|
||||
self.video_generation_time = Histogram('fastvideo_video_generation_seconds', 'Video generation time', ['model_type'])
|
||||
|
||||
def _get_model_name(self, model_path: Optional[str]) -> str:
|
||||
return model_path.split('/')[-1] if model_path else "unknown"
|
||||
|
||||
def _record_metrics(self, model_name: str, status: str, duration: float, response: Optional[VideoGenerationResponse] = None) -> None:
|
||||
self.request_count.labels(model_type=model_name, status=status).inc()
|
||||
self.request_duration.labels(model_type=model_name).observe(duration)
|
||||
# Initialize the video generators
|
||||
self.t2v_generator = None # Initialize to None
|
||||
self.i2v_generator = None # Initialize to None
|
||||
self.t2v_default_params = None # Initialize to None
|
||||
self.i2v_default_params = None # Initialize to None
|
||||
|
||||
if response and hasattr(response, 'generation_time') and response.generation_time:
|
||||
self.video_generation_time.labels(model_type=model_name).observe(response.generation_time)
|
||||
# Ensure output directory exists
|
||||
os.makedirs(output_path, exist_ok=True)
|
||||
time.sleep(10)
|
||||
self._initialize_models() # Ensure models are initialized
|
||||
|
||||
def _initialize_models(self):
|
||||
# Set VSA environment variable
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
|
||||
# os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "FLASH_ATTN"
|
||||
|
||||
# Import only when needed - use direct imports to avoid module-level execution
|
||||
from fastvideo.entrypoints.video_generator import VideoGenerator
|
||||
from fastvideo.configs.sample.base import SamplingParam
|
||||
|
||||
# Initialize T2V model
|
||||
if self.t2v_generator is None:
|
||||
print(f"Initializing T2V model: {self.t2v_model_path}")
|
||||
self.t2v_generator = VideoGenerator.from_pretrained(
|
||||
model_path=self.t2v_model_path,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
# Adjust these offload parameters if you have < 32GB of VRAM
|
||||
text_encoder_cpu_offload=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
VSA_sparsity=0.8,
|
||||
# master_port=port,
|
||||
)
|
||||
self.t2v_default_params = SamplingParam.from_pretrained(self.t2v_model_path)
|
||||
print("✅ T2V model initialized successfully")
|
||||
|
||||
# Initialize I2V model
|
||||
# if self.i2v_generator is None:
|
||||
if False:
|
||||
print(f"Initializing I2V model: {self.i2v_model_path}")
|
||||
self.i2v_generator = VideoGenerator.from_pretrained(
|
||||
model_path=self.i2v_model_path,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
# Adjust these offload parameters if you have < 32GB of VRAM
|
||||
text_encoder_cpu_offload=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
VSA_sparsity=0.8,
|
||||
# master_port=port,
|
||||
)
|
||||
self.i2v_default_params = SamplingParam.from_pretrained(self.i2v_model_path)
|
||||
print("✅ I2V model initialized successfully")
|
||||
|
||||
@app.post("/generate_video", response_model=VideoGenerationResponse)
|
||||
@limiter.limit("10/minute")
|
||||
@limiter.limit("50/minute") # Allow 2 requests per minute per IP
|
||||
async def generate_video(self, request: Request, video_request: VideoGenerationRequest) -> VideoGenerationResponse:
|
||||
"""Route the request to the appropriate model deployment based on model_path."""
|
||||
start_time = time.time()
|
||||
model_name = self._get_model_name(video_request.model_path)
|
||||
|
||||
try:
|
||||
if video_request.model_path not in self.t2v_deployments:
|
||||
raise ValueError(f"Model {video_request.model_path} not found")
|
||||
# Select the appropriate model and parameters based on model_type
|
||||
if video_request.model_type.lower() == "i2v":
|
||||
generator = self.i2v_generator
|
||||
params = deepcopy(self.i2v_default_params)
|
||||
print(f"Using I2V model for generation")
|
||||
else:
|
||||
generator = self.t2v_generator
|
||||
params = deepcopy(self.t2v_default_params)
|
||||
print(f"Using T2V model for generation")
|
||||
|
||||
response_ref = self.t2v_deployments[video_request.model_path].generate_video.remote(video_request)
|
||||
response = await response_ref
|
||||
|
||||
self._record_metrics(model_name, "success", time.time() - start_time, response)
|
||||
return response
|
||||
# Update parameters with request values
|
||||
params.prompt = video_request.prompt
|
||||
# Only override negative prompt if user explicitly opts in
|
||||
if video_request.use_negative_prompt:
|
||||
params.negative_prompt = video_request.negative_prompt
|
||||
|
||||
except Exception as e:
|
||||
self._record_metrics(model_name, "error", time.time() - start_time)
|
||||
params.seed = video_request.seed
|
||||
params.guidance_scale = video_request.guidance_scale
|
||||
params.num_frames = video_request.num_frames
|
||||
params.height = video_request.height
|
||||
params.width = video_request.width
|
||||
params.num_inference_steps = video_request.num_inference_steps
|
||||
|
||||
# Handle seed randomization
|
||||
if video_request.randomize_seed:
|
||||
params.seed = torch.randint(0, 1000000, (1,)).item()
|
||||
|
||||
# Ensure negative_prompt is a non-None string; FastVideo validation disallows None
|
||||
if params.negative_prompt is None:
|
||||
params.negative_prompt = "" # empty string satisfies validator
|
||||
|
||||
# Set up output path and video saving
|
||||
params.save_video = True
|
||||
params.output_path = self.output_path
|
||||
# params.return_frames = False # avoid keeping frames in memory
|
||||
|
||||
# Create a clean filename from the prompt
|
||||
safe_prompt = video_request.prompt[:100].replace(' ', '_').replace('/', '_').replace('\\', '_')
|
||||
|
||||
# Store desired video name inside the SamplingParam to avoid unknown kwarg errors
|
||||
setattr(params, "output_video_name", safe_prompt)
|
||||
|
||||
# Handle image_path for I2V
|
||||
if video_request.image_path:
|
||||
params.image_path = video_request.image_path
|
||||
|
||||
# Generate the video with proper output path and filename
|
||||
result = generator.generate_video(
|
||||
prompt=video_request.prompt,
|
||||
sampling_param=params,
|
||||
save_video=True, # Match the params.save_video setting
|
||||
)
|
||||
|
||||
# The actual output path where the video was saved
|
||||
output_path = os.path.join(self.output_path, f"{safe_prompt}.mp4")
|
||||
|
||||
# Verify the file exists
|
||||
if not os.path.exists(output_path):
|
||||
raise FileNotFoundError(f"Video was not saved to expected location: {output_path}")
|
||||
|
||||
frames = result.get("frames", [])
|
||||
|
||||
# Encode frames to base64 for web transmission only if requested
|
||||
encoded_frames = None
|
||||
if video_request.return_frames and frames:
|
||||
try:
|
||||
encoded_frames = encode_frames_to_base64(frames)
|
||||
except Exception as e:
|
||||
print(f"Warning: Failed to encode frames: {e}")
|
||||
encoded_frames = None
|
||||
|
||||
response = VideoGenerationResponse(
|
||||
output_path=output_path,
|
||||
frames=encoded_frames,
|
||||
seed=params.seed,
|
||||
success=True
|
||||
)
|
||||
|
||||
# Memory cleanup to avoid OOM in repeated generations
|
||||
import gc
|
||||
gc.collect()
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
return response
|
||||
except Exception as e:
|
||||
return VideoGenerationResponse(
|
||||
video_data=None,
|
||||
output_path="",
|
||||
seed=video_request.seed,
|
||||
success=False,
|
||||
error_message=str(e),
|
||||
generation_time=0,
|
||||
inference_time=0,
|
||||
encoding_time=0,
|
||||
total_time=0,
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
@app.get("/health")
|
||||
@limiter.limit("10/minute")
|
||||
async def health_check(self, request: Request) -> Dict[str, str]:
|
||||
@limiter.limit("10/minute") # Allow 10 health checks per minute per IP
|
||||
async def health_check(self, request: Request):
|
||||
return {"status": "healthy"}
|
||||
|
||||
@app.get("/metrics")
|
||||
async def metrics(self) -> Response:
|
||||
return Response(generate_latest(), media_type="text/plain")
|
||||
|
||||
|
||||
def validate_configuration(model_paths: List[str], replicas: List[int]) -> None:
|
||||
assert len(model_paths) == len(replicas), "Number of models and replicas must match"
|
||||
assert sum(replicas) <= NUM_GPUS, f"Total replicas ({sum(replicas)}) must be <= {NUM_GPUS}"
|
||||
|
||||
for model, replica_count in zip(model_paths, replicas):
|
||||
assert model in SUPPORTED_MODELS, f"Model {model} not supported"
|
||||
assert replica_count > 0, f"Replicas must be greater than 0"
|
||||
|
||||
|
||||
def start_ray_serve(
|
||||
*,
|
||||
t2v_model_paths: str,
|
||||
t2v_model_replicas: str,
|
||||
t2v_model_path: str = "Wan-AI/Wan2.2-TI2V-5B-Diffusers",
|
||||
i2v_model_path: str = "Wan-AI/Wan2.2-TI2V-5B-Diffusers",
|
||||
output_path: str = "outputs",
|
||||
host: str = "0.0.0.0",
|
||||
port: int = 8000,
|
||||
) -> None:
|
||||
port: int = 8000
|
||||
):
|
||||
"""Start the Ray Serve backend"""
|
||||
# Initialize Ray
|
||||
if not ray.is_initialized():
|
||||
ray.init()
|
||||
|
||||
model_paths = t2v_model_paths.split(",")
|
||||
replicas = [int(r) for r in t2v_model_replicas.split(",")]
|
||||
validate_configuration(model_paths, replicas)
|
||||
|
||||
t2v_deps = {}
|
||||
for model_path, replica_count in zip(model_paths, replicas):
|
||||
t2v_dep = T2VModelDeployment.options(num_replicas=replica_count).bind(model_path, output_path)
|
||||
t2v_deps[model_path] = t2v_dep
|
||||
|
||||
api = FastVideoAPI.bind(t2v_deps)
|
||||
serve.run(api, route_prefix="/", name="fast_video")
|
||||
|
||||
|
||||
# Deploy the API
|
||||
api = FastVideoAPI.bind(t2v_model_path, i2v_model_path, output_path)
|
||||
serve.run(api, route_prefix="/", name="fast_video") # detach
|
||||
|
||||
print(f"Ray Serve backend started at http://{host}:{port}")
|
||||
for model_path, replica_count in zip(model_paths, replicas):
|
||||
print(f"T2V Model: {model_path} | Replicas: {replica_count}")
|
||||
print(f"T2V Model: {t2v_model_path}")
|
||||
print(f"I2V Model: {i2v_model_path}")
|
||||
print(f"Health check: http://{host}:{port}/health")
|
||||
print(f"Video generation endpoint: http://{host}:{port}/generate_video")
|
||||
|
||||
|
||||
def setup_signal_handlers() -> None:
|
||||
signal.signal(signal.SIGINT, lambda *_: sys.exit(0))
|
||||
signal.signal(signal.SIGTERM, lambda *_: sys.exit(0))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description="FastVideo Ray Serve Backend")
|
||||
parser.add_argument("--t2v_model_paths",
|
||||
parser.add_argument("--t2v_model_path",
|
||||
type=str,
|
||||
default="FastVideo/FastWan2.1-T2V-1.3B-Diffusers,FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers",
|
||||
help="Comma separated list of paths to the T2V model(s)")
|
||||
parser.add_argument("--t2v_model_replicas",
|
||||
default="FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
help="Path to the T2V model")
|
||||
parser.add_argument("--i2v_model_path",
|
||||
type=str,
|
||||
default="4,4",
|
||||
help="Comma separated list of number of replicas for the T2V model(s)")
|
||||
default="Wan-AI/Wan2.2-TI2V-5B-Diffusers",
|
||||
help="Path to the I2V model")
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default="outputs",
|
||||
@@ -360,20 +318,20 @@ if __name__ == "__main__":
|
||||
help="Port to bind to")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
model_paths = args.t2v_model_paths.split(",")
|
||||
replicas = [int(r) for r in args.t2v_model_replicas.split(",")]
|
||||
validate_configuration(model_paths, replicas)
|
||||
|
||||
start_ray_serve(
|
||||
t2v_model_paths=args.t2v_model_paths,
|
||||
t2v_model_replicas=args.t2v_model_replicas,
|
||||
t2v_model_path=args.t2v_model_path,
|
||||
i2v_model_path=args.i2v_model_path,
|
||||
output_path=args.output_path,
|
||||
host=args.host,
|
||||
port=args.port,
|
||||
)
|
||||
|
||||
setup_signal_handlers()
|
||||
# ---- keep the process alive ---------------------------------
|
||||
import signal, sys, time
|
||||
signal.signal(signal.SIGINT, lambda *_: sys.exit(0)) # Ctrl-C
|
||||
signal.signal(signal.SIGTERM, lambda *_: sys.exit(0)) # docker stop etc.
|
||||
|
||||
print("✅ FastVideo backend is running. Press Ctrl-C to stop.")
|
||||
while True:
|
||||
time.sleep(3600)
|
||||
@@ -0,0 +1,334 @@
|
||||
import time
|
||||
import os
|
||||
import torch
|
||||
import base64
|
||||
import io
|
||||
from copy import deepcopy
|
||||
from typing import Dict, Any, Optional, List
|
||||
|
||||
import ray
|
||||
from ray import serve
|
||||
from fastapi import FastAPI, Request
|
||||
from pydantic import BaseModel
|
||||
from PIL import Image
|
||||
import numpy as np
|
||||
from slowapi import Limiter, _rate_limit_exceeded_handler
|
||||
from slowapi.util import get_remote_address
|
||||
from slowapi.errors import RateLimitExceeded
|
||||
|
||||
|
||||
class VideoGenerationRequest(BaseModel):
|
||||
prompt: str
|
||||
negative_prompt: Optional[str] = None
|
||||
use_negative_prompt: bool = False
|
||||
seed: int = 42
|
||||
guidance_scale: float = 7.5
|
||||
num_frames: int = 21
|
||||
height: int = 448
|
||||
width: int = 832
|
||||
num_inference_steps: int = 20
|
||||
randomize_seed: bool = False
|
||||
return_frames: bool = False # Whether to return base64 encoded frames
|
||||
image_path: Optional[str] = None # Path to input image for I2V
|
||||
model_type: str = "t2v" # "t2v" or "i2v" to specify which model to use
|
||||
|
||||
|
||||
class VideoGenerationResponse(BaseModel):
|
||||
output_path: str
|
||||
seed: int
|
||||
success: bool
|
||||
error_message: Optional[str] = None
|
||||
frames: Optional[List[str]] = None # Base64 encoded frames
|
||||
|
||||
|
||||
def encode_frames_to_base64(frames: List[np.ndarray]) -> List[str]:
|
||||
"""Convert numpy frames (0-255) to base64-encoded PNG images"""
|
||||
if not frames:
|
||||
return []
|
||||
|
||||
encoded_frames = []
|
||||
|
||||
for i, frame in enumerate(frames):
|
||||
try:
|
||||
# Ensure frame is numpy array
|
||||
if not isinstance(frame, np.ndarray):
|
||||
print(f"Warning: Frame {i} is not a numpy array, skipping")
|
||||
continue
|
||||
|
||||
# Ensure frame is uint8
|
||||
if frame.dtype != np.uint8:
|
||||
# Clip values to 0-255 range and convert to uint8
|
||||
frame = np.clip(frame, 0, 255).astype(np.uint8)
|
||||
|
||||
# Convert numpy array to PIL Image
|
||||
if len(frame.shape) == 3 and frame.shape[2] == 3:
|
||||
# RGB image
|
||||
pil_image = Image.fromarray(frame, mode='RGB')
|
||||
elif len(frame.shape) == 3 and frame.shape[2] == 4:
|
||||
# RGBA image
|
||||
pil_image = Image.fromarray(frame, mode='RGBA')
|
||||
elif len(frame.shape) == 2:
|
||||
# Grayscale image
|
||||
pil_image = Image.fromarray(frame, mode='L')
|
||||
else:
|
||||
print(f"Warning: Frame {i} has unsupported shape {frame.shape}, skipping")
|
||||
continue
|
||||
|
||||
# Save to bytes buffer as PNG
|
||||
buffer = io.BytesIO()
|
||||
pil_image.save(buffer, format='PNG')
|
||||
buffer.seek(0)
|
||||
|
||||
# Encode to base64
|
||||
img_base64 = base64.b64encode(buffer.getvalue()).decode('utf-8')
|
||||
encoded_frames.append(f"data:image/png;base64,{img_base64}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Warning: Failed to encode frame {i}: {e}")
|
||||
continue
|
||||
|
||||
return encoded_frames
|
||||
|
||||
|
||||
# Create FastAPI app with rate limiting
|
||||
app = FastAPI()
|
||||
|
||||
# Initialize rate limiter
|
||||
limiter = Limiter(key_func=get_remote_address)
|
||||
app.state.limiter = limiter
|
||||
app.add_exception_handler(RateLimitExceeded, _rate_limit_exceeded_handler)
|
||||
|
||||
|
||||
@serve.deployment(
|
||||
num_replicas=3, # Set to 3 for cluster 0
|
||||
ray_actor_options={
|
||||
"num_cpus": 10,
|
||||
"num_gpus": 1,
|
||||
"runtime_env": {"conda": "fv"},
|
||||
},
|
||||
)
|
||||
@serve.ingress(app)
|
||||
class FastVideoMultiGPUAPI:
|
||||
def __init__(self, t2v_model_path: str, i2v_model_path: str, output_path: str, gpu_id: int = 0):
|
||||
self.t2v_model_path = t2v_model_path
|
||||
self.i2v_model_path = i2v_model_path
|
||||
self.output_path = output_path
|
||||
self.gpu_id = gpu_id
|
||||
|
||||
# Initialize the video generators
|
||||
self.t2v_generator = None
|
||||
self.i2v_generator = None
|
||||
self.t2v_default_params = None
|
||||
self.i2v_default_params = None
|
||||
|
||||
# Ensure output directory exists
|
||||
os.makedirs(output_path, exist_ok=True)
|
||||
time.sleep(10)
|
||||
self._initialize_models()
|
||||
|
||||
def _initialize_models(self):
|
||||
# Set VSA environment variable
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "FLASH_ATTN"
|
||||
|
||||
# Import only when needed
|
||||
from fastvideo.entrypoints.video_generator import VideoGenerator
|
||||
from fastvideo.configs.sample.base import SamplingParam
|
||||
|
||||
# Initialize T2V model
|
||||
if False: # Disabled for now
|
||||
print(f"Initializing T2V model on GPU {self.gpu_id}: {self.t2v_model_path}")
|
||||
self.t2v_generator = VideoGenerator.from_pretrained(
|
||||
model_path=self.t2v_model_path,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
text_encoder_cpu_offload=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
VSA_sparsity=0.8,
|
||||
)
|
||||
self.t2v_default_params = SamplingParam.from_pretrained(self.t2v_model_path)
|
||||
print(f"✅ T2V model initialized successfully on GPU {self.gpu_id}")
|
||||
|
||||
# Initialize I2V model
|
||||
if self.i2v_generator is None:
|
||||
print(f"Initializing I2V model on GPU {self.gpu_id}: {self.i2v_model_path}")
|
||||
self.i2v_generator = VideoGenerator.from_pretrained(
|
||||
model_path=self.i2v_model_path,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
text_encoder_cpu_offload=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
VSA_sparsity=0.8,
|
||||
)
|
||||
self.i2v_default_params = SamplingParam.from_pretrained(self.i2v_model_path)
|
||||
print(f"✅ I2V model initialized successfully on GPU {self.gpu_id}")
|
||||
|
||||
@app.post("/generate_video", response_model=VideoGenerationResponse)
|
||||
@limiter.limit("2/minute") # Allow 2 requests per minute per IP
|
||||
async def generate_video(self, request: Request, video_request: VideoGenerationRequest) -> VideoGenerationResponse:
|
||||
try:
|
||||
# Select the appropriate model and parameters based on model_type
|
||||
if video_request.model_type.lower() == "i2v":
|
||||
generator = self.i2v_generator
|
||||
params = deepcopy(self.i2v_default_params)
|
||||
print(f"Using I2V model for generation on GPU {self.gpu_id}")
|
||||
else:
|
||||
generator = self.t2v_generator
|
||||
params = deepcopy(self.t2v_default_params)
|
||||
print(f"Using T2V model for generation on GPU {self.gpu_id}")
|
||||
|
||||
# Update parameters with request values
|
||||
params.prompt = video_request.prompt
|
||||
|
||||
# Handle seed randomization
|
||||
if video_request.randomize_seed:
|
||||
params.seed = torch.randint(0, 1000000, (1,)).item()
|
||||
|
||||
# Ensure negative_prompt is a non-None string
|
||||
if params.negative_prompt is None:
|
||||
params.negative_prompt = ""
|
||||
|
||||
# Set up output path and video saving
|
||||
params.save_video = True
|
||||
params.output_path = self.output_path
|
||||
|
||||
# Create a clean filename from the prompt
|
||||
safe_prompt = video_request.prompt[:100].replace(' ', '_').replace('/', '_').replace('\\', '_')
|
||||
setattr(params, "output_video_name", safe_prompt)
|
||||
|
||||
# Handle image_path for I2V
|
||||
if video_request.image_path:
|
||||
params.image_path = video_request.image_path
|
||||
|
||||
# Generate the video
|
||||
result = generator.generate_video(
|
||||
prompt=video_request.prompt,
|
||||
sampling_param=params,
|
||||
save_video=True,
|
||||
)
|
||||
|
||||
frames = result.get("frames", [])
|
||||
|
||||
# Encode frames to base64 for web transmission only if requested
|
||||
encoded_frames = None
|
||||
if video_request.return_frames and frames:
|
||||
try:
|
||||
encoded_frames = encode_frames_to_base64(frames)
|
||||
except Exception as e:
|
||||
print(f"Warning: Failed to encode frames: {e}")
|
||||
encoded_frames = None
|
||||
|
||||
response = VideoGenerationResponse(
|
||||
output_path="",
|
||||
frames=encoded_frames,
|
||||
seed=params.seed,
|
||||
success=True
|
||||
)
|
||||
|
||||
# Memory cleanup to avoid OOM in repeated generations
|
||||
import gc
|
||||
gc.collect()
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
return response
|
||||
except Exception as e:
|
||||
return VideoGenerationResponse(
|
||||
output_path="",
|
||||
seed=video_request.seed,
|
||||
success=False,
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
@app.get("/health")
|
||||
@limiter.limit("10/minute") # Allow 10 health checks per minute per IP
|
||||
async def health_check(self, request: Request):
|
||||
return {"status": "healthy", "gpu_id": self.gpu_id}
|
||||
|
||||
|
||||
def start_ray_serve_multi_gpu(
|
||||
t2v_model_path: str = "FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
i2v_model_path: str = "Wan-AI/Wan2.1-I2V-14B-480P-Diffusers",
|
||||
output_path: str = "outputs",
|
||||
host: str = "0.0.0.0",
|
||||
port: int = 8000,
|
||||
num_gpus: int = 8,
|
||||
cluster_id: int = 0
|
||||
):
|
||||
"""Start the Ray Serve backend with multiple GPU replicas"""
|
||||
# Initialize Ray
|
||||
if not ray.is_initialized():
|
||||
ray.init()
|
||||
|
||||
# Use unique application name based on cluster_id
|
||||
app_name = f"fast_video_cluster_{cluster_id}"
|
||||
|
||||
# Deploy the API
|
||||
api = FastVideoMultiGPUAPI.bind(t2v_model_path, i2v_model_path, output_path)
|
||||
serve.run(api, route_prefix=f"/cluster_{cluster_id}", name=app_name)
|
||||
|
||||
print(f"Ray Serve multi-GPU backend started at http://{host}:{port}")
|
||||
print(f"T2V Model: {t2v_model_path}")
|
||||
print(f"I2V Model: {i2v_model_path}")
|
||||
print(f"Number of GPU replicas: {num_gpus}")
|
||||
print(f"Cluster ID: {cluster_id}")
|
||||
print(f"Application name: {app_name}")
|
||||
print(f"Health check: http://{host}:{port}/cluster_{cluster_id}/health")
|
||||
print(f"Video generation endpoint: http://{host}:{port}/cluster_{cluster_id}/generate_video")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description="FastVideo Ray Serve Multi-GPU Backend")
|
||||
parser.add_argument("--t2v_model_path",
|
||||
type=str,
|
||||
default="FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
help="Path to the T2V model")
|
||||
parser.add_argument("--i2v_model_path",
|
||||
type=str,
|
||||
default="Wan-AI/Wan2.1-I2V-14B-480P-Diffusers",
|
||||
help="Path to the I2V model")
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default="outputs",
|
||||
help="Path to save generated videos")
|
||||
parser.add_argument("--host",
|
||||
type=str,
|
||||
default="0.0.0.0",
|
||||
help="Host to bind to")
|
||||
parser.add_argument("--port",
|
||||
type=int,
|
||||
default=8000,
|
||||
help="Port to bind to")
|
||||
parser.add_argument("--num_gpus",
|
||||
type=int,
|
||||
default=8,
|
||||
help="Number of GPU replicas")
|
||||
parser.add_argument("--cluster_id",
|
||||
type=int,
|
||||
default=0,
|
||||
help="Cluster ID for unique naming")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
start_ray_serve_multi_gpu(
|
||||
t2v_model_path=args.t2v_model_path,
|
||||
i2v_model_path=args.i2v_model_path,
|
||||
output_path=args.output_path,
|
||||
host=args.host,
|
||||
port=args.port,
|
||||
num_gpus=args.num_gpus,
|
||||
cluster_id=args.cluster_id,
|
||||
)
|
||||
|
||||
# Keep the process alive
|
||||
import signal, sys, time
|
||||
signal.signal(signal.SIGINT, lambda *_: sys.exit(0))
|
||||
signal.signal(signal.SIGTERM, lambda *_: sys.exit(0))
|
||||
|
||||
print("✅ FastVideo multi-GPU backend is running. Press Ctrl-C to stop.")
|
||||
while True:
|
||||
time.sleep(3600)
|
||||
@@ -1,3 +0,0 @@
|
||||
python examples/inference/gradio/start_ray_serve_app.py \
|
||||
--t2v_model_paths "FastVideo/FastWan2.1-T2V-1.3B-Diffusers,FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers" \
|
||||
--t2v_model_replicas "4,4"
|
||||
@@ -12,245 +12,220 @@ import threading
|
||||
import signal
|
||||
import requests
|
||||
from pathlib import Path
|
||||
from typing import Dict, Any, Optional
|
||||
|
||||
# Add the project root to the Python path
|
||||
project_root = Path(__file__).parent.parent.parent.parent
|
||||
sys.path.insert(0, str(project_root))
|
||||
|
||||
|
||||
DEFAULT_BACKEND_HOST = "0.0.0.0"
|
||||
DEFAULT_BACKEND_PORT = 8000
|
||||
DEFAULT_FRONTEND_HOST = "0.0.0.0"
|
||||
DEFAULT_FRONTEND_PORT = 7860
|
||||
DEFAULT_OUTPUT_PATH = "outputs"
|
||||
DEFAULT_T2V_MODELS = "FastVideo/FastWan2.1-T2V-1.3B-Diffusers,FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers"
|
||||
DEFAULT_T2V_REPLICAS = "4,4"
|
||||
|
||||
HEALTH_CHECK_TIMEOUT = 5
|
||||
HEALTH_CHECK_MAX_RETRIES = 100
|
||||
HEALTH_CHECK_INTERVAL = 2
|
||||
PROCESS_SHUTDOWN_TIMEOUT = 5
|
||||
PROCESS_MONITOR_INTERVAL = 1
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent.parent.parent
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
|
||||
|
||||
class ServiceManager:
|
||||
def check_backend_health(backend_url: str, max_retries: int = 100) -> bool:
|
||||
"""Check if the backend is healthy"""
|
||||
health_url = f"{backend_url}/health"
|
||||
|
||||
def __init__(self, args: argparse.Namespace):
|
||||
self.args = args
|
||||
self.backend_process: Optional[subprocess.Popen] = None
|
||||
self.frontend_process: Optional[subprocess.Popen] = None
|
||||
self.backend_url = f"http://{args.backend_host}:{args.backend_port}"
|
||||
|
||||
def check_backend_health(self, max_retries: int = HEALTH_CHECK_MAX_RETRIES) -> bool:
|
||||
health_url = f"{self.backend_url}/health"
|
||||
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
response = requests.get(health_url, timeout=HEALTH_CHECK_TIMEOUT)
|
||||
if response.status_code == 200:
|
||||
print(f"✅ Backend is healthy at {self.backend_url}")
|
||||
return True
|
||||
except requests.exceptions.RequestException:
|
||||
pass
|
||||
|
||||
if attempt < max_retries - 1:
|
||||
print(f"⏳ Waiting for backend to start... ({attempt + 1}/{max_retries})")
|
||||
time.sleep(HEALTH_CHECK_INTERVAL)
|
||||
|
||||
print(f"❌ Backend failed to start within {max_retries * HEALTH_CHECK_INTERVAL} seconds")
|
||||
return False
|
||||
|
||||
def _create_monitor_thread(self, process: subprocess.Popen, service_name: str) -> threading.Thread:
|
||||
def monitor():
|
||||
if process.stdout:
|
||||
for line in process.stdout:
|
||||
print(f"[{service_name}] {line.rstrip()}")
|
||||
|
||||
thread = threading.Thread(target=monitor, daemon=True)
|
||||
thread.start()
|
||||
return thread
|
||||
|
||||
def _start_service(self, script_name: str, args_dict: Dict[str, Any], service_name: str) -> subprocess.Popen:
|
||||
script_path = Path(__file__).parent / script_name
|
||||
|
||||
cmd = [sys.executable, str(script_path)]
|
||||
for key, value in args_dict.items():
|
||||
cmd.extend([f"--{key}", str(value)])
|
||||
|
||||
print(f"🚀 Starting {service_name}...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
self._create_monitor_thread(process, service_name.upper())
|
||||
return process
|
||||
|
||||
def start_backend(self) -> subprocess.Popen:
|
||||
backend_args = {
|
||||
"t2v_model_paths": self.args.t2v_model_paths,
|
||||
"t2v_model_replicas": self.args.t2v_model_replicas,
|
||||
"output_path": self.args.output_path,
|
||||
"host": self.args.backend_host,
|
||||
"port": self.args.backend_port
|
||||
}
|
||||
|
||||
self.backend_process = self._start_service("ray_serve_backend.py", backend_args, "backend")
|
||||
return self.backend_process
|
||||
|
||||
def start_frontend(self) -> subprocess.Popen:
|
||||
frontend_args = {
|
||||
"backend_url": self.backend_url,
|
||||
"t2v_model_paths": self.args.t2v_model_paths,
|
||||
"host": self.args.frontend_host,
|
||||
"port": self.args.frontend_port
|
||||
}
|
||||
|
||||
self.frontend_process = self._start_service("gradio_frontend.py", frontend_args, "frontend")
|
||||
return self.frontend_process
|
||||
|
||||
def shutdown_services(self) -> None:
|
||||
print("\n🛑 Shutting down services...")
|
||||
|
||||
processes = []
|
||||
if self.frontend_process:
|
||||
self.frontend_process.terminate()
|
||||
processes.append(("frontend", self.frontend_process))
|
||||
|
||||
if self.backend_process:
|
||||
self.backend_process.terminate()
|
||||
processes.append(("backend", self.backend_process))
|
||||
|
||||
for name, process in processes:
|
||||
try:
|
||||
process.wait(timeout=PROCESS_SHUTDOWN_TIMEOUT)
|
||||
print(f"✅ {name.capitalize()} stopped gracefully")
|
||||
except subprocess.TimeoutExpired:
|
||||
print(f"⚠️ Force killing {name} process...")
|
||||
process.kill()
|
||||
|
||||
print("✅ All services stopped")
|
||||
|
||||
def monitor_processes(self) -> None:
|
||||
if not self.backend_process or not self.frontend_process:
|
||||
print("❌ Processes not properly initialized")
|
||||
return
|
||||
|
||||
for i in range(max_retries):
|
||||
try:
|
||||
while True:
|
||||
if self.frontend_process.poll() is not None:
|
||||
print("❌ Frontend process died unexpectedly")
|
||||
break
|
||||
|
||||
if self.backend_process.poll() is not None:
|
||||
print("❌ Backend process died unexpectedly")
|
||||
break
|
||||
|
||||
time.sleep(PROCESS_MONITOR_INTERVAL)
|
||||
|
||||
except KeyboardInterrupt:
|
||||
response = requests.get(health_url, timeout=5)
|
||||
if response.status_code == 200:
|
||||
print(f"✅ Backend is healthy at {backend_url}")
|
||||
return True
|
||||
except requests.exceptions.RequestException:
|
||||
pass
|
||||
|
||||
self.shutdown_services()
|
||||
|
||||
|
||||
def setup_signal_handlers(service_manager: ServiceManager) -> None:
|
||||
def signal_handler(signum: int, frame: Any) -> None:
|
||||
service_manager.shutdown_services()
|
||||
sys.exit(0)
|
||||
if i < max_retries - 1:
|
||||
print(f"⏳ Waiting for backend to start... ({i+1}/{max_retries})")
|
||||
time.sleep(2)
|
||||
|
||||
signal.signal(signal.SIGINT, signal_handler)
|
||||
signal.signal(signal.SIGTERM, signal_handler)
|
||||
print(f"❌ Backend failed to start within {max_retries * 2} seconds")
|
||||
return False
|
||||
|
||||
|
||||
def print_startup_info(args: argparse.Namespace) -> None:
|
||||
def start_backend(args):
|
||||
"""Start the Ray Serve backend"""
|
||||
backend_script = Path(__file__).parent / "ray_serve_backend.py"
|
||||
|
||||
cmd = [
|
||||
sys.executable, str(backend_script),
|
||||
"--t2v_model_path", args.t2v_model_path,
|
||||
"--i2v_model_path", args.i2v_model_path,
|
||||
"--output_path", args.output_path,
|
||||
"--host", args.backend_host,
|
||||
"--port", str(args.backend_port)
|
||||
]
|
||||
|
||||
print(f"🚀 Starting Ray Serve backend...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
# Start the backend process
|
||||
backend_process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
# Monitor backend output
|
||||
def monitor_backend():
|
||||
for line in backend_process.stdout:
|
||||
print(f"[BACKEND] {line.rstrip()}")
|
||||
|
||||
monitor_thread = threading.Thread(target=monitor_backend, daemon=True)
|
||||
monitor_thread.start()
|
||||
|
||||
return backend_process
|
||||
|
||||
|
||||
def start_frontend(args):
|
||||
"""Start the Gradio frontend"""
|
||||
frontend_script = Path(__file__).parent / "gradio_frontend.py"
|
||||
backend_url = f"http://{args.backend_host}:{args.backend_port}"
|
||||
|
||||
cmd = [
|
||||
sys.executable, str(frontend_script),
|
||||
"--backend_url", backend_url,
|
||||
"--t2v_model_path", args.t2v_model_path,
|
||||
"--i2v_model_path", args.i2v_model_path,
|
||||
"--host", args.frontend_host,
|
||||
"--port", str(args.frontend_port)
|
||||
]
|
||||
|
||||
print(f"🎨 Starting Gradio frontend...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
# Start the frontend process
|
||||
frontend_process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
# Monitor frontend output
|
||||
def monitor_frontend():
|
||||
for line in frontend_process.stdout:
|
||||
print(f"[FRONTEND] {line.rstrip()}")
|
||||
|
||||
monitor_thread = threading.Thread(target=monitor_frontend, daemon=True)
|
||||
monitor_thread.start()
|
||||
|
||||
return frontend_process
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="FastVideo Ray Serve App")
|
||||
|
||||
# Model and output settings
|
||||
parser.add_argument("--t2v_model_path",
|
||||
type=str,
|
||||
default="Wan-AI/Wan2.2-TI2V-5BDiffusers",
|
||||
help="Path to the T2V model")
|
||||
parser.add_argument("--i2v_model_path",
|
||||
type=str,
|
||||
default="Wan-AI/Wan2.2-TI2V-5B-Diffusers",
|
||||
help="Path to the I2V model")
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default="outputs",
|
||||
help="Path to save generated videos")
|
||||
|
||||
# Backend settings
|
||||
parser.add_argument("--backend_host",
|
||||
type=str,
|
||||
default="0.0.0.0",
|
||||
help="Backend host to bind to")
|
||||
parser.add_argument("--backend_port",
|
||||
type=int,
|
||||
default=8000,
|
||||
help="Backend port to bind to")
|
||||
|
||||
# Frontend settings
|
||||
parser.add_argument("--frontend_host",
|
||||
type=str,
|
||||
default="0.0.0.0",
|
||||
help="Frontend host to bind to")
|
||||
parser.add_argument("--frontend_port",
|
||||
type=int,
|
||||
default=7860,
|
||||
help="Frontend port to bind to")
|
||||
|
||||
# Other settings
|
||||
parser.add_argument("--skip_backend_check",
|
||||
action="store_true",
|
||||
help="Skip backend health check")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Ensure output directory exists
|
||||
os.makedirs(args.output_path, exist_ok=True)
|
||||
|
||||
print("🎬 FastVideo Ray Serve App")
|
||||
print("=" * 50)
|
||||
print(f"T2V Models: {args.t2v_model_paths}")
|
||||
print(f"T2V Model Replicas: {args.t2v_model_replicas}")
|
||||
print(f"T2V Model: {args.t2v_model_path}")
|
||||
print(f"I2V Model: {args.i2v_model_path}")
|
||||
print(f"Output: {args.output_path}")
|
||||
print(f"Backend: http://{args.backend_host}:{args.backend_port}")
|
||||
print(f"Frontend: http://{args.frontend_host}:{args.frontend_port}")
|
||||
print("=" * 50)
|
||||
|
||||
|
||||
def parse_arguments() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="FastVideo Ray Serve App")
|
||||
|
||||
parser.add_argument("--t2v_model_paths",
|
||||
type=str,
|
||||
default=DEFAULT_T2V_MODELS,
|
||||
help="Comma separated list of paths to the T2V model(s)")
|
||||
parser.add_argument("--t2v_model_replicas",
|
||||
type=str,
|
||||
default=DEFAULT_T2V_REPLICAS,
|
||||
help="Comma separated list of number of replicas for the T2V model(s)")
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default=DEFAULT_OUTPUT_PATH,
|
||||
help="Path to save generated videos")
|
||||
# Start backend
|
||||
backend_process = start_backend(args)
|
||||
|
||||
parser.add_argument("--backend_host",
|
||||
type=str,
|
||||
default=DEFAULT_BACKEND_HOST,
|
||||
help="Backend host to bind to")
|
||||
parser.add_argument("--backend_port",
|
||||
type=int,
|
||||
default=DEFAULT_BACKEND_PORT,
|
||||
help="Backend port to bind to")
|
||||
# Wait for backend to be ready
|
||||
backend_url = f"http://{args.backend_host}:{args.backend_port}"
|
||||
|
||||
parser.add_argument("--frontend_host",
|
||||
type=str,
|
||||
default=DEFAULT_FRONTEND_HOST,
|
||||
help="Frontend host to bind to")
|
||||
parser.add_argument("--frontend_port",
|
||||
type=int,
|
||||
default=DEFAULT_FRONTEND_PORT,
|
||||
help="Frontend port to bind to")
|
||||
if not args.skip_backend_check:
|
||||
if not check_backend_health(backend_url):
|
||||
print("❌ Backend failed to start. Terminating...")
|
||||
backend_process.terminate()
|
||||
sys.exit(1)
|
||||
|
||||
parser.add_argument("--skip_backend_check",
|
||||
action="store_true",
|
||||
help="Skip backend health check")
|
||||
# Start frontend
|
||||
frontend_process = start_frontend(args)
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_arguments()
|
||||
print("\n🎉 Both services are starting up!")
|
||||
print(f"📺 Frontend will be available at: http://{args.frontend_host}:{args.frontend_port}")
|
||||
print(f"🔧 Backend API will be available at: {backend_url}")
|
||||
print("\nPress Ctrl+C to stop both services...")
|
||||
# return
|
||||
|
||||
os.makedirs(args.output_path, exist_ok=True)
|
||||
print_startup_info(args)
|
||||
# Signal handler for graceful shutdown
|
||||
def signal_handler(signum, frame):
|
||||
print("\n🛑 Shutting down services...")
|
||||
frontend_process.terminate()
|
||||
backend_process.terminate()
|
||||
|
||||
# Wait for processes to terminate
|
||||
try:
|
||||
frontend_process.wait(timeout=5)
|
||||
backend_process.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
print("⚠️ Force killing processes...")
|
||||
frontend_process.kill()
|
||||
backend_process.kill()
|
||||
|
||||
print("✅ Services stopped")
|
||||
sys.exit(0)
|
||||
|
||||
service_manager = ServiceManager(args)
|
||||
setup_signal_handlers(service_manager)
|
||||
signal.signal(signal.SIGINT, signal_handler)
|
||||
signal.signal(signal.SIGTERM, signal_handler)
|
||||
|
||||
# Monitor processes
|
||||
try:
|
||||
service_manager.start_backend()
|
||||
|
||||
if not args.skip_backend_check:
|
||||
if not service_manager.check_backend_health():
|
||||
print("❌ Backend failed to start. Terminating...")
|
||||
service_manager.shutdown_services()
|
||||
sys.exit(1)
|
||||
|
||||
service_manager.start_frontend()
|
||||
|
||||
print("\n🎉 Both services are starting up!")
|
||||
print(f"📺 Frontend will be available at: http://{args.frontend_host}:{args.frontend_port}")
|
||||
print(f"🔧 Backend API will be available at: http://{args.backend_host}:{args.backend_port}")
|
||||
print("\nPress Ctrl+C to stop both services...")
|
||||
|
||||
service_manager.monitor_processes()
|
||||
while True:
|
||||
# Check if processes are still running
|
||||
if frontend_process.poll() is not None:
|
||||
print("❌ Frontend process died unexpectedly")
|
||||
break
|
||||
|
||||
if backend_process.poll() is not None:
|
||||
print("❌ Backend process died unexpectedly")
|
||||
break
|
||||
|
||||
time.sleep(1)
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Unexpected error: {e}")
|
||||
service_manager.shutdown_services()
|
||||
sys.exit(1)
|
||||
except KeyboardInterrupt:
|
||||
signal_handler(signal.SIGINT, None)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -0,0 +1,465 @@
|
||||
"""
|
||||
Startup script for FastVideo scalable architecture:
|
||||
ngrok -> nginx reverse proxy -> frontend1/frontend2 -> backend1×8/backend2×8
|
||||
|
||||
This script starts:
|
||||
1. Multiple backend instances (8 GPU replicas each)
|
||||
2. Multiple frontend instances (2 instances)
|
||||
3. Nginx reverse proxy
|
||||
4. Optional ngrok tunnel
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import threading
|
||||
import signal
|
||||
import requests
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
# Add the project root to the Python path
|
||||
project_root = Path(__file__).parent.parent.parent.parent
|
||||
sys.path.insert(0, str(project_root))
|
||||
|
||||
|
||||
def check_service_health(url: str, max_retries: int = 50) -> bool:
|
||||
"""Check if a service is healthy"""
|
||||
for i in range(max_retries):
|
||||
try:
|
||||
response = requests.get(url, timeout=5)
|
||||
if response.status_code == 200:
|
||||
print(f"✅ Service is healthy at {url}")
|
||||
return True
|
||||
except requests.exceptions.RequestException:
|
||||
pass
|
||||
|
||||
if i < max_retries - 1:
|
||||
print(f"⏳ Waiting for service to start... ({i+1}/{max_retries})")
|
||||
time.sleep(2)
|
||||
|
||||
print(f"❌ Service failed to start within {max_retries * 2} seconds")
|
||||
return False
|
||||
|
||||
|
||||
def start_backend_cluster(args, cluster_id: int):
|
||||
"""Start one backend cluster (Ray-Serve application)."""
|
||||
backend_script = Path(__file__).parent / "ray_serve_backend_scalable.py"
|
||||
|
||||
# All Ray Serve apps share the same HTTP server (default 8000).
|
||||
# We still forward the port flag for completeness, but keep it
|
||||
# identical for every cluster.
|
||||
base_port = args.backend_base_port
|
||||
|
||||
cmd = [
|
||||
sys.executable, str(backend_script),
|
||||
"--t2v_model_path", args.t2v_model_path,
|
||||
"--i2v_model_path", args.i2v_model_path,
|
||||
"--output_path", args.output_path,
|
||||
"--host", args.backend_host,
|
||||
"--port", str(base_port),
|
||||
"--num_gpus", str(args.num_gpus_per_cluster),
|
||||
"--cluster_id", str(cluster_id),
|
||||
]
|
||||
|
||||
print(f"🚀 Starting Backend Cluster {cluster_id + 1} (HTTP port {base_port})...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
# Start the backend process
|
||||
backend_process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
# Monitor backend output
|
||||
def monitor_backend():
|
||||
for line in backend_process.stdout:
|
||||
print(f"[BACKEND-{cluster_id + 1}] {line.rstrip()}")
|
||||
|
||||
monitor_thread = threading.Thread(target=monitor_backend, daemon=True)
|
||||
monitor_thread.start()
|
||||
|
||||
return backend_process, base_port
|
||||
|
||||
|
||||
def start_frontend_instance(args, instance_id: int, backend_url: str):
|
||||
"""Start a single frontend instance"""
|
||||
frontend_script = Path(__file__).parent / "gradio_frontend.py"
|
||||
frontend_port = args.frontend_base_port + instance_id
|
||||
|
||||
# Update backend URL to include cluster-specific path
|
||||
cluster_id = instance_id % args.num_backend_clusters
|
||||
backend_url_with_cluster = f"{backend_url}/cluster_{cluster_id}"
|
||||
|
||||
cmd = [
|
||||
sys.executable, str(frontend_script),
|
||||
"--backend_url", backend_url_with_cluster,
|
||||
"--t2v_model_path", args.t2v_model_path,
|
||||
"--i2v_model_path", args.i2v_model_path,
|
||||
"--host", args.frontend_host,
|
||||
"--port", str(frontend_port)
|
||||
]
|
||||
|
||||
print(f"🎨 Starting Frontend {instance_id + 1} on port {frontend_port}...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
# Start the frontend process
|
||||
frontend_process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
# Monitor frontend output
|
||||
def monitor_frontend():
|
||||
for line in frontend_process.stdout:
|
||||
print(f"[FRONTEND-{instance_id + 1}] {line.rstrip()}")
|
||||
|
||||
monitor_thread = threading.Thread(target=monitor_frontend, daemon=True)
|
||||
monitor_thread.start()
|
||||
|
||||
return frontend_process, frontend_port
|
||||
|
||||
|
||||
def start_nginx(args):
|
||||
"""Start nginx reverse proxy"""
|
||||
nginx_conf = Path(__file__).parent / "nginx.conf"
|
||||
|
||||
# Update nginx configuration with actual ports
|
||||
update_nginx_config(args)
|
||||
|
||||
cmd = [
|
||||
"nginx",
|
||||
"-c", str(nginx_conf),
|
||||
"-g", "daemon off;"
|
||||
]
|
||||
|
||||
print(f"🌐 Starting Nginx reverse proxy...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
# Start nginx process
|
||||
nginx_process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
# Monitor nginx output
|
||||
def monitor_nginx():
|
||||
for line in nginx_process.stdout:
|
||||
print(f"[NGINX] {line.rstrip()}")
|
||||
|
||||
monitor_thread = threading.Thread(target=monitor_nginx, daemon=True)
|
||||
monitor_thread.start()
|
||||
|
||||
return nginx_process
|
||||
|
||||
|
||||
def update_nginx_config(args):
|
||||
"""Rewrite nginx.conf with the correct ports – NO “/cluster_X” in upstreams."""
|
||||
nginx_conf = Path(__file__).parent / "nginx.conf"
|
||||
nginx_conf_backup = Path(__file__).parent / "nginx.conf.backup"
|
||||
|
||||
if not nginx_conf_backup.exists():
|
||||
nginx_conf_backup.write_text(nginx_conf.read_text())
|
||||
|
||||
config_content = nginx_conf_backup.read_text()
|
||||
|
||||
# ── 1. front-end pool ───────────────────────────────────────────────
|
||||
frontend_servers = "\n ".join(
|
||||
f"server 127.0.0.1:{args.frontend_base_port + i} "
|
||||
f"weight=1 max_fails=3 fail_timeout=30s;"
|
||||
for i in range(args.num_frontends)
|
||||
)
|
||||
config_content = config_content.replace(
|
||||
"# Upstream for frontend load balancing",
|
||||
f"# Upstream for frontend load balancing\n upstream frontend_servers {{\n"
|
||||
f" # Round-robin load balancing between frontends\n {frontend_servers}"
|
||||
)
|
||||
|
||||
# Shared Ray-Serve HTTP port
|
||||
backend_port = args.backend_base_port # default 8000
|
||||
backend_line = (f"server 127.0.0.1:{backend_port} "
|
||||
f"weight=1 max_fails=3 fail_timeout=30s;")
|
||||
|
||||
# ── 2. backend-1 pool ───────────────────────────────────────────────
|
||||
config_content = config_content.replace(
|
||||
"# Upstream for backend1 load balancing",
|
||||
f"# Upstream for backend1 load balancing\n upstream backend1_servers {{\n"
|
||||
f" {backend_line}"
|
||||
)
|
||||
|
||||
# ── 3. backend-2 pool ───────────────────────────────────────────────
|
||||
config_content = config_content.replace(
|
||||
"# Upstream for backend2 load balancing",
|
||||
f"# Upstream for backend2 load balancing\n upstream backend2_servers {{\n"
|
||||
f" {backend_line}"
|
||||
)
|
||||
|
||||
# ── 4. strip any stray “/cluster_X” fragments ───────────────────────
|
||||
config_content = config_content.replace("/cluster_0", "").replace("/cluster_1", "")
|
||||
|
||||
# ── 5. use user-writable log directory ---------------------------------
|
||||
log_dir = Path(args.output_path).resolve()
|
||||
config_content = config_content.replace(
|
||||
"access_log /var/log/nginx/access.log;",
|
||||
f"access_log {log_dir}/nginx_access.log;")
|
||||
config_content = config_content.replace(
|
||||
"error_log /var/log/nginx/error.log;",
|
||||
f"error_log {log_dir}/nginx_error.log;")
|
||||
|
||||
nginx_conf.write_text(config_content)
|
||||
print("✅ nginx.conf updated (no path suffixes & custom log paths)")
|
||||
|
||||
|
||||
def start_ngrok(args):
|
||||
"""Start ngrok tunnel"""
|
||||
if not args.use_ngrok:
|
||||
return None
|
||||
|
||||
cmd = [
|
||||
"ngrok",
|
||||
"http",
|
||||
str(args.nginx_port),
|
||||
"--log=stdout"
|
||||
]
|
||||
|
||||
print(f"🌍 Starting ngrok tunnel to port {args.nginx_port}...")
|
||||
print(f"Command: {' '.join(cmd)}")
|
||||
|
||||
# Start ngrok process
|
||||
ngrok_process = subprocess.Popen(
|
||||
cmd,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
# Monitor ngrok output
|
||||
def monitor_ngrok():
|
||||
for line in ngrok_process.stdout:
|
||||
print(f"[NGROK] {line.rstrip()}")
|
||||
|
||||
monitor_thread = threading.Thread(target=monitor_ngrok, daemon=True)
|
||||
monitor_thread.start()
|
||||
|
||||
return ngrok_process
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="FastVideo Scalable Architecture Launcher")
|
||||
|
||||
# Model and output settings
|
||||
parser.add_argument("--t2v_model_path",
|
||||
type=str,
|
||||
default="FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
help="Path to the T2V model")
|
||||
parser.add_argument("--i2v_model_path",
|
||||
type=str,
|
||||
default="Wan-AI/Wan2.1-I2V-14B-480P-Diffusers",
|
||||
help="Path to the I2V model")
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default="outputs",
|
||||
help="Path to save generated videos")
|
||||
|
||||
# Backend settings
|
||||
parser.add_argument("--backend_host",
|
||||
type=str,
|
||||
default="0.0.0.0",
|
||||
help="Backend host to bind to")
|
||||
parser.add_argument("--backend_base_port",
|
||||
type=int,
|
||||
default=8000,
|
||||
help="Base port for backend clusters")
|
||||
parser.add_argument("--num_backend_clusters",
|
||||
type=int,
|
||||
default=2,
|
||||
help="Number of backend clusters")
|
||||
parser.add_argument("--num_gpus_per_cluster",
|
||||
type=int,
|
||||
default=3, # Changed from 8 to 3 (3+3=6 GPUs total, leaving 1 GPU buffer)
|
||||
help="Number of GPUs per backend cluster")
|
||||
|
||||
# Frontend settings
|
||||
parser.add_argument("--frontend_host",
|
||||
type=str,
|
||||
default="0.0.0.0",
|
||||
help="Frontend host to bind to")
|
||||
parser.add_argument("--frontend_base_port",
|
||||
type=int,
|
||||
default=7860,
|
||||
help="Base port for frontend instances")
|
||||
parser.add_argument("--num_frontends",
|
||||
type=int,
|
||||
default=2,
|
||||
help="Number of frontend instances")
|
||||
|
||||
# Nginx settings
|
||||
parser.add_argument("--nginx_port",
|
||||
type=int,
|
||||
default=80,
|
||||
help="Port for nginx reverse proxy")
|
||||
|
||||
# Ngrok settings
|
||||
parser.add_argument("--use_ngrok",
|
||||
action="store_true",
|
||||
help="Start ngrok tunnel")
|
||||
|
||||
# Other settings
|
||||
parser.add_argument("--skip_health_check",
|
||||
action="store_true",
|
||||
help="Skip health checks")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Ensure output directory exists
|
||||
os.makedirs(args.output_path, exist_ok=True)
|
||||
|
||||
print(" FastVideo Scalable Architecture")
|
||||
print("=" * 60)
|
||||
print(f"Architecture: ngrok -> nginx -> frontend1/frontend2 -> backend1×{args.num_gpus_per_cluster}/backend2×{args.num_gpus_per_cluster}")
|
||||
print(f"T2V Model: {args.t2v_model_path}")
|
||||
print(f"I2V Model: {args.i2v_model_path}")
|
||||
print(f"Output: {args.output_path}")
|
||||
print(f"Backend Clusters: {args.num_backend_clusters}")
|
||||
print(f"GPUs per Cluster: {args.num_gpus_per_cluster}")
|
||||
print(f"Total GPUs needed: {args.num_backend_clusters * args.num_gpus_per_cluster}")
|
||||
print(f"Frontend Instances: {args.num_frontends}")
|
||||
print(f"Nginx Port: {args.nginx_port}")
|
||||
print(f"Use Ngrok: {args.use_ngrok}")
|
||||
print("=" * 60)
|
||||
|
||||
# Start backend clusters
|
||||
backend_processes = []
|
||||
backend_urls = []
|
||||
|
||||
for i in range(args.num_backend_clusters):
|
||||
process, _ = start_backend_cluster(args, i)
|
||||
backend_processes.append(process)
|
||||
backend_urls.append(f"http://{args.backend_host}:{args.backend_base_port}")
|
||||
|
||||
# Wait for backends to be ready
|
||||
if not args.skip_health_check:
|
||||
print("\n⏳ Waiting for backend clusters to start...")
|
||||
for i, url in enumerate(backend_urls):
|
||||
if not check_service_health(f"{url}/cluster_{i}/health"):
|
||||
print(f"❌ Backend cluster {i + 1} failed to start. Terminating...")
|
||||
for process in backend_processes:
|
||||
process.terminate()
|
||||
sys.exit(1)
|
||||
|
||||
# Start frontend instances
|
||||
frontend_processes = []
|
||||
frontend_urls = []
|
||||
|
||||
for i in range(args.num_frontends):
|
||||
# Each frontend connects to a different backend cluster
|
||||
backend_url = backend_urls[i % len(backend_urls)]
|
||||
process, port = start_frontend_instance(args, i, backend_url)
|
||||
frontend_processes.append(process)
|
||||
frontend_urls.append(f"http://{args.frontend_host}:{port}")
|
||||
|
||||
# Wait for frontends to be ready
|
||||
if not args.skip_health_check:
|
||||
print("\n⏳ Waiting for frontend instances to start...")
|
||||
for i, url in enumerate(frontend_urls):
|
||||
if not check_service_health(url):
|
||||
print(f"❌ Frontend {i + 1} failed to start. Terminating...")
|
||||
for process in backend_processes + frontend_processes:
|
||||
process.terminate()
|
||||
sys.exit(1)
|
||||
|
||||
# Start nginx reverse proxy
|
||||
nginx_process = start_nginx(args)
|
||||
|
||||
# Wait for nginx to be ready
|
||||
if not args.skip_health_check:
|
||||
print("\n⏳ Waiting for nginx to start...")
|
||||
if not check_service_health(f"http://localhost:{args.nginx_port}/health"):
|
||||
print("❌ Nginx failed to start. Terminating...")
|
||||
for process in backend_processes + frontend_processes + [nginx_process]:
|
||||
process.terminate()
|
||||
sys.exit(1)
|
||||
|
||||
# Start ngrok tunnel (optional)
|
||||
ngrok_process = start_ngrok(args)
|
||||
|
||||
print("\n🎉 All services are starting up!")
|
||||
print(f"🌐 Nginx reverse proxy: http://localhost:{args.nginx_port}")
|
||||
for i, url in enumerate(frontend_urls):
|
||||
print(f"📺 Frontend {i + 1}: {url}")
|
||||
for i, url in enumerate(backend_urls):
|
||||
print(f" Backend Cluster {i + 1}: {url}")
|
||||
if args.use_ngrok:
|
||||
print("🌍 Ngrok tunnel is starting...")
|
||||
print("\nPress Ctrl+C to stop all services...")
|
||||
|
||||
# Signal handler for graceful shutdown
|
||||
def signal_handler(signum, frame):
|
||||
print("\n🛑 Shutting down all services...")
|
||||
all_processes = backend_processes + frontend_processes + [nginx_process]
|
||||
if ngrok_process:
|
||||
all_processes.append(ngrok_process)
|
||||
|
||||
for process in all_processes:
|
||||
if process:
|
||||
process.terminate()
|
||||
|
||||
# Wait for processes to terminate
|
||||
try:
|
||||
for process in all_processes:
|
||||
if process:
|
||||
process.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
print("⚠️ Force killing processes...")
|
||||
for process in all_processes:
|
||||
if process:
|
||||
process.kill()
|
||||
|
||||
print("✅ All services stopped")
|
||||
sys.exit(0)
|
||||
|
||||
signal.signal(signal.SIGINT, signal_handler)
|
||||
signal.signal(signal.SIGTERM, signal_handler)
|
||||
|
||||
# Monitor processes
|
||||
try:
|
||||
while True:
|
||||
# Check if processes are still running
|
||||
for i, process in enumerate(backend_processes):
|
||||
if process.poll() is not None:
|
||||
print(f"❌ Backend cluster {i + 1} process died unexpectedly")
|
||||
break
|
||||
|
||||
for i, process in enumerate(frontend_processes):
|
||||
if process.poll() is not None:
|
||||
print(f"❌ Frontend {i + 1} process died unexpectedly")
|
||||
break
|
||||
|
||||
if nginx_process and nginx_process.poll() is not None:
|
||||
print("❌ Nginx process died unexpectedly")
|
||||
break
|
||||
|
||||
if ngrok_process and ngrok_process.poll() is not None:
|
||||
print("❌ Ngrok process died unexpectedly")
|
||||
break
|
||||
|
||||
time.sleep(1)
|
||||
|
||||
except KeyboardInterrupt:
|
||||
signal_handler(signal.SIGINT, None)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,147 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Test script for T2V and I2V functionality in FastVideo Gradio app.
|
||||
This script tests both the backend and frontend modifications.
|
||||
"""
|
||||
|
||||
import requests
|
||||
import json
|
||||
import os
|
||||
from PIL import Image
|
||||
import numpy as np
|
||||
|
||||
def test_backend_t2v():
|
||||
"""Test the backend T2V functionality directly"""
|
||||
backend_url = "http://localhost:8000"
|
||||
|
||||
try:
|
||||
# Test T2V request data
|
||||
request_data = {
|
||||
"prompt": "A beautiful sunset over the ocean with gentle waves",
|
||||
"negative_prompt": "",
|
||||
"use_negative_prompt": False,
|
||||
"seed": 42,
|
||||
"guidance_scale": 7.5,
|
||||
"num_frames": 21,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_inference_steps": 20,
|
||||
"randomize_seed": False,
|
||||
"return_frames": True,
|
||||
"image_path": None,
|
||||
"model_type": "t2v"
|
||||
}
|
||||
|
||||
# Send request to backend
|
||||
response = requests.post(
|
||||
f"{backend_url}/generate_video",
|
||||
json=request_data,
|
||||
timeout=300 # 5 minutes timeout
|
||||
)
|
||||
|
||||
if response.status_code == 200:
|
||||
result = response.json()
|
||||
print("✅ Backend T2V test successful!")
|
||||
print(f"Success: {result.get('success')}")
|
||||
print(f"Seed used: {result.get('seed')}")
|
||||
if result.get('frames'):
|
||||
print(f"Frames returned: {len(result.get('frames'))}")
|
||||
else:
|
||||
print("No frames returned")
|
||||
else:
|
||||
print(f"❌ Backend T2V test failed with status {response.status_code}")
|
||||
print(f"Response: {response.text}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Backend T2V test failed with exception: {e}")
|
||||
|
||||
def test_backend_i2v():
|
||||
"""Test the backend I2V functionality directly"""
|
||||
backend_url = "http://localhost:8000"
|
||||
|
||||
# Create a simple test image
|
||||
test_image = Image.new('RGB', (256, 256), color='red')
|
||||
temp_image_path = "test_image.png"
|
||||
test_image.save(temp_image_path)
|
||||
|
||||
try:
|
||||
# Test I2V request data
|
||||
request_data = {
|
||||
"prompt": "The red square gently animates with subtle movement",
|
||||
"negative_prompt": "",
|
||||
"use_negative_prompt": False,
|
||||
"seed": 42,
|
||||
"guidance_scale": 7.5,
|
||||
"num_frames": 21,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_inference_steps": 20,
|
||||
"randomize_seed": False,
|
||||
"return_frames": True,
|
||||
"image_path": temp_image_path,
|
||||
"model_type": "i2v"
|
||||
}
|
||||
|
||||
# Send request to backend
|
||||
response = requests.post(
|
||||
f"{backend_url}/generate_video",
|
||||
json=request_data,
|
||||
timeout=300 # 5 minutes timeout
|
||||
)
|
||||
|
||||
if response.status_code == 200:
|
||||
result = response.json()
|
||||
print("✅ Backend I2V test successful!")
|
||||
print(f"Success: {result.get('success')}")
|
||||
print(f"Seed used: {result.get('seed')}")
|
||||
if result.get('frames'):
|
||||
print(f"Frames returned: {len(result.get('frames'))}")
|
||||
else:
|
||||
print("No frames returned")
|
||||
else:
|
||||
print(f"❌ Backend I2V test failed with status {response.status_code}")
|
||||
print(f"Response: {response.text}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Backend I2V test failed with exception: {e}")
|
||||
|
||||
finally:
|
||||
# Clean up test image
|
||||
if os.path.exists(temp_image_path):
|
||||
os.remove(temp_image_path)
|
||||
|
||||
def test_backend_health():
|
||||
"""Test if the backend is running"""
|
||||
backend_url = "http://localhost:8000"
|
||||
|
||||
try:
|
||||
response = requests.get(f"{backend_url}/health", timeout=5)
|
||||
if response.status_code == 200:
|
||||
print("✅ Backend is healthy")
|
||||
return True
|
||||
else:
|
||||
print(f"❌ Backend health check failed: {response.status_code}")
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f"❌ Backend health check failed: {e}")
|
||||
return False
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("🧪 Testing FastVideo T2V and I2V functionality...")
|
||||
print("=" * 50)
|
||||
|
||||
# Test backend health first
|
||||
if test_backend_health():
|
||||
# Test T2V functionality
|
||||
print("\n📝 Testing T2V functionality...")
|
||||
test_backend_t2v()
|
||||
|
||||
# Test I2V functionality
|
||||
print("\n🖼️ Testing I2V functionality...")
|
||||
test_backend_i2v()
|
||||
else:
|
||||
print("⚠️ Backend is not running. Please start the backend first.")
|
||||
print("You can start it with: python start_ray_serve_app.py")
|
||||
|
||||
print("=" * 50)
|
||||
print("Test completed!")
|
||||
@@ -10,7 +10,7 @@ def main():
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
pin_cpu_memory=False,
|
||||
lora_path="benjamin-paine/steamboat-willie-1.3b",
|
||||
lora_nickname="steamboat"
|
||||
)
|
||||
@@ -18,7 +18,7 @@ def main():
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 81,
|
||||
"guidance_scale": 6.0,
|
||||
"guidance_scale": 5.0,
|
||||
"num_inference_steps": 32,
|
||||
"seed": 42,
|
||||
}
|
||||
|
||||
@@ -11,23 +11,22 @@ def main():
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
lora_path="checkpoints/wan_t2v_finetune_lora/checkpoint-160/transformer",
|
||||
pin_cpu_memory=False,
|
||||
lora_path="checkpoints/wan_t2v_finetune_lora/checkpoint-1250/transformer",
|
||||
lora_nickname="crush_smol"
|
||||
)
|
||||
generator.unmerge_lora_weights()
|
||||
kwargs = {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77,
|
||||
"guidance_scale": 6.0,
|
||||
"guidance_scale": 5.0,
|
||||
"num_inference_steps": 50,
|
||||
"seed": 42,
|
||||
}
|
||||
# Generate video with LoRA style
|
||||
prompt = "A large metal cylinder is seen pressing down on a pile of Oreo cookies, flattening them as if they were under a hydraulic press."
|
||||
prompt = "A large metal cylinder is seen pressing down on a pile of colorful candies, flattening them as if they were under a hydraulic press. The candies are crushed and broken into small pieces, creating a mess on the table."
|
||||
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
@@ -35,12 +34,6 @@ def main():
|
||||
save_video=True,
|
||||
**kwargs
|
||||
)
|
||||
prompt = "A large metal cylinder is seen compressing colorful clay into a compact shape, demonstrating the power of a hydraulic press."
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
**kwargs
|
||||
)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -14,7 +14,7 @@ def main():
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
pin_cpu_memory=False,
|
||||
)
|
||||
load_time = time.perf_counter() - start_time
|
||||
print(f"Model loading time: {load_time:.2f} seconds")
|
||||
|
||||
@@ -12,7 +12,7 @@ def main():
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
pin_cpu_memory=False,
|
||||
)
|
||||
load_time = time.perf_counter() - start_time
|
||||
print(f"Model loading time: {load_time:.2f} seconds")
|
||||
|
||||
@@ -54,7 +54,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "40"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -69,6 +69,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
|
||||
@@ -88,7 +88,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "40"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -103,6 +103,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
|
||||
@@ -101,6 +101,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--dit_precision "fp32"
|
||||
|
||||
@@ -101,6 +101,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--dit_precision "fp32"
|
||||
|
||||
@@ -7,7 +7,7 @@ export TOKENIZERS_PARALLELISM=false
|
||||
|
||||
MODEL_PATH="Wan-AI/Wan2.1-I2V-14B-480P-Diffusers"
|
||||
DATA_DIR="data/crush-smol_processed_i2v/combined_parquet_dataset/"
|
||||
VALIDATION_DATASET_FILE="$(dirname "$0")/validation.json"
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_i2v_14b_480p/crush_smol/validation.json"
|
||||
NUM_GPUS=8
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
@@ -54,7 +54,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "40"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -69,6 +69,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
|
||||
@@ -88,7 +88,7 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "40"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -103,6 +103,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
|
||||
@@ -54,7 +54,7 @@ validation_args=(
|
||||
--validation_preprocessed_path "$VALIDATION_DIR"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "40"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -69,6 +69,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
|
||||
@@ -1,24 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
GPU_NUM=2 # 2,4,8
|
||||
MODEL_PATH="Wan-AI/Wan2.1-I2V-14B-480P-Diffusers"
|
||||
DATASET_PATH="data/crush-smol/"
|
||||
OUTPUT_DIR="data/crush-smol_processed_i2v/"
|
||||
|
||||
torchrun --nproc_per_node=$GPU_NUM \
|
||||
-m fastvideo.pipelines.preprocess.v1_preprocessing_new \
|
||||
--model_path $MODEL_PATH \
|
||||
--mode preprocess \
|
||||
--workload_type i2v \
|
||||
--preprocess.dataset_type merged \
|
||||
--preprocess.dataset_path $DATASET_PATH \
|
||||
--preprocess.dataset_output_dir $OUTPUT_DIR \
|
||||
--preprocess.preprocess_video_batch_size 2 \
|
||||
--preprocess.dataloader_num_workers 0 \
|
||||
--preprocess.max_height 480 \
|
||||
--preprocess.max_width 832 \
|
||||
--preprocess.num_frames 77 \
|
||||
--preprocess.train_fps 16 \
|
||||
--preprocess.samples_per_file 8 \
|
||||
--preprocess.flush_frequency 8 \
|
||||
--preprocess.video_length_tolerance_range 5
|
||||
@@ -7,7 +7,7 @@ export TOKENIZERS_PARALLELISM=false
|
||||
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATA_DIR="data/crush-smol_processed_t2v/combined_parquet_dataset/"
|
||||
VALIDATION_DATASET_FILE="$(dirname "$0")/validation.json"
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1_3b/crush_smol/validation.json"
|
||||
NUM_GPUS=4
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
@@ -52,9 +52,9 @@ dataset_args=(
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_steps 50
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
@@ -69,6 +69,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
@@ -77,7 +78,6 @@ miscellaneous_args=(
|
||||
--num_euler_timesteps 50
|
||||
--ema_start_step 0
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
# --resume_from_checkpoint "checkpoints/wan_t2v_finetune/checkpoint-2500"
|
||||
)
|
||||
|
||||
torchrun \
|
||||
|
||||
@@ -85,14 +85,14 @@ validation_args=(
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 5e-5
|
||||
--mixed_precision "bf16"
|
||||
--checkpointing_steps 400
|
||||
--checkpointing_steps 500
|
||||
--weight_decay 1e-4
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
@@ -100,6 +100,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
|
||||
@@ -6,8 +6,8 @@ export WANDB_MODE=online
|
||||
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATA_DIR="data/crush-smol_processed_t2v/combined_parquet_dataset/"
|
||||
VALIDATION_DATASET_FILE="$(dirname "$0")/validation.json"
|
||||
NUM_GPUS=1
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1_3b/crush_smol/validation.json"
|
||||
NUM_GPUS=2
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
|
||||
@@ -52,16 +52,16 @@ dataset_args=(
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_steps 50
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "6.0"
|
||||
--validation_guidance_scale "1.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 5e-5
|
||||
--mixed_precision "bf16"
|
||||
--checkpointing_steps 400
|
||||
--checkpointing_steps 500
|
||||
--weight_decay 1e-4
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
@@ -69,6 +69,7 @@ optimizer_args=(
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--allow_tf32
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
@@ -76,7 +77,6 @@ miscellaneous_args=(
|
||||
--dit_precision "fp32"
|
||||
--num_euler_timesteps 50
|
||||
--ema_start_step 0
|
||||
--resume_from_checkpoint "checkpoints/wan_t2v_finetune_lora/checkpoint-160"
|
||||
)
|
||||
|
||||
torchrun \
|
||||
|
||||
@@ -1,24 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
GPU_NUM=2 # 2,4,8
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATASET_PATH="data/crush-smol/"
|
||||
OUTPUT_DIR="data/crush-smol_processed_t2v/"
|
||||
|
||||
torchrun --nproc_per_node=$GPU_NUM \
|
||||
-m fastvideo.pipelines.preprocess.v1_preprocessing_new \
|
||||
--model_path $MODEL_PATH \
|
||||
--mode preprocess \
|
||||
--workload_type t2v \
|
||||
--preprocess.dataset_type merged \
|
||||
--preprocess.dataset_path $DATASET_PATH \
|
||||
--preprocess.dataset_output_dir $OUTPUT_DIR \
|
||||
--preprocess.preprocess_video_batch_size 2 \
|
||||
--preprocess.dataloader_num_workers 0 \
|
||||
--preprocess.max_height 480 \
|
||||
--preprocess.max_width 832 \
|
||||
--preprocess.num_frames 77 \
|
||||
--preprocess.train_fps 16 \
|
||||
--preprocess.samples_per_file 8 \
|
||||
--preprocess.flush_frequency 8 \
|
||||
--preprocess.video_length_tolerance_range 5
|
||||
@@ -55,7 +55,6 @@ class DistributedAttention(nn.Module):
|
||||
self.backend = backend_name_to_enum(attn_backend.get_name())
|
||||
self.dtype = dtype
|
||||
|
||||
@torch.compiler.disable
|
||||
def forward(
|
||||
self,
|
||||
q: torch.Tensor,
|
||||
@@ -137,7 +136,6 @@ class DistributedAttention_VSA(DistributedAttention):
|
||||
"""Distributed attention layer with VSA support.
|
||||
"""
|
||||
|
||||
@torch.compiler.disable
|
||||
def forward(
|
||||
self,
|
||||
q: torch.Tensor,
|
||||
|
||||
@@ -1,36 +1,9 @@
|
||||
import dataclasses
|
||||
from enum import Enum
|
||||
from typing import Any, Optional
|
||||
|
||||
from fastvideo.configs.utils import update_config_from_args
|
||||
from fastvideo.logger import init_logger
|
||||
from fastvideo.utils import FlexibleArgumentParser, StoreBoolean
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
|
||||
class DatasetType(str, Enum):
|
||||
"""
|
||||
Enumeration for different dataset types.
|
||||
"""
|
||||
HF = "hf"
|
||||
MERGED = "merged"
|
||||
|
||||
@classmethod
|
||||
def from_string(cls, value: str) -> "DatasetType":
|
||||
"""Convert string to DatasetType enum."""
|
||||
try:
|
||||
return cls(value.lower())
|
||||
except ValueError:
|
||||
raise ValueError(
|
||||
f"Invalid dataset type: {value}. Must be one of: {', '.join([m.value for m in cls])}"
|
||||
) from None
|
||||
|
||||
@classmethod
|
||||
def choices(cls) -> list[str]:
|
||||
"""Get all available choices as strings for argparse."""
|
||||
return [dataset_type.value for dataset_type in cls]
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class PreprocessConfig:
|
||||
@@ -39,7 +12,6 @@ class PreprocessConfig:
|
||||
# Model and dataset configuration
|
||||
model_path: str = ""
|
||||
dataset_path: str = ""
|
||||
dataset_type: DatasetType = DatasetType.HF
|
||||
dataset_output_dir: str = "./output"
|
||||
|
||||
# Dataloader configuration
|
||||
@@ -63,9 +35,6 @@ class PreprocessConfig:
|
||||
# Model configuration
|
||||
training_cfg_rate: float = 0.0
|
||||
|
||||
# framework configuration
|
||||
seed: int = 42
|
||||
|
||||
@staticmethod
|
||||
def add_cli_args(parser: FlexibleArgumentParser,
|
||||
prefix: str = "preprocess") -> FlexibleArgumentParser:
|
||||
@@ -83,12 +52,6 @@ class PreprocessConfig:
|
||||
type=str,
|
||||
default=PreprocessConfig.dataset_path,
|
||||
help="Path to the dataset directory for preprocessing")
|
||||
preprocess_args.add_argument(
|
||||
f"--{prefix_with_dot}dataset-type",
|
||||
type=str,
|
||||
choices=DatasetType.choices(),
|
||||
default=PreprocessConfig.dataset_type.value,
|
||||
help="Type of the dataset")
|
||||
preprocess_args.add_argument(
|
||||
f"--{prefix_with_dot}dataset-output-dir",
|
||||
type=str,
|
||||
@@ -160,10 +123,6 @@ class PreprocessConfig:
|
||||
type=float,
|
||||
default=PreprocessConfig.training_cfg_rate,
|
||||
help="Training CFG rate")
|
||||
preprocess_args.add_argument(f"--{prefix_with_dot}seed",
|
||||
type=int,
|
||||
default=PreprocessConfig.seed,
|
||||
help="Seed for random number generator")
|
||||
|
||||
return parser
|
||||
|
||||
@@ -171,10 +130,6 @@ class PreprocessConfig:
|
||||
def from_kwargs(cls, kwargs: dict[str,
|
||||
Any]) -> Optional["PreprocessConfig"]:
|
||||
"""Create PreprocessConfig from keyword arguments."""
|
||||
if 'dataset_type' in kwargs and isinstance(kwargs['dataset_type'], str):
|
||||
kwargs['dataset_type'] = DatasetType.from_string(
|
||||
kwargs['dataset_type'])
|
||||
|
||||
preprocess_config = cls()
|
||||
if not update_config_from_args(
|
||||
preprocess_config, kwargs, prefix="preprocess", pop_args=True):
|
||||
|
||||
@@ -92,21 +92,15 @@ class WanVideoArchConfig(DiTArchConfig):
|
||||
pos_embed_seq_len: int | None = None
|
||||
exclude_lora_layers: list[str] = field(default_factory=lambda: ["embedder"])
|
||||
|
||||
# Causal Wan
|
||||
local_attn_size: int = -1 # Window size for temporal local attention (-1 indicates global attention)
|
||||
sink_size: int = 0 # Size of the attention sink, we keep the first `sink_size` frames unchanged when rolling the KV cache
|
||||
num_frames_per_block: int = 3
|
||||
sliding_window_num_frames: int = 21
|
||||
|
||||
def __post_init__(self):
|
||||
super().__post_init__()
|
||||
self.out_channels = self.out_channels or self.in_channels
|
||||
self.hidden_size = self.num_attention_heads * self.attention_head_dim
|
||||
self.num_channels_latents = self.out_channels
|
||||
self.num_channels_latents = self.in_channels if self.added_kv_proj_dim is None else self.out_channels
|
||||
|
||||
|
||||
@dataclass
|
||||
class WanVideoConfig(DiTConfig):
|
||||
arch_config: DiTArchConfig = field(default_factory=WanVideoArchConfig)
|
||||
|
||||
prefix: str = "Wan"
|
||||
prefix: str = "Wan"
|
||||
|
||||
@@ -85,6 +85,9 @@ class PipelineConfig:
|
||||
# DMD parameters
|
||||
dmd_denoising_steps: list[int] | None = field(default=None)
|
||||
|
||||
# Wan2.2 TI2V parameters
|
||||
ti2v_task: bool = False
|
||||
|
||||
# Compilation
|
||||
# enable_torch_compile: bool = False
|
||||
|
||||
|
||||
@@ -7,9 +7,8 @@ from collections.abc import Callable
|
||||
from fastvideo.configs.pipelines.base import PipelineConfig
|
||||
from fastvideo.configs.pipelines.hunyuan import FastHunyuanConfig, HunyuanConfig
|
||||
from fastvideo.configs.pipelines.stepvideo import StepVideoT2VConfig
|
||||
from fastvideo.configs.pipelines.wan import (FastWan2_1_T2V_480P_Config,
|
||||
FastWan2_2_TI2V_5B_Config,
|
||||
SelfForcingWanT2V480PConfig,
|
||||
from fastvideo.configs.pipelines.wan import (FastWanT2V480PConfig,
|
||||
Wan2_2_TI2V_5B_Config,
|
||||
WanI2V480PConfig, WanI2V720PConfig,
|
||||
WanT2V480PConfig, WanT2V720PConfig)
|
||||
from fastvideo.logger import init_logger
|
||||
@@ -27,15 +26,13 @@ PIPE_NAME_TO_CONFIG: dict[str, type[PipelineConfig]] = {
|
||||
"Wan-AI/Wan2.1-I2V-14B-480P-Diffusers": WanI2V480PConfig,
|
||||
"Wan-AI/Wan2.1-I2V-14B-720P-Diffusers": WanI2V720PConfig,
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers": WanT2V720PConfig,
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers": FastWan2_1_T2V_480P_Config,
|
||||
"FastVideo/FastWan2.1-T2V-14B-480P-Diffusers": FastWan2_1_T2V_480P_Config,
|
||||
"FastVideo/FastWan2.2-TI2V-5B-Diffusers": FastWan2_2_TI2V_5B_Config,
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers": FastWanT2V480PConfig,
|
||||
"FastVideo/FastWan2.1-T2V-14B-480P-Diffusers": FastWanT2V480PConfig,
|
||||
"FastVideo/stepvideo-t2v-diffusers": StepVideoT2VConfig,
|
||||
"FastVideo/Wan2.1-VSA-T2V-14B-720P-Diffusers": WanT2V720PConfig,
|
||||
"Wan-AI/Wan2.2-TI2V-5B-Diffusers": WanT2V720PConfig,
|
||||
"Wan-AI/Wan2.2-T2V-A14B-Diffusers": WanT2V480PConfig,
|
||||
"Wan-AI/Wan2.2-I2V-A14B-Diffusers": WanI2V480PConfig,
|
||||
"wlsaidhi/SFWan2.1-T2V-1.3B-Diffusers": SelfForcingWanT2V480PConfig,
|
||||
"Wan-AI/Wan2.2-TI2V-5B-Diffusers": Wan2_2_TI2V_5B_Config,
|
||||
# "Wan-AI/Wan2.2-T2V-A14B-Diffusers": Wan2_2_T2V_A14B_Config,
|
||||
# "Wan-AI/Wan2.2-I2V-A14B-Diffusers": Wan2_2_I2V_A14B_Config,
|
||||
# Add other specific weight variants
|
||||
}
|
||||
|
||||
@@ -56,7 +53,7 @@ PIPELINE_FALLBACK_CONFIG: dict[str, type[PipelineConfig]] = {
|
||||
"wanpipeline":
|
||||
WanT2V480PConfig, # Base Wan config as fallback for any Wan variant
|
||||
"wanimagetovideo": WanI2V480PConfig,
|
||||
"wandmdpipeline": FastWan2_1_T2V_480P_Config,
|
||||
"wandmdpipeline": FastWanT2V480PConfig,
|
||||
"stepvideo": StepVideoT2VConfig
|
||||
# Other fallbacks by architecture
|
||||
}
|
||||
|
||||
@@ -98,7 +98,7 @@ class WanI2V720PConfig(WanI2V480PConfig):
|
||||
|
||||
|
||||
@dataclass
|
||||
class FastWan2_1_T2V_480P_Config(WanT2V480PConfig):
|
||||
class FastWanT2V480PConfig(WanT2V480PConfig):
|
||||
"""Base configuration for FastWan T2V 1.3B 480P pipeline architecture with DMD"""
|
||||
|
||||
# WanConfig-specific parameters with defaults
|
||||
@@ -115,6 +115,9 @@ class FastWan2_1_T2V_480P_Config(WanT2V480PConfig):
|
||||
|
||||
@dataclass
|
||||
class Wan2_2_TI2V_5B_Config(WanT2V480PConfig):
|
||||
"""Base configuration for FastWan T2V 1.3B 480P pipeline architecture with DMD"""
|
||||
|
||||
# Denoising stage
|
||||
flow_shift: int = 5
|
||||
ti2v_task: bool = True
|
||||
|
||||
@@ -123,13 +126,6 @@ class Wan2_2_TI2V_5B_Config(WanT2V480PConfig):
|
||||
self.vae_config.load_decoder = True
|
||||
|
||||
|
||||
@dataclass
|
||||
class FastWan2_2_TI2V_5B_Config(Wan2_2_TI2V_5B_Config):
|
||||
flow_shift: int = 5
|
||||
dmd_denoising_steps: list[int] | None = field(
|
||||
default_factory=lambda: [1000, 757, 522])
|
||||
|
||||
|
||||
@dataclass
|
||||
class Wan2_2_T2V_A14B_Config(WanT2V480PConfig):
|
||||
pass
|
||||
@@ -138,14 +134,3 @@ class Wan2_2_T2V_A14B_Config(WanT2V480PConfig):
|
||||
@dataclass
|
||||
class Wan2_2_I2V_A14B_Config(WanT2V480PConfig):
|
||||
pass
|
||||
|
||||
|
||||
# =============================================
|
||||
# ============= Causal Self-Forcing =============
|
||||
# =============================================
|
||||
@dataclass
|
||||
class SelfForcingWanT2V480PConfig(WanT2V480PConfig):
|
||||
is_causal: bool = True
|
||||
flow_shift: int = 5
|
||||
dmd_denoising_steps: list[int] | None = field(
|
||||
default_factory=lambda: [1000, 750, 500, 250])
|
||||
|
||||