Compare commits

...
Author SHA1 Message Date
rlsu9 a4b08a373b syn 2025-01-23 21:47:33 +00:00
rlsu9 1143f2382e add fa3 2025-01-19 02:13:03 +00:00
rlsu9 38f591dd08 add inference syn data 2025-01-19 01:54:24 +00:00
Yongqi Chen b53cf7425c Lora README update (#155) 2025-01-18 12:30:53 -08:00
22 changed files with 1287 additions and 78 deletions
+14 -16
View File
@@ -120,51 +120,49 @@ Then you can run the finetune with:
```
bash scripts/finetune/finetune_mochi.sh # for mochi
```
**Note that for finetuning, we did not tune the hyperparameters in the provided script**
**Note that for finetuning, we did not tune the hyperparameters in the provided script.**
### ⚡ Lora Finetune
Demos and prompts of Black-Myth-Wukong can be found in [here](https://huggingface.co/FastVideo/Hunyuan-Black-Myth-Wukong-lora-weight). You can download the Lora weight through:
Hunyuan supports Lora fine-tuning of videos up to 720p. Demos and prompts of Black-Myth-Wukong can be found in [here](https://huggingface.co/FastVideo/Hunyuan-Black-Myth-Wukong-lora-weight). You can download the Lora weight through:
```bash
python scripts/huggingface/download_hf.py --repo_id=FastVideo/Hunyuan-Black-Myth-Wukong-lora-weight --local_dir=data/Hunyuan-Black-Myth-Wukong-lora-weight --repo_type=model
```
#### Minimum Hardware Requirement
- 40 GB GPU memory each for 2 GPUs with lora.
- 30 GB GPU memory each for 2 GPUs with CPU offload and lora.
Currently, both Mochi and Hunyuan models support Lora finetuning through diffusers. To generate personalized videos from your own dataset, you'll need to follow three main steps: dataset preparation, finetuning, and inference.
#### Dataset Preparation
We provide scripts to better help you get started to train on your own characters!
You can run this to organize your dataset to get the videos2caption.json before preprocess. Specify your video folder and corresponding caption folder(Caption files should be .txt files and have the same name with its video):
You can run this to organize your dataset to get the videos2caption.json before preprocess. Specify your video folder and corresponding caption folder (caption files should be .txt files and have the same name with its video):
```
python scripts/dataset_preparation/prepare_json_file.py --video_dir data/input_videos/ --prompt_dir data/captions/ --output_path data/output_folder/videos2caption.json --verbose
```
Also, we provide script to resize your videos:
```
python scripts/data_preprocess/resize_videos.py \
--input_dir data/raw_videos/ \
--output_dir data/resized_videos/ \
--width 1280 \
--height 720 \
--fps 30
python scripts/data_preprocess/resize_videos.py
```
#### Finetuning
After basic dataset preparation and preprocess, you can start to finetune your model using Lora:
```
bash scripts/finetune/finetune_hunyuan_hf_lora.sh
bash scripts/finetune/finetune_mochi_lora.sh
```
#### Inference
For inference with Lora checkpoint, you can run the following scripts with Additional parameter --lora_checkpoint_dir:
For inference with Lora checkpoint, you can run the following scripts with additional parameter `--lora_checkpoint_dir`:
```
bash scripts/inference/inference_hunyuan_hf.sh
bash scripts/inference/inference_mochi_hf.sh
```
#### Minimum Hardware Requirement
- 40 GB GPU memory each for 2 GPUs with lora
- 30 GB GPU memory each for 2 GPUs with CPU offload and lora.
**We also provide scripts for Mochi in the same directory.**
#### Finetune with Both Image and Video
Our codebase support finetuning with both image and video.
```bash
bash scripts/finetune/finetune_hunyuan.sh
bash scripts/finetune/finetune_mochi_lora_mix.sh
```
For Image-Video Mixture Fine-tuning, make sure to enable the --group_frame option in your script.
For Image-Video Mixture Fine-tuning, make sure to enable the `--group_frame` option in your script.
## 📑 Development Plan
+33 -21
View File
@@ -1,36 +1,48 @@
from einops import rearrange
try:
from flash_attn_interface import flash_attn_func as flash_attn_func_v3
has_v3 = True
except:
pass
from flash_attn import flash_attn_varlen_qkvpacked_func
from flash_attn.bert_padding import pad_input, unpad_input
from einops import rearrange
def flash_attn_no_pad(qkv,
key_padding_mask,
causal=False,
dropout_p=0.0,
softmax_scale=None):
def flash_attn_no_pad(
qkv, key_padding_mask, causal=False, dropout_p=0.0, softmax_scale=None, attn_impl="fa3",
):
# adapted from https://github.com/Dao-AILab/flash-attention/blob/13403e81157ba37ca525890f2f0f2137edf75311/flash_attn/flash_attention.py#L27
batch_size = qkv.shape[0]
seqlen = qkv.shape[1]
nheads = qkv.shape[-2]
x = rearrange(qkv, "b s three h d -> b s (three h d)")
x_unpad, indices, cu_seqlens, max_s, used_seqlens_in_batch = unpad_input(
x, key_padding_mask)
x_unpad = rearrange(x_unpad,
"nnz (three h d) -> nnz three h d",
three=3,
h=nheads)
output_unpad = flash_attn_varlen_qkvpacked_func(
x_unpad,
cu_seqlens,
max_s,
dropout_p,
softmax_scale=softmax_scale,
causal=causal,
x, key_padding_mask
)
x_unpad = rearrange(x_unpad, "nnz (three h d) -> nnz three h d", three=3, h=nheads)
if attn_impl == "fa3":
assert has_v3
q, k, v = x_unpad[:, 0].unsqueeze(0), x_unpad[:, 1].unsqueeze(0), x_unpad[:, 2].unsqueeze(0)
assert dropout_p == 0.0
output_unpad = flash_attn_func_v3(
q,
k,
v,
)[0].squeeze(0)
else:
output_unpad = flash_attn_varlen_qkvpacked_func(
x_unpad,
cu_seqlens,
max_s,
dropout_p,
softmax_scale=softmax_scale,
causal=causal,
)
output = rearrange(
pad_input(rearrange(output_unpad, "nnz h d -> nnz (h d)"), indices,
batch_size, seqlen),
pad_input(
rearrange(output_unpad, "nnz h d -> nnz (h d)"), indices, batch_size, seqlen
),
"b s (h d) -> b s h d",
h=nheads,
)
@@ -606,6 +606,7 @@ class HunyuanVideoPipeline(DiffusionPipeline):
enable_vae_sp: bool = False,
n_tokens: Optional[int] = None,
embedded_guidance_scale: Optional[float] = None,
only_save_latent: bool = False,
**kwargs,
):
r"""
@@ -978,8 +979,13 @@ class HunyuanVideoPipeline(DiffusionPipeline):
latents = (latents / self.vae.config.scaling_factor +
self.vae.config.shift_factor)
else:
latents = latents / self.vae.config.scaling_factor
latents = latents / self.vae.config.scaling_factor # 1, 16, 32, 90, 160
if only_save_latent:
latents = latents.cpu().float()
self.maybe_free_model_hooks()
return latents
with torch.autocast(device_type="cuda",
dtype=vae_dtype,
enabled=vae_autocast_enabled):
@@ -993,7 +999,6 @@ class HunyuanVideoPipeline(DiffusionPipeline):
if expand_temporal_dim or image.shape[2] == 1:
image = image.squeeze(2)
else:
image = latents
+4
View File
@@ -369,6 +369,7 @@ class HunyuanVideoSampler(Inference):
embedded_guidance_scale=None,
batch_size=1,
num_videos_per_prompt=1,
only_save_latent=False,
**kwargs,
):
"""
@@ -524,7 +525,10 @@ class HunyuanVideoSampler(Inference):
vae_ver=self.args.vae,
enable_tiling=self.args.vae_tiling,
enable_vae_sp=self.args.vae_sp,
only_save_latent=only_save_latent,
)[0]
if only_save_latent:
return samples
out_dict["samples"] = samples
out_dict["prompts"] = prompt
@@ -0,0 +1,286 @@
import argparse
import os
from pathlib import Path
import imageio
import numpy as np
import torch
import torch.distributed as dist
import torchvision
from einops import rearrange
import json
from fastvideo.models.hunyuan.inference import HunyuanVideoSampler
from fastvideo.utils.parallel_states import (
initialize_sequence_parallel_state, nccl_info)
def initialize_distributed():
local_rank = int(os.getenv("RANK", 0))
world_size = int(os.getenv("WORLD_SIZE", 1))
print("world_size", world_size)
torch.cuda.set_device(local_rank)
dist.init_process_group(backend="nccl",
init_method="env://",
world_size=world_size,
rank=local_rank)
initialize_sequence_parallel_state(world_size)
def main(args):
print(nccl_info.sp_size)
print(args)
models_root_path = Path(args.model_path)
if not models_root_path.exists():
raise ValueError(f"`models_root` not exists: {models_root_path}")
save_path = args.output_path
os.makedirs(os.path.dirname(save_path), exist_ok=True)
# Handle both JSON and TXT files
prompts = []
if args.prompt_json.endswith('.json'):
with open(args.prompt_json) as f:
prompt_json = json.load(f)
prompts = [(item["caption"], item["latent_path"]) for item in prompt_json]
else:
with open(args.prompt_json) as f:
for line in f:
prompt = line.strip()
if prompt:
# Use first 50 chars of prompt as filename
file_name = "".join(c for c in prompt[:50] if c.isalnum() or c.isspace())
file_name = file_name.strip().replace(" ", "_")
prompts.append((prompt, file_name))
# Filter for unprocessed prompts in latent-only mode
if args.only_save_latent:
filtered_prompts = []
for prompt, file_name in prompts:
latent_path = os.path.join(args.output_path, "latent", f"{file_name}.pt")
if not os.path.exists(latent_path):
filtered_prompts.append((prompt, file_name))
else:
print(f"Latent file exists for {file_name}, skipping...")
prompts = filtered_prompts
if not prompts:
print("All prompts processed. Exiting...")
return
hunyuan_video_sampler = HunyuanVideoSampler.from_pretrained(
models_root_path, args=args)
args = hunyuan_video_sampler.args
prompt_start_end_idx = [int(i) for i in args.prompt_start_end_idx.split(",")]
prompts = prompts[prompt_start_end_idx[0]:prompt_start_end_idx[1]]
for prompt, file_name in prompts:
outputs = hunyuan_video_sampler.predict(
prompt=prompt,
height=args.height,
width=args.width,
video_length=args.num_frames,
seed=args.seed,
negative_prompt=args.neg_prompt,
infer_steps=args.num_inference_steps,
guidance_scale=args.guidance_scale,
num_videos_per_prompt=args.num_videos,
flow_shift=args.flow_shift,
batch_size=args.batch_size,
embedded_guidance_scale=args.embedded_cfg_scale,
only_save_latent=args.only_save_latent,
)
if args.only_save_latent:
os.makedirs(os.path.join(args.output_path, "latent"), exist_ok=True)
latent_path = os.path.join(args.output_path, "latent", f"{file_name}.pt")
torch.save(outputs.to(torch.bfloat16), latent_path)
else:
videos = rearrange(outputs["samples"], "b c t h w -> t b c h w")
outputs = []
for x in videos:
x = torchvision.utils.make_grid(x, nrow=6)
x = x.transpose(0, 1).transpose(1, 2).squeeze(-1)
outputs.append((x * 255).numpy().astype(np.uint8))
os.makedirs(os.path.dirname(args.output_path), exist_ok=True)
imageio.mimsave(os.path.join(args.output_path, f"{file_name}.mp4"),
outputs,
fps=args.fps)
#
if __name__ == "__main__":
parser = argparse.ArgumentParser()
# Basic parameters
parser.add_argument("--prompt_json", type=str, required=True, help="prompt file for inference")
parser.add_argument("--num_frames", type=int, default=16)
parser.add_argument("--height", type=int, default=256)
parser.add_argument("--width", type=int, default=256)
parser.add_argument("--num_inference_steps", type=int, default=50)
parser.add_argument("--model_path", type=str, default="data/hunyuan")
parser.add_argument("--output_path", type=str, default="./outputs/video")
parser.add_argument("--fps", type=int, default=24)
# Additional parameters
parser.add_argument(
"--prompt_start_end_idx",
type=str,
default="0,99",
help=" 0 stands for starting layer index, 9 for stride",
)
parser.add_argument(
"--only_save_latent",
type=bool,
default=False,
help="Whether save latent or not.",
)
parser.add_argument(
"--denoise-type",
type=str,
default="flow",
help="Denoise type for noised inputs.",
)
parser.add_argument("--seed",
type=int,
default=None,
help="Seed for evaluation.")
parser.add_argument("--neg_prompt",
type=str,
default=None,
help="Negative prompt for sampling.")
parser.add_argument(
"--guidance_scale",
type=float,
default=1.0,
help="Classifier free guidance scale.",
)
parser.add_argument(
"--embedded_cfg_scale",
type=float,
default=6.0,
help="Embedded classifier free guidance scale.",
)
parser.add_argument("--flow_shift",
type=int,
default=7,
help="Flow shift parameter.")
parser.add_argument("--batch_size",
type=int,
default=1,
help="Batch size for inference.")
parser.add_argument(
"--num_videos",
type=int,
default=1,
help="Number of videos to generate per prompt.",
)
parser.add_argument(
"--load-key",
type=str,
default="module",
help=
"Key to load the model states. 'module' for the main model, 'ema' for the EMA model.",
)
parser.add_argument(
"--use-cpu-offload",
action="store_true",
help="Use CPU offload for the model load.",
)
parser.add_argument(
"--dit-weight",
type=str,
default=
"data/hunyuan/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt",
)
parser.add_argument(
"--reproduce",
action="store_true",
help=
"Enable reproducibility by setting random seeds and deterministic algorithms.",
)
parser.add_argument(
"--disable-autocast",
action="store_true",
help=
"Disable autocast for denoising loop and vae decoding in pipeline sampling.",
)
# Flow Matching
parser.add_argument(
"--flow-reverse",
action="store_true",
help="If reverse, learning/sampling from t=1 -> t=0.",
)
parser.add_argument("--flow-solver",
type=str,
default="euler",
help="Solver for flow matching.")
parser.add_argument(
"--use-linear-quadratic-schedule",
action="store_true",
help=
"Use linear quadratic schedule for flow matching. Following MovieGen (https://ai.meta.com/static-resource/movie-gen-research-paper)",
)
parser.add_argument(
"--linear-schedule-end",
type=int,
default=25,
help="End step for linear quadratic schedule for flow matching.",
)
# Model parameters
parser.add_argument("--model", type=str, default="HYVideo-T/2-cfgdistill")
parser.add_argument("--latent-channels", type=int, default=16)
parser.add_argument("--precision",
type=str,
default="bf16",
choices=["fp32", "fp16", "bf16"])
parser.add_argument("--rope-theta",
type=int,
default=256,
help="Theta used in RoPE.")
parser.add_argument("--vae", type=str, default="884-16c-hy")
parser.add_argument("--vae-precision",
type=str,
default="fp16",
choices=["fp32", "fp16", "bf16"])
parser.add_argument("--vae-tiling", action="store_true", default=True)
parser.add_argument("--vae-sp", action="store_true", default=False)
parser.add_argument("--text-encoder", type=str, default="llm")
parser.add_argument(
"--text-encoder-precision",
type=str,
default="fp16",
choices=["fp32", "fp16", "bf16"],
)
parser.add_argument("--text-states-dim", type=int, default=4096)
parser.add_argument("--text-len", type=int, default=256)
parser.add_argument("--tokenizer", type=str, default="llm")
parser.add_argument("--prompt-template",
type=str,
default="dit-llm-encode")
parser.add_argument("--prompt-template-video",
type=str,
default="dit-llm-encode-video")
parser.add_argument("--hidden-state-skip-layer", type=int, default=2)
parser.add_argument("--apply-final-norm", action="store_true")
parser.add_argument("--text-encoder-2", type=str, default="clipL")
parser.add_argument(
"--text-encoder-precision-2",
type=str,
default="fp16",
choices=["fp32", "fp16", "bf16"],
)
parser.add_argument("--text-states-dim-2", type=int, default=768)
parser.add_argument("--tokenizer-2", type=str, default="clipL")
parser.add_argument("--text-len-2", type=int, default=77)
args = parser.parse_args()
# process for vae sequence parallel
if args.vae_sp and not args.vae_tiling:
raise ValueError(
"Currently enabling vae_sp requires enabling vae_tiling, please set --vae-tiling to True."
)
main(args)
@@ -1,16 +1,16 @@
import argparse
import json
import os
import torch
import torch.distributed as dist
from diffusers.utils import export_to_video
import torch
from fastvideo.models.mochi_hf.pipeline_mochi import MochiPipeline
import os
from diffusers.utils import export_to_video
import argparse
def generate_video_and_latent(pipe, prompt, height, width, num_frames,
num_inference_steps, guidance_scale):
def generate_video_and_latent(
pipe, prompt, height, width, num_frames, num_inference_steps, guidance_scale
):
# Set the random seed for reproducibility
generator = torch.Generator("cuda").manual_seed(12345)
# Generate videos from the input prompt
@@ -25,8 +25,7 @@ def generate_video_and_latent(pipe, prompt, height, width, num_frames,
output_type="latent_and_video",
)
# prompt_embed has negative prompt at index 0
return noise[0], video[0], latent[0], prompt_embed[
1], prompt_attention_mask[1]
return noise[0], video[0], latent[0], prompt_embed[1], prompt_attention_mask[1]
# return dummy tensor to debug first
# return torch.zeros(1, 3, 480, 848), torch.zeros(1, 256, 16, 16)
@@ -40,22 +39,19 @@ if __name__ == "__main__":
parser.add_argument("--num_inference_steps", type=int, default=64)
parser.add_argument("--guidance_scale", type=float, default=4.5)
parser.add_argument("--model_path", type=str, default="data/mochi")
parser.add_argument("--prompt_path",
type=str,
default="data/dummyVid/videos2caption.json")
parser.add_argument("--dataset_output_dir",
type=str,
default="data/dummySynthetic")
parser.add_argument(
"--prompt_path", type=str, default="data/dummyVid/videos2caption.json"
)
parser.add_argument("--dataset_output_dir", type=str, default="data/dummySynthetic")
args = parser.parse_args()
local_rank = int(os.getenv("RANK", 0))
world_size = int(os.getenv("WORLD_SIZE", 1))
print("world_size", world_size, "local rank", local_rank)
torch.cuda.set_device(local_rank)
dist.init_process_group(backend="nccl",
init_method="env://",
world_size=world_size,
rank=local_rank)
dist.init_process_group(
backend="nccl", init_method="env://", world_size=world_size, rank=local_rank
)
if not isinstance(args.prompt_path, list):
args.prompt_path = [args.prompt_path]
@@ -63,8 +59,7 @@ if __name__ == "__main__":
text_prompt = open(args.prompt_path[0], "r").readlines()
text_prompt = [i.strip() for i in text_prompt]
pipe = MochiPipeline.from_pretrained(args.model_path,
torch_dtype=torch.bfloat16)
pipe = MochiPipeline.from_pretrained(args.model_path, torch_dtype=torch.bfloat16)
pipe.enable_vae_tiling()
pipe.enable_model_cpu_offload(gpu_id=local_rank)
# make dir if not exist
@@ -73,10 +68,10 @@ if __name__ == "__main__":
os.makedirs(os.path.join(args.dataset_output_dir, "noise"), exist_ok=True)
os.makedirs(os.path.join(args.dataset_output_dir, "video"), exist_ok=True)
os.makedirs(os.path.join(args.dataset_output_dir, "latent"), exist_ok=True)
os.makedirs(os.path.join(args.dataset_output_dir, "prompt_embed"),
exist_ok=True)
os.makedirs(os.path.join(args.dataset_output_dir, "prompt_attention_mask"),
exist_ok=True)
os.makedirs(os.path.join(args.dataset_output_dir, "prompt_embed"), exist_ok=True)
os.makedirs(
os.path.join(args.dataset_output_dir, "prompt_attention_mask"), exist_ok=True
)
data = []
for i, prompt in enumerate(text_prompt):
if i % world_size != local_rank:
@@ -98,17 +93,17 @@ if __name__ == "__main__":
)
# save latent
video_name = str(i)
noise_path = os.path.join(args.dataset_output_dir, "noise",
video_name + ".pt")
latent_path = os.path.join(args.dataset_output_dir, "latent",
video_name + ".pt")
prompt_embed_path = os.path.join(args.dataset_output_dir,
"prompt_embed", video_name + ".pt")
video_path = os.path.join(args.dataset_output_dir, "video",
video_name + ".mp4")
prompt_attention_mask_path = os.path.join(args.dataset_output_dir,
"prompt_attention_mask",
video_name + ".pt")
noise_path = os.path.join(args.dataset_output_dir, "noise", video_name + ".pt")
latent_path = os.path.join(
args.dataset_output_dir, "latent", video_name + ".pt"
)
prompt_embed_path = os.path.join(
args.dataset_output_dir, "prompt_embed", video_name + ".pt"
)
video_path = os.path.join(args.dataset_output_dir, "video", video_name + ".mp4")
prompt_attention_mask_path = os.path.join(
args.dataset_output_dir, "prompt_attention_mask", video_name + ".pt"
)
# save latent
torch.save(noise, noise_path)
torch.save(latent, latent_path)
@@ -132,6 +127,7 @@ if __name__ == "__main__":
# save json
if local_rank == 0:
all_data = [item for sublist in gathered_data for item in sublist]
with open(os.path.join(args.dataset_output_dir, "videos2caption.json"),
"w") as f:
with open(
os.path.join(args.dataset_output_dir, "videos2caption.json"), "w"
) as f:
json.dump(all_data, f, indent=4)
@@ -0,0 +1,62 @@
#!/bin/bash
#SBATCH --job-name=fastvideo
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log/slurm_lora_1e4.out
#SBATCH --error=slurm_log/slurm_lora_1e4.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
export MASTER_PORT=29803
torchrun --nnodes 1 --nproc_per_node 8 --master_port $MASTER_PORT \
fastvideo/train.py \
--seed 1024 \
--pretrained_model_name_or_path ~/data/hunyuan_diffusers \
--model_type hunyuan_hf \
--cache_dir data/.cache \
--data_json_path ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--validation_prompt_dir ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/validation \
--gradient_checkpointing \
--train_batch_size 8 \
--num_latent_t 32 \
--sp_size 8 \
--train_sp_batch_size 1 \
--dataloader_num_workers 4 \
--gradient_accumulation_steps 1 \
--max_train_steps 8000 \
--learning_rate 1e-4 \
--mixed_precision bf16 \
--checkpointing_steps 100 \
--validation_steps 100 \
--validation_sampling_steps 50 \
--checkpoints_total_limit 3 \
--allow_tf32 \
--ema_start_step 0 \
--cfg 0.0 \
--ema_decay 0.999 \
--log_validation \
--output_dir data/outputs/SBA_lora_1e4_r32 \
--tracker_project_name SBA \
--num_frames 125 \
--num_width 1280 \
--num_height 768 \
--validation_guidance_scale "1.0" \
--shift 7 \
--use_lora \
--lora_rank 32 \
--lora_alpha 32
@@ -0,0 +1,100 @@
#!/bin/bash
#SBATCH --job-name=fv_syn_hunyuan
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log/slurm_ori_ft_syn.out
#SBATCH --error=slurm_log/slurm_ori_ft_syn.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
export MASTER_PORT=29803
torchrun --nnodes 1 --nproc_per_node 6 \
fastvideo/train.py \
--seed 42 \
--pretrained_model_name_or_path data/hunyuan \
--dit_model_name_or_path data/hunyuan/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt\
--model_type "hunyuan" \
--cache_dir data/.cache \
--data_json_path ../FastVideo-Internal/data/synthetic_MixKit/videos2caption.json \
--validation_prompt_dir ../FastVideo-Internal/data/synthetic_MixKit/validation \
--gradient_checkpointing \
--train_batch_size=1 \
--num_latent_t 30 \
--sp_size 6 \
--train_sp_batch_size 1 \
--dataloader_num_workers 4 \
--gradient_accumulation_steps=2 \
--max_train_steps=2000 \
--learning_rate=1e-5 \
--mixed_precision=bf16 \
--checkpointing_steps=200 \
--validation_steps 100 \
--validation_sampling_steps 50 \
--checkpoints_total_limit 3 \
--allow_tf32 \
--ema_start_step 0 \
--cfg 0.0 \
--ema_decay 0.999 \
--log_validation \
--output_dir=data/outputs/SBA_hunyuan_ori_ft_1e5 \
--tracker_project_name SBA-Syn \
--num_frames 117 \
--num_height 768 \
--num_width 1280 \
--shift 7 \
--validation_guidance_scale "1.0" \
--training_guidance "6.0" \
--run_name "SBA_hunyuan_ori_ft_1e5_g6" \
torchrun --nnodes 1 --nproc_per_node 6 \
fastvideo/train.py \
--seed 42 \
--pretrained_model_name_or_path data/hunyuan \
--dit_model_name_or_path data/hunyuan/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt\
--model_type "hunyuan" \
--cache_dir data/.cache \
--data_json_path ../FastVideo-Internal/data/synthetic_MixKit/videos2caption.json \
--validation_prompt_dir ../FastVideo-Internal/data/synthetic_MixKit/validation \
--gradient_checkpointing \
--train_batch_size=1 \
--num_latent_t 30 \
--sp_size 6 \
--train_sp_batch_size 1 \
--dataloader_num_workers 4 \
--gradient_accumulation_steps=2 \
--max_train_steps=2000 \
--learning_rate=1e-5 \
--mixed_precision=bf16 \
--checkpointing_steps=200 \
--validation_steps 100 \
--validation_sampling_steps 50 \
--checkpoints_total_limit 3 \
--allow_tf32 \
--ema_start_step 0 \
--cfg 0.0 \
--ema_decay 0.999 \
--log_validation \
--output_dir=data/outputs/SBA_hunyuan_ori_ft_1e5 \
--tracker_project_name SBA-Syn \
--num_frames 117 \
--num_height 768 \
--num_width 1280 \
--shift 7 \
--validation_guidance_scale "1.0" \
--training_guidance "1.0" \
--run_name "SBA_hunyuan_ori_ft_1e5_g1" \
+12
View File
@@ -0,0 +1,12 @@
import json
src_json = "../HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json"
des_json = "./videos2caption.json"
syn_num = 8
with open(src_json) as f:
data = json.load(f)
data = data[:syn_num]
# from IPython import embed
# embed()
with open(des_json, "w") as f2:
json.dump(data, f2, indent=4, ensure_ascii=False)
+65
View File
@@ -0,0 +1,65 @@
#!/bin/bash
#SBATCH --job-name=hyhf-lora-5e6-r32
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=4
#SBATCH --mem=512G
#SBATCH --output=slurm_log/slurm.out
#SBATCH --error=slurm_log/slurm.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
cd ~/yongqi/FastVideo
export MASTER_PORT=29803
torchrun --nnodes 1 --nproc_per_node 8 --master_port $MASTER_PORT \
fastvideo/train.py \
--seed 1024 \
--pretrained_model_name_or_path ~/data/hunyuan_diffusers \
--model_type hunyuan_hf \
--cache_dir data/.cache \
--data_json_path ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--validation_prompt_dir ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/validation \
--gradient_checkpointing \
--train_batch_size 4 \
--num_latent_t 32 \
--sp_size 8 \
--train_sp_batch_size 1 \
--dataloader_num_workers 4 \
--gradient_accumulation_steps 1 \
--max_train_steps 8000 \
--learning_rate 5e-6 \
--mixed_precision bf16 \
--checkpointing_steps 200 \
--validation_steps 100 \
--validation_sampling_steps 50 \
--checkpoints_total_limit 3 \
--allow_tf32 \
--ema_start_step 0 \
--cfg 0.0 \
--ema_decay 0.999 \
--log_validation \
--output_dir data/outputs/SBA_lora_5e6_r32 \
--tracker_project_name SBA \
--num_frames 125 \
--num_width 1280 \
--num_height 768 \
--validation_guidance_scale "1.0" \
--use_lora \
--lora_rank 32 \
--lora_alpha 64
echo "Run completed at:- "
date
@@ -0,0 +1,60 @@
#!/bin/bash
#SBATCH --job-name=syn-1
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-1.out
#SBATCH --error=slurm_log_last_dance/syn-1.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=0
num_promtps_all=200
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ./assets/prompt_vb.txt \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-10
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-10.out
#SBATCH --error=slurm_log_last_dance/syn-10.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=2160
num_promtps_all=240
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-11
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-11.out
#SBATCH --error=slurm_log_last_dance/syn-11.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=2400
num_promtps_all=240
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-2
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-2.out
#SBATCH --error=slurm_log_last_dance/syn-2.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=198
num_promtps_all=200
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ./assets/prompt_vb.txt \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-3
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-3.out
#SBATCH --error=slurm_log_last_dance/syn-3.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=396
num_promtps_all=200
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ./assets/prompt_vb.txt \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-4
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-4.out
#SBATCH --error=slurm_log_last_dance/syn-4.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=594
num_promtps_all=200
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ./assets/prompt_vb.txt \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-5
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-5.out
#SBATCH --error=slurm_log_last_dance/syn-5.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=960
num_promtps_all=240
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-6
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-6.out
#SBATCH --error=slurm_log_last_dance/syn-6.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=1200
num_promtps_all=240
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-7
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-7.out
#SBATCH --error=slurm_log_last_dance/syn-7.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=1440
num_promtps_all=240
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-8
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-8.out
#SBATCH --error=slurm_log_last_dance/syn-8.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=1680
num_promtps_all=240
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
@@ -0,0 +1,59 @@
#!/bin/bash
#SBATCH --job-name=syn-9
#SBATCH --partition=mbzuai
#SBATCH --nodes=1
#SBATCH --ntasks-per-node=8
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=8
#SBATCH --mem=960G
#SBATCH --output=slurm_log_last_dance/syn-9.out
#SBATCH --error=slurm_log_last_dance/syn-9.err
#SBATCH --exclusive
#SBATCH --time=72:00:00
conda init
source ~/conda/miniconda/bin/activate
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
cd /mbz/users/hao.zhang/yongqi/FastVideo
echo " "
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
echo " GPUs per node:= " $SLURM_JOB_GPUS
echo " Running on multiple nodes/GPU devices"
echo ""
echo " Run started at:- "
date
# Calculate prompts per GPU
prompt_start_idx_all=1920
num_promtps_all=240
num_gpus=8
prompts_per_gpu=$((num_promtps_all / num_gpus))
export MODEL_BASE=data/hunyuan
for gpu_id in $(seq 0 $((num_gpus-1))); do
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
end_idx=$((start_idx + prompts_per_gpu))
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 6 \
--flow_shift 7 \
--flow-reverse \
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
--seed 1024 \
--output_path outputs_latents/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--only_save_latent True \
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
done
# Wait for all background jobs to complete
wait
+19
View File
@@ -18,3 +18,22 @@ torchrun --nnodes=1 --nproc_per_node=$num_gpus --master_port 29503 \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--vae-sp
num_gpus=6
export MODEL_BASE=data/hunyuan
torchrun --nnodes=1 --nproc_per_node=$num_gpus --master_port 29503 \
fastvideo/sample/sample_t2v_hunyuan.py \
--height 768 \
--width 1280 \
--num_frames 117 \
--num_inference_steps 50 \
--guidance_scale 1 \
--embedded_cfg_scale 8 \
--flow_shift 7 \
--flow-reverse \
--prompt ./assets/prompt_mixkit.txt \
--seed 1024 \
--output_path outputs_video/mixkit-8/ \
--model_path $MODEL_BASE \
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
--vae-sp