Compare commits
3
Commits
v0.1.2
...
generate_syn
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a4b08a373b | ||
|
|
1143f2382e | ||
|
|
38f591dd08 |
@@ -1,36 +1,48 @@
|
||||
from einops import rearrange
|
||||
try:
|
||||
from flash_attn_interface import flash_attn_func as flash_attn_func_v3
|
||||
has_v3 = True
|
||||
except:
|
||||
pass
|
||||
from flash_attn import flash_attn_varlen_qkvpacked_func
|
||||
from flash_attn.bert_padding import pad_input, unpad_input
|
||||
from einops import rearrange
|
||||
|
||||
|
||||
def flash_attn_no_pad(qkv,
|
||||
key_padding_mask,
|
||||
causal=False,
|
||||
dropout_p=0.0,
|
||||
softmax_scale=None):
|
||||
def flash_attn_no_pad(
|
||||
qkv, key_padding_mask, causal=False, dropout_p=0.0, softmax_scale=None, attn_impl="fa3",
|
||||
):
|
||||
# adapted from https://github.com/Dao-AILab/flash-attention/blob/13403e81157ba37ca525890f2f0f2137edf75311/flash_attn/flash_attention.py#L27
|
||||
batch_size = qkv.shape[0]
|
||||
seqlen = qkv.shape[1]
|
||||
nheads = qkv.shape[-2]
|
||||
x = rearrange(qkv, "b s three h d -> b s (three h d)")
|
||||
x_unpad, indices, cu_seqlens, max_s, used_seqlens_in_batch = unpad_input(
|
||||
x, key_padding_mask)
|
||||
|
||||
x_unpad = rearrange(x_unpad,
|
||||
"nnz (three h d) -> nnz three h d",
|
||||
three=3,
|
||||
h=nheads)
|
||||
output_unpad = flash_attn_varlen_qkvpacked_func(
|
||||
x_unpad,
|
||||
cu_seqlens,
|
||||
max_s,
|
||||
dropout_p,
|
||||
softmax_scale=softmax_scale,
|
||||
causal=causal,
|
||||
x, key_padding_mask
|
||||
)
|
||||
|
||||
x_unpad = rearrange(x_unpad, "nnz (three h d) -> nnz three h d", three=3, h=nheads)
|
||||
if attn_impl == "fa3":
|
||||
assert has_v3
|
||||
q, k, v = x_unpad[:, 0].unsqueeze(0), x_unpad[:, 1].unsqueeze(0), x_unpad[:, 2].unsqueeze(0)
|
||||
assert dropout_p == 0.0
|
||||
output_unpad = flash_attn_func_v3(
|
||||
q,
|
||||
k,
|
||||
v,
|
||||
)[0].squeeze(0)
|
||||
else:
|
||||
output_unpad = flash_attn_varlen_qkvpacked_func(
|
||||
x_unpad,
|
||||
cu_seqlens,
|
||||
max_s,
|
||||
dropout_p,
|
||||
softmax_scale=softmax_scale,
|
||||
causal=causal,
|
||||
)
|
||||
output = rearrange(
|
||||
pad_input(rearrange(output_unpad, "nnz h d -> nnz (h d)"), indices,
|
||||
batch_size, seqlen),
|
||||
pad_input(
|
||||
rearrange(output_unpad, "nnz h d -> nnz (h d)"), indices, batch_size, seqlen
|
||||
),
|
||||
"b s (h d) -> b s h d",
|
||||
h=nheads,
|
||||
)
|
||||
|
||||
@@ -606,6 +606,7 @@ class HunyuanVideoPipeline(DiffusionPipeline):
|
||||
enable_vae_sp: bool = False,
|
||||
n_tokens: Optional[int] = None,
|
||||
embedded_guidance_scale: Optional[float] = None,
|
||||
only_save_latent: bool = False,
|
||||
**kwargs,
|
||||
):
|
||||
r"""
|
||||
@@ -978,8 +979,13 @@ class HunyuanVideoPipeline(DiffusionPipeline):
|
||||
latents = (latents / self.vae.config.scaling_factor +
|
||||
self.vae.config.shift_factor)
|
||||
else:
|
||||
latents = latents / self.vae.config.scaling_factor
|
||||
|
||||
latents = latents / self.vae.config.scaling_factor # 1, 16, 32, 90, 160
|
||||
|
||||
if only_save_latent:
|
||||
latents = latents.cpu().float()
|
||||
self.maybe_free_model_hooks()
|
||||
return latents
|
||||
|
||||
with torch.autocast(device_type="cuda",
|
||||
dtype=vae_dtype,
|
||||
enabled=vae_autocast_enabled):
|
||||
@@ -993,7 +999,6 @@ class HunyuanVideoPipeline(DiffusionPipeline):
|
||||
|
||||
if expand_temporal_dim or image.shape[2] == 1:
|
||||
image = image.squeeze(2)
|
||||
|
||||
else:
|
||||
image = latents
|
||||
|
||||
|
||||
@@ -369,6 +369,7 @@ class HunyuanVideoSampler(Inference):
|
||||
embedded_guidance_scale=None,
|
||||
batch_size=1,
|
||||
num_videos_per_prompt=1,
|
||||
only_save_latent=False,
|
||||
**kwargs,
|
||||
):
|
||||
"""
|
||||
@@ -524,7 +525,10 @@ class HunyuanVideoSampler(Inference):
|
||||
vae_ver=self.args.vae,
|
||||
enable_tiling=self.args.vae_tiling,
|
||||
enable_vae_sp=self.args.vae_sp,
|
||||
only_save_latent=only_save_latent,
|
||||
)[0]
|
||||
if only_save_latent:
|
||||
return samples
|
||||
out_dict["samples"] = samples
|
||||
out_dict["prompts"] = prompt
|
||||
|
||||
|
||||
@@ -0,0 +1,286 @@
|
||||
import argparse
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
import imageio
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.distributed as dist
|
||||
import torchvision
|
||||
from einops import rearrange
|
||||
import json
|
||||
from fastvideo.models.hunyuan.inference import HunyuanVideoSampler
|
||||
from fastvideo.utils.parallel_states import (
|
||||
initialize_sequence_parallel_state, nccl_info)
|
||||
|
||||
|
||||
def initialize_distributed():
|
||||
local_rank = int(os.getenv("RANK", 0))
|
||||
world_size = int(os.getenv("WORLD_SIZE", 1))
|
||||
print("world_size", world_size)
|
||||
torch.cuda.set_device(local_rank)
|
||||
dist.init_process_group(backend="nccl",
|
||||
init_method="env://",
|
||||
world_size=world_size,
|
||||
rank=local_rank)
|
||||
initialize_sequence_parallel_state(world_size)
|
||||
|
||||
|
||||
def main(args):
|
||||
print(nccl_info.sp_size)
|
||||
print(args)
|
||||
models_root_path = Path(args.model_path)
|
||||
if not models_root_path.exists():
|
||||
raise ValueError(f"`models_root` not exists: {models_root_path}")
|
||||
|
||||
save_path = args.output_path
|
||||
os.makedirs(os.path.dirname(save_path), exist_ok=True)
|
||||
|
||||
# Handle both JSON and TXT files
|
||||
prompts = []
|
||||
if args.prompt_json.endswith('.json'):
|
||||
with open(args.prompt_json) as f:
|
||||
prompt_json = json.load(f)
|
||||
prompts = [(item["caption"], item["latent_path"]) for item in prompt_json]
|
||||
else:
|
||||
with open(args.prompt_json) as f:
|
||||
for line in f:
|
||||
prompt = line.strip()
|
||||
if prompt:
|
||||
# Use first 50 chars of prompt as filename
|
||||
file_name = "".join(c for c in prompt[:50] if c.isalnum() or c.isspace())
|
||||
file_name = file_name.strip().replace(" ", "_")
|
||||
prompts.append((prompt, file_name))
|
||||
|
||||
# Filter for unprocessed prompts in latent-only mode
|
||||
if args.only_save_latent:
|
||||
filtered_prompts = []
|
||||
for prompt, file_name in prompts:
|
||||
latent_path = os.path.join(args.output_path, "latent", f"{file_name}.pt")
|
||||
if not os.path.exists(latent_path):
|
||||
filtered_prompts.append((prompt, file_name))
|
||||
else:
|
||||
print(f"Latent file exists for {file_name}, skipping...")
|
||||
|
||||
prompts = filtered_prompts
|
||||
if not prompts:
|
||||
print("All prompts processed. Exiting...")
|
||||
return
|
||||
|
||||
hunyuan_video_sampler = HunyuanVideoSampler.from_pretrained(
|
||||
models_root_path, args=args)
|
||||
args = hunyuan_video_sampler.args
|
||||
|
||||
prompt_start_end_idx = [int(i) for i in args.prompt_start_end_idx.split(",")]
|
||||
prompts = prompts[prompt_start_end_idx[0]:prompt_start_end_idx[1]]
|
||||
|
||||
for prompt, file_name in prompts:
|
||||
outputs = hunyuan_video_sampler.predict(
|
||||
prompt=prompt,
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
video_length=args.num_frames,
|
||||
seed=args.seed,
|
||||
negative_prompt=args.neg_prompt,
|
||||
infer_steps=args.num_inference_steps,
|
||||
guidance_scale=args.guidance_scale,
|
||||
num_videos_per_prompt=args.num_videos,
|
||||
flow_shift=args.flow_shift,
|
||||
batch_size=args.batch_size,
|
||||
embedded_guidance_scale=args.embedded_cfg_scale,
|
||||
only_save_latent=args.only_save_latent,
|
||||
)
|
||||
if args.only_save_latent:
|
||||
os.makedirs(os.path.join(args.output_path, "latent"), exist_ok=True)
|
||||
latent_path = os.path.join(args.output_path, "latent", f"{file_name}.pt")
|
||||
torch.save(outputs.to(torch.bfloat16), latent_path)
|
||||
else:
|
||||
videos = rearrange(outputs["samples"], "b c t h w -> t b c h w")
|
||||
outputs = []
|
||||
for x in videos:
|
||||
x = torchvision.utils.make_grid(x, nrow=6)
|
||||
x = x.transpose(0, 1).transpose(1, 2).squeeze(-1)
|
||||
outputs.append((x * 255).numpy().astype(np.uint8))
|
||||
os.makedirs(os.path.dirname(args.output_path), exist_ok=True)
|
||||
imageio.mimsave(os.path.join(args.output_path, f"{file_name}.mp4"),
|
||||
outputs,
|
||||
fps=args.fps)
|
||||
|
||||
#
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
|
||||
# Basic parameters
|
||||
parser.add_argument("--prompt_json", type=str, required=True, help="prompt file for inference")
|
||||
parser.add_argument("--num_frames", type=int, default=16)
|
||||
parser.add_argument("--height", type=int, default=256)
|
||||
parser.add_argument("--width", type=int, default=256)
|
||||
parser.add_argument("--num_inference_steps", type=int, default=50)
|
||||
parser.add_argument("--model_path", type=str, default="data/hunyuan")
|
||||
parser.add_argument("--output_path", type=str, default="./outputs/video")
|
||||
parser.add_argument("--fps", type=int, default=24)
|
||||
|
||||
# Additional parameters
|
||||
parser.add_argument(
|
||||
"--prompt_start_end_idx",
|
||||
type=str,
|
||||
default="0,99",
|
||||
help=" 0 stands for starting layer index, 9 for stride",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--only_save_latent",
|
||||
type=bool,
|
||||
default=False,
|
||||
help="Whether save latent or not.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--denoise-type",
|
||||
type=str,
|
||||
default="flow",
|
||||
help="Denoise type for noised inputs.",
|
||||
)
|
||||
parser.add_argument("--seed",
|
||||
type=int,
|
||||
default=None,
|
||||
help="Seed for evaluation.")
|
||||
parser.add_argument("--neg_prompt",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Negative prompt for sampling.")
|
||||
parser.add_argument(
|
||||
"--guidance_scale",
|
||||
type=float,
|
||||
default=1.0,
|
||||
help="Classifier free guidance scale.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--embedded_cfg_scale",
|
||||
type=float,
|
||||
default=6.0,
|
||||
help="Embedded classifier free guidance scale.",
|
||||
)
|
||||
parser.add_argument("--flow_shift",
|
||||
type=int,
|
||||
default=7,
|
||||
help="Flow shift parameter.")
|
||||
parser.add_argument("--batch_size",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Batch size for inference.")
|
||||
parser.add_argument(
|
||||
"--num_videos",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Number of videos to generate per prompt.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--load-key",
|
||||
type=str,
|
||||
default="module",
|
||||
help=
|
||||
"Key to load the model states. 'module' for the main model, 'ema' for the EMA model.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--use-cpu-offload",
|
||||
action="store_true",
|
||||
help="Use CPU offload for the model load.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--dit-weight",
|
||||
type=str,
|
||||
default=
|
||||
"data/hunyuan/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--reproduce",
|
||||
action="store_true",
|
||||
help=
|
||||
"Enable reproducibility by setting random seeds and deterministic algorithms.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--disable-autocast",
|
||||
action="store_true",
|
||||
help=
|
||||
"Disable autocast for denoising loop and vae decoding in pipeline sampling.",
|
||||
)
|
||||
|
||||
# Flow Matching
|
||||
parser.add_argument(
|
||||
"--flow-reverse",
|
||||
action="store_true",
|
||||
help="If reverse, learning/sampling from t=1 -> t=0.",
|
||||
)
|
||||
parser.add_argument("--flow-solver",
|
||||
type=str,
|
||||
default="euler",
|
||||
help="Solver for flow matching.")
|
||||
parser.add_argument(
|
||||
"--use-linear-quadratic-schedule",
|
||||
action="store_true",
|
||||
help=
|
||||
"Use linear quadratic schedule for flow matching. Following MovieGen (https://ai.meta.com/static-resource/movie-gen-research-paper)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--linear-schedule-end",
|
||||
type=int,
|
||||
default=25,
|
||||
help="End step for linear quadratic schedule for flow matching.",
|
||||
)
|
||||
|
||||
# Model parameters
|
||||
parser.add_argument("--model", type=str, default="HYVideo-T/2-cfgdistill")
|
||||
parser.add_argument("--latent-channels", type=int, default=16)
|
||||
parser.add_argument("--precision",
|
||||
type=str,
|
||||
default="bf16",
|
||||
choices=["fp32", "fp16", "bf16"])
|
||||
parser.add_argument("--rope-theta",
|
||||
type=int,
|
||||
default=256,
|
||||
help="Theta used in RoPE.")
|
||||
|
||||
parser.add_argument("--vae", type=str, default="884-16c-hy")
|
||||
parser.add_argument("--vae-precision",
|
||||
type=str,
|
||||
default="fp16",
|
||||
choices=["fp32", "fp16", "bf16"])
|
||||
parser.add_argument("--vae-tiling", action="store_true", default=True)
|
||||
parser.add_argument("--vae-sp", action="store_true", default=False)
|
||||
|
||||
parser.add_argument("--text-encoder", type=str, default="llm")
|
||||
parser.add_argument(
|
||||
"--text-encoder-precision",
|
||||
type=str,
|
||||
default="fp16",
|
||||
choices=["fp32", "fp16", "bf16"],
|
||||
)
|
||||
parser.add_argument("--text-states-dim", type=int, default=4096)
|
||||
parser.add_argument("--text-len", type=int, default=256)
|
||||
parser.add_argument("--tokenizer", type=str, default="llm")
|
||||
parser.add_argument("--prompt-template",
|
||||
type=str,
|
||||
default="dit-llm-encode")
|
||||
parser.add_argument("--prompt-template-video",
|
||||
type=str,
|
||||
default="dit-llm-encode-video")
|
||||
parser.add_argument("--hidden-state-skip-layer", type=int, default=2)
|
||||
parser.add_argument("--apply-final-norm", action="store_true")
|
||||
|
||||
parser.add_argument("--text-encoder-2", type=str, default="clipL")
|
||||
parser.add_argument(
|
||||
"--text-encoder-precision-2",
|
||||
type=str,
|
||||
default="fp16",
|
||||
choices=["fp32", "fp16", "bf16"],
|
||||
)
|
||||
parser.add_argument("--text-states-dim-2", type=int, default=768)
|
||||
parser.add_argument("--tokenizer-2", type=str, default="clipL")
|
||||
parser.add_argument("--text-len-2", type=int, default=77)
|
||||
|
||||
args = parser.parse_args()
|
||||
# process for vae sequence parallel
|
||||
if args.vae_sp and not args.vae_tiling:
|
||||
raise ValueError(
|
||||
"Currently enabling vae_sp requires enabling vae_tiling, please set --vae-tiling to True."
|
||||
)
|
||||
main(args)
|
||||
+34
-38
@@ -1,16 +1,16 @@
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
|
||||
import torch
|
||||
import torch.distributed as dist
|
||||
from diffusers.utils import export_to_video
|
||||
|
||||
import torch
|
||||
from fastvideo.models.mochi_hf.pipeline_mochi import MochiPipeline
|
||||
import os
|
||||
from diffusers.utils import export_to_video
|
||||
import argparse
|
||||
|
||||
|
||||
def generate_video_and_latent(pipe, prompt, height, width, num_frames,
|
||||
num_inference_steps, guidance_scale):
|
||||
def generate_video_and_latent(
|
||||
pipe, prompt, height, width, num_frames, num_inference_steps, guidance_scale
|
||||
):
|
||||
# Set the random seed for reproducibility
|
||||
generator = torch.Generator("cuda").manual_seed(12345)
|
||||
# Generate videos from the input prompt
|
||||
@@ -25,8 +25,7 @@ def generate_video_and_latent(pipe, prompt, height, width, num_frames,
|
||||
output_type="latent_and_video",
|
||||
)
|
||||
# prompt_embed has negative prompt at index 0
|
||||
return noise[0], video[0], latent[0], prompt_embed[
|
||||
1], prompt_attention_mask[1]
|
||||
return noise[0], video[0], latent[0], prompt_embed[1], prompt_attention_mask[1]
|
||||
|
||||
# return dummy tensor to debug first
|
||||
# return torch.zeros(1, 3, 480, 848), torch.zeros(1, 256, 16, 16)
|
||||
@@ -40,22 +39,19 @@ if __name__ == "__main__":
|
||||
parser.add_argument("--num_inference_steps", type=int, default=64)
|
||||
parser.add_argument("--guidance_scale", type=float, default=4.5)
|
||||
parser.add_argument("--model_path", type=str, default="data/mochi")
|
||||
parser.add_argument("--prompt_path",
|
||||
type=str,
|
||||
default="data/dummyVid/videos2caption.json")
|
||||
parser.add_argument("--dataset_output_dir",
|
||||
type=str,
|
||||
default="data/dummySynthetic")
|
||||
parser.add_argument(
|
||||
"--prompt_path", type=str, default="data/dummyVid/videos2caption.json"
|
||||
)
|
||||
parser.add_argument("--dataset_output_dir", type=str, default="data/dummySynthetic")
|
||||
args = parser.parse_args()
|
||||
|
||||
local_rank = int(os.getenv("RANK", 0))
|
||||
world_size = int(os.getenv("WORLD_SIZE", 1))
|
||||
print("world_size", world_size, "local rank", local_rank)
|
||||
torch.cuda.set_device(local_rank)
|
||||
dist.init_process_group(backend="nccl",
|
||||
init_method="env://",
|
||||
world_size=world_size,
|
||||
rank=local_rank)
|
||||
dist.init_process_group(
|
||||
backend="nccl", init_method="env://", world_size=world_size, rank=local_rank
|
||||
)
|
||||
|
||||
if not isinstance(args.prompt_path, list):
|
||||
args.prompt_path = [args.prompt_path]
|
||||
@@ -63,8 +59,7 @@ if __name__ == "__main__":
|
||||
text_prompt = open(args.prompt_path[0], "r").readlines()
|
||||
text_prompt = [i.strip() for i in text_prompt]
|
||||
|
||||
pipe = MochiPipeline.from_pretrained(args.model_path,
|
||||
torch_dtype=torch.bfloat16)
|
||||
pipe = MochiPipeline.from_pretrained(args.model_path, torch_dtype=torch.bfloat16)
|
||||
pipe.enable_vae_tiling()
|
||||
pipe.enable_model_cpu_offload(gpu_id=local_rank)
|
||||
# make dir if not exist
|
||||
@@ -73,10 +68,10 @@ if __name__ == "__main__":
|
||||
os.makedirs(os.path.join(args.dataset_output_dir, "noise"), exist_ok=True)
|
||||
os.makedirs(os.path.join(args.dataset_output_dir, "video"), exist_ok=True)
|
||||
os.makedirs(os.path.join(args.dataset_output_dir, "latent"), exist_ok=True)
|
||||
os.makedirs(os.path.join(args.dataset_output_dir, "prompt_embed"),
|
||||
exist_ok=True)
|
||||
os.makedirs(os.path.join(args.dataset_output_dir, "prompt_attention_mask"),
|
||||
exist_ok=True)
|
||||
os.makedirs(os.path.join(args.dataset_output_dir, "prompt_embed"), exist_ok=True)
|
||||
os.makedirs(
|
||||
os.path.join(args.dataset_output_dir, "prompt_attention_mask"), exist_ok=True
|
||||
)
|
||||
data = []
|
||||
for i, prompt in enumerate(text_prompt):
|
||||
if i % world_size != local_rank:
|
||||
@@ -98,17 +93,17 @@ if __name__ == "__main__":
|
||||
)
|
||||
# save latent
|
||||
video_name = str(i)
|
||||
noise_path = os.path.join(args.dataset_output_dir, "noise",
|
||||
video_name + ".pt")
|
||||
latent_path = os.path.join(args.dataset_output_dir, "latent",
|
||||
video_name + ".pt")
|
||||
prompt_embed_path = os.path.join(args.dataset_output_dir,
|
||||
"prompt_embed", video_name + ".pt")
|
||||
video_path = os.path.join(args.dataset_output_dir, "video",
|
||||
video_name + ".mp4")
|
||||
prompt_attention_mask_path = os.path.join(args.dataset_output_dir,
|
||||
"prompt_attention_mask",
|
||||
video_name + ".pt")
|
||||
noise_path = os.path.join(args.dataset_output_dir, "noise", video_name + ".pt")
|
||||
latent_path = os.path.join(
|
||||
args.dataset_output_dir, "latent", video_name + ".pt"
|
||||
)
|
||||
prompt_embed_path = os.path.join(
|
||||
args.dataset_output_dir, "prompt_embed", video_name + ".pt"
|
||||
)
|
||||
video_path = os.path.join(args.dataset_output_dir, "video", video_name + ".mp4")
|
||||
prompt_attention_mask_path = os.path.join(
|
||||
args.dataset_output_dir, "prompt_attention_mask", video_name + ".pt"
|
||||
)
|
||||
# save latent
|
||||
torch.save(noise, noise_path)
|
||||
torch.save(latent, latent_path)
|
||||
@@ -132,6 +127,7 @@ if __name__ == "__main__":
|
||||
# save json
|
||||
if local_rank == 0:
|
||||
all_data = [item for sublist in gathered_data for item in sublist]
|
||||
with open(os.path.join(args.dataset_output_dir, "videos2caption.json"),
|
||||
"w") as f:
|
||||
with open(
|
||||
os.path.join(args.dataset_output_dir, "videos2caption.json"), "w"
|
||||
) as f:
|
||||
json.dump(all_data, f, indent=4)
|
||||
@@ -0,0 +1,62 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=fastvideo
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log/slurm_lora_1e4.out
|
||||
#SBATCH --error=slurm_log/slurm_lora_1e4.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
export MASTER_PORT=29803
|
||||
torchrun --nnodes 1 --nproc_per_node 8 --master_port $MASTER_PORT \
|
||||
fastvideo/train.py \
|
||||
--seed 1024 \
|
||||
--pretrained_model_name_or_path ~/data/hunyuan_diffusers \
|
||||
--model_type hunyuan_hf \
|
||||
--cache_dir data/.cache \
|
||||
--data_json_path ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--validation_prompt_dir ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/validation \
|
||||
--gradient_checkpointing \
|
||||
--train_batch_size 8 \
|
||||
--num_latent_t 32 \
|
||||
--sp_size 8 \
|
||||
--train_sp_batch_size 1 \
|
||||
--dataloader_num_workers 4 \
|
||||
--gradient_accumulation_steps 1 \
|
||||
--max_train_steps 8000 \
|
||||
--learning_rate 1e-4 \
|
||||
--mixed_precision bf16 \
|
||||
--checkpointing_steps 100 \
|
||||
--validation_steps 100 \
|
||||
--validation_sampling_steps 50 \
|
||||
--checkpoints_total_limit 3 \
|
||||
--allow_tf32 \
|
||||
--ema_start_step 0 \
|
||||
--cfg 0.0 \
|
||||
--ema_decay 0.999 \
|
||||
--log_validation \
|
||||
--output_dir data/outputs/SBA_lora_1e4_r32 \
|
||||
--tracker_project_name SBA \
|
||||
--num_frames 125 \
|
||||
--num_width 1280 \
|
||||
--num_height 768 \
|
||||
--validation_guidance_scale "1.0" \
|
||||
--shift 7 \
|
||||
--use_lora \
|
||||
--lora_rank 32 \
|
||||
--lora_alpha 32
|
||||
@@ -0,0 +1,100 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=fv_syn_hunyuan
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log/slurm_ori_ft_syn.out
|
||||
#SBATCH --error=slurm_log/slurm_ori_ft_syn.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
export MASTER_PORT=29803
|
||||
torchrun --nnodes 1 --nproc_per_node 6 \
|
||||
fastvideo/train.py \
|
||||
--seed 42 \
|
||||
--pretrained_model_name_or_path data/hunyuan \
|
||||
--dit_model_name_or_path data/hunyuan/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt\
|
||||
--model_type "hunyuan" \
|
||||
--cache_dir data/.cache \
|
||||
--data_json_path ../FastVideo-Internal/data/synthetic_MixKit/videos2caption.json \
|
||||
--validation_prompt_dir ../FastVideo-Internal/data/synthetic_MixKit/validation \
|
||||
--gradient_checkpointing \
|
||||
--train_batch_size=1 \
|
||||
--num_latent_t 30 \
|
||||
--sp_size 6 \
|
||||
--train_sp_batch_size 1 \
|
||||
--dataloader_num_workers 4 \
|
||||
--gradient_accumulation_steps=2 \
|
||||
--max_train_steps=2000 \
|
||||
--learning_rate=1e-5 \
|
||||
--mixed_precision=bf16 \
|
||||
--checkpointing_steps=200 \
|
||||
--validation_steps 100 \
|
||||
--validation_sampling_steps 50 \
|
||||
--checkpoints_total_limit 3 \
|
||||
--allow_tf32 \
|
||||
--ema_start_step 0 \
|
||||
--cfg 0.0 \
|
||||
--ema_decay 0.999 \
|
||||
--log_validation \
|
||||
--output_dir=data/outputs/SBA_hunyuan_ori_ft_1e5 \
|
||||
--tracker_project_name SBA-Syn \
|
||||
--num_frames 117 \
|
||||
--num_height 768 \
|
||||
--num_width 1280 \
|
||||
--shift 7 \
|
||||
--validation_guidance_scale "1.0" \
|
||||
--training_guidance "6.0" \
|
||||
--run_name "SBA_hunyuan_ori_ft_1e5_g6" \
|
||||
|
||||
torchrun --nnodes 1 --nproc_per_node 6 \
|
||||
fastvideo/train.py \
|
||||
--seed 42 \
|
||||
--pretrained_model_name_or_path data/hunyuan \
|
||||
--dit_model_name_or_path data/hunyuan/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt\
|
||||
--model_type "hunyuan" \
|
||||
--cache_dir data/.cache \
|
||||
--data_json_path ../FastVideo-Internal/data/synthetic_MixKit/videos2caption.json \
|
||||
--validation_prompt_dir ../FastVideo-Internal/data/synthetic_MixKit/validation \
|
||||
--gradient_checkpointing \
|
||||
--train_batch_size=1 \
|
||||
--num_latent_t 30 \
|
||||
--sp_size 6 \
|
||||
--train_sp_batch_size 1 \
|
||||
--dataloader_num_workers 4 \
|
||||
--gradient_accumulation_steps=2 \
|
||||
--max_train_steps=2000 \
|
||||
--learning_rate=1e-5 \
|
||||
--mixed_precision=bf16 \
|
||||
--checkpointing_steps=200 \
|
||||
--validation_steps 100 \
|
||||
--validation_sampling_steps 50 \
|
||||
--checkpoints_total_limit 3 \
|
||||
--allow_tf32 \
|
||||
--ema_start_step 0 \
|
||||
--cfg 0.0 \
|
||||
--ema_decay 0.999 \
|
||||
--log_validation \
|
||||
--output_dir=data/outputs/SBA_hunyuan_ori_ft_1e5 \
|
||||
--tracker_project_name SBA-Syn \
|
||||
--num_frames 117 \
|
||||
--num_height 768 \
|
||||
--num_width 1280 \
|
||||
--shift 7 \
|
||||
--validation_guidance_scale "1.0" \
|
||||
--training_guidance "1.0" \
|
||||
--run_name "SBA_hunyuan_ori_ft_1e5_g1" \
|
||||
@@ -0,0 +1,12 @@
|
||||
import json
|
||||
src_json = "../HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json"
|
||||
des_json = "./videos2caption.json"
|
||||
syn_num = 8
|
||||
with open(src_json) as f:
|
||||
data = json.load(f)
|
||||
data = data[:syn_num]
|
||||
# from IPython import embed
|
||||
# embed()
|
||||
|
||||
with open(des_json, "w") as f2:
|
||||
json.dump(data, f2, indent=4, ensure_ascii=False)
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=hyhf-lora-5e6-r32
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=4
|
||||
#SBATCH --mem=512G
|
||||
#SBATCH --output=slurm_log/slurm.out
|
||||
#SBATCH --error=slurm_log/slurm.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
cd ~/yongqi/FastVideo
|
||||
export MASTER_PORT=29803
|
||||
torchrun --nnodes 1 --nproc_per_node 8 --master_port $MASTER_PORT \
|
||||
fastvideo/train.py \
|
||||
--seed 1024 \
|
||||
--pretrained_model_name_or_path ~/data/hunyuan_diffusers \
|
||||
--model_type hunyuan_hf \
|
||||
--cache_dir data/.cache \
|
||||
--data_json_path ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--validation_prompt_dir ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/validation \
|
||||
--gradient_checkpointing \
|
||||
--train_batch_size 4 \
|
||||
--num_latent_t 32 \
|
||||
--sp_size 8 \
|
||||
--train_sp_batch_size 1 \
|
||||
--dataloader_num_workers 4 \
|
||||
--gradient_accumulation_steps 1 \
|
||||
--max_train_steps 8000 \
|
||||
--learning_rate 5e-6 \
|
||||
--mixed_precision bf16 \
|
||||
--checkpointing_steps 200 \
|
||||
--validation_steps 100 \
|
||||
--validation_sampling_steps 50 \
|
||||
--checkpoints_total_limit 3 \
|
||||
--allow_tf32 \
|
||||
--ema_start_step 0 \
|
||||
--cfg 0.0 \
|
||||
--ema_decay 0.999 \
|
||||
--log_validation \
|
||||
--output_dir data/outputs/SBA_lora_5e6_r32 \
|
||||
--tracker_project_name SBA \
|
||||
--num_frames 125 \
|
||||
--num_width 1280 \
|
||||
--num_height 768 \
|
||||
--validation_guidance_scale "1.0" \
|
||||
--use_lora \
|
||||
--lora_rank 32 \
|
||||
--lora_alpha 64
|
||||
|
||||
echo "Run completed at:- "
|
||||
date
|
||||
@@ -0,0 +1,60 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-1
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-1.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-1.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=0
|
||||
num_promtps_all=200
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ./assets/prompt_vb.txt \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-10
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-10.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-10.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=2160
|
||||
num_promtps_all=240
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-11
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-11.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-11.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=2400
|
||||
num_promtps_all=240
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-2
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-2.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-2.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=198
|
||||
num_promtps_all=200
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ./assets/prompt_vb.txt \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-3
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-3.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-3.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=396
|
||||
num_promtps_all=200
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ./assets/prompt_vb.txt \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-4
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-4.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-4.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=594
|
||||
num_promtps_all=200
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ./assets/prompt_vb.txt \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-5
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-5.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-5.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=960
|
||||
num_promtps_all=240
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-6
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-6.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-6.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=1200
|
||||
num_promtps_all=240
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-7
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-7.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-7.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=1440
|
||||
num_promtps_all=240
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-8
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-8.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-8.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=1680
|
||||
num_promtps_all=240
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=syn-9
|
||||
#SBATCH --partition=mbzuai
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=8
|
||||
#SBATCH --mem=960G
|
||||
#SBATCH --output=slurm_log_last_dance/syn-9.out
|
||||
#SBATCH --error=slurm_log_last_dance/syn-9.err
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --time=72:00:00
|
||||
|
||||
conda init
|
||||
source ~/conda/miniconda/bin/activate
|
||||
PYTHON_VIRTUAL_ENVIRONMENT=fastvideo
|
||||
conda activate $PYTHON_VIRTUAL_ENVIRONMENT
|
||||
cd /mbz/users/hao.zhang/yongqi/FastVideo
|
||||
|
||||
echo " "
|
||||
echo " Number of nodes:= " $SLURM_JOB_NUM_NODES
|
||||
echo " GPUs per node:= " $SLURM_JOB_GPUS
|
||||
echo " Running on multiple nodes/GPU devices"
|
||||
echo ""
|
||||
echo " Run started at:- "
|
||||
date
|
||||
|
||||
# Calculate prompts per GPU
|
||||
prompt_start_idx_all=1920
|
||||
num_promtps_all=240
|
||||
num_gpus=8
|
||||
prompts_per_gpu=$((num_promtps_all / num_gpus))
|
||||
export MODEL_BASE=data/hunyuan
|
||||
|
||||
for gpu_id in $(seq 0 $((num_gpus-1))); do
|
||||
start_idx=$((prompt_start_idx_all + gpu_id * prompts_per_gpu))
|
||||
end_idx=$((start_idx + prompts_per_gpu))
|
||||
prompt_start_end_idx_single_gpu="${start_idx},${end_idx}"
|
||||
|
||||
CUDA_VISIBLE_DEVICES=$gpu_id python fastvideo/sample/generate_synthetic_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 6 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt_json ~/data/HD-MixKit-Hunyuan-Distill-125x768x1280/videos2caption.json \
|
||||
--seed 1024 \
|
||||
--output_path outputs_latents/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--only_save_latent True \
|
||||
--prompt_start_end_idx ${prompt_start_end_idx_single_gpu} &
|
||||
done
|
||||
|
||||
# Wait for all background jobs to complete
|
||||
wait
|
||||
@@ -18,3 +18,22 @@ torchrun --nnodes=1 --nproc_per_node=$num_gpus --master_port 29503 \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--vae-sp
|
||||
|
||||
num_gpus=6
|
||||
export MODEL_BASE=data/hunyuan
|
||||
torchrun --nnodes=1 --nproc_per_node=$num_gpus --master_port 29503 \
|
||||
fastvideo/sample/sample_t2v_hunyuan.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
--num_frames 117 \
|
||||
--num_inference_steps 50 \
|
||||
--guidance_scale 1 \
|
||||
--embedded_cfg_scale 8 \
|
||||
--flow_shift 7 \
|
||||
--flow-reverse \
|
||||
--prompt ./assets/prompt_mixkit.txt \
|
||||
--seed 1024 \
|
||||
--output_path outputs_video/mixkit-8/ \
|
||||
--model_path $MODEL_BASE \
|
||||
--dit-weight ${MODEL_BASE}/hunyuan-video-t2v-720p/transformers/mp_rank_00_model_states.pt \
|
||||
--vae-sp
|
||||
|
||||
Reference in New Issue
Block a user