Initial commit

This commit is contained in:
SHYuanBest
2026-03-04 03:31:47 +00:00
commit dd6a26ebaa
205 changed files with 45297 additions and 0 deletions
@@ -0,0 +1,14 @@
compute_environment: LOCAL_MACHINE
distributed_type: DEEPSPEED
deepspeed_config:
deepspeed_config_file: scripts/accelerate_configs/zero2.json
deepspeed_multinode_launcher: standard
fsdp_config: {}
machine_rank: 0
main_training_function: main
rdzv_backend: static
same_network: true
tpu_env: []
tpu_use_cluster: false
tpu_use_sudo: false
use_cpu: false
@@ -0,0 +1,14 @@
compute_environment: LOCAL_MACHINE
distributed_type: DEEPSPEED
deepspeed_config:
deepspeed_config_file: scripts/accelerate_configs/zero3.json
deepspeed_multinode_launcher: standard
fsdp_config: {}
machine_rank: 0
main_training_function: main
rdzv_backend: static
same_network: true
tpu_env: []
tpu_use_cluster: false
tpu_use_sudo: false
use_cpu: false
@@ -0,0 +1,28 @@
{
"_class_name": "UniPCMultistepScheduler",
"_diffusers_version": "0.33.0.dev0",
"beta_end": 0.02,
"beta_schedule": "linear",
"beta_start": 0.0001,
"disable_corrector": [],
"dynamic_thresholding_ratio": 0.995,
"final_sigmas_type": "zero",
"flow_shift": 3.0,
"lower_order_final": true,
"num_train_timesteps": 1000,
"predict_x0": true,
"prediction_type": "flow_prediction",
"rescale_betas_zero_snr": false,
"sample_max_value": 1.0,
"solver_order": 2,
"solver_p": null,
"solver_type": "bh2",
"steps_offset": 0,
"thresholding": false,
"timestep_spacing": "linspace",
"trained_betas": null,
"use_beta_sigmas": false,
"use_exponential_sigmas": false,
"use_flow_sigmas": true,
"use_karras_sigmas": false
}
+25
View File
@@ -0,0 +1,25 @@
{
"fp16": {
"enabled": false,
"loss_scale": 0,
"loss_scale_window": 1000,
"initial_scale_power": 16,
"hysteresis": 2,
"min_loss_scale": 1
},
"bf16": {
"enabled": "auto"
},
"communication_data_type": "fp32",
"gradient_clipping": 1.0,
"train_micro_batch_size_per_gpu": "auto",
"train_batch_size": "auto",
"gradient_accumulation_steps": "auto",
"zero_optimization": {
"stage": 2,
"overlap_comm": true,
"contiguous_gradients": true,
"reduce_bucket_size": 1e9,
"allgather_bucket_size": 536870912
}
}
+30
View File
@@ -0,0 +1,30 @@
{
"fp16": {
"enabled": false,
"loss_scale": 0,
"loss_scale_window": 1000,
"initial_scale_power": 16,
"hysteresis": 2,
"min_loss_scale": 1
},
"bf16": {
"enabled": "auto"
},
"communication_data_type": "fp32",
"gradient_clipping": 1.0,
"train_micro_batch_size_per_gpu": "auto",
"train_batch_size": "auto",
"gradient_accumulation_steps": "auto",
"zero_optimization": {
"stage": 3,
"overlap_comm": true,
"contiguous_gradients": true,
"stage3_gather_16bit_weights_on_model_save": true,
"sub_group_size": 536870912,
"reduce_bucket_size": 536870912,
"stage3_prefetch_bucket_size": 536870912,
"stage3_param_persistence_threshold": 524288,
"stage3_max_live_parameters": 536870912,
"stage3_max_reuse_distance": 536870912
}
}
+14
View File
@@ -0,0 +1,14 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Base" \
--transformer_path "BestWishYsh/Helios-Base" \
--sample_type "i2v" \
--image_path "example/wave.jpg" \
--prompt "A towering emerald wave surges forward, its crest curling with raw power and energy. Sunlight glints off the translucent water, illuminating the intricate textures and deep green hues within the wave’s body. A thick spray erupts from the breaking crest, casting a misty veil that dances above the churning surface. As the perspective widens, the immense scale of the wave becomes apparent, revealing the restless expanse of the ocean stretching beyond. The scene captures the ocean’s untamed beauty and relentless force, with every droplet and ripple shimmering in the light. The dynamic motion and vivid colors evoke both awe and respect for nature’s might." \
--guidance_scale 5.0 \
--enable_compile \
--output_folder "./output_helios/helios-base"
# --use_cfg_zero_star \
# --use_zero_init \
# --zero_steps 1 \
+13
View File
@@ -0,0 +1,13 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Base" \
--transformer_path "BestWishYsh/Helios-Base" \
--sample_type "t2v" \
--prompt "A vibrant tropical fish swimming gracefully among colorful coral reefs in a clear, turquoise ocean. The fish has bright blue and yellow scales with a small, distinctive orange spot on its side, its fins moving fluidly. The coral reefs are alive with a variety of marine life, including small schools of colorful fish and sea turtles gliding by. The water is crystal clear, allowing for a view of the sandy ocean floor below. The reef itself is adorned with a mix of hard and soft corals in shades of red, orange, and green. The photo captures the fish from a slightly elevated angle, emphasizing its lively movements and the vivid colors of its surroundings. A close-up shot with dynamic movement." \
--guidance_scale 5.0 \
--enable_compile \
--output_folder "./output_helios/helios-base"
# --use_cfg_zero_star \
# --use_zero_init \
# --zero_steps 1 \
+14
View File
@@ -0,0 +1,14 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Base" \
--transformer_path "BestWishYsh/Helios-Base" \
--sample_type "v2v" \
--video_path "example/car.mp4" \
--prompt "A bright yellow Lamborghini Huracn Tecnica speeds along a curving mountain road, surrounded by lush green trees under a partly cloudy sky. The car's sleek design and vibrant color stand out against the natural backdrop, emphasizing its dynamic movement. The road curves gently, with a guardrail visible on one side, adding depth to the scene. The motion blur captures the sense of speed and energy, creating a thrilling and exhilarating atmosphere. A front-facing shot from a slightly elevated angle, highlighting the car's aggressive stance and the surrounding greenery." \
--guidance_scale 5.0 \
--enable_compile \
--output_folder "./output_helios/helios-base"
# --use_cfg_zero_star \
# --use_zero_init \
# --zero_steps 1 \
+16
View File
@@ -0,0 +1,16 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Distilled" \
--transformer_path "BestWishYsh/Helios-Distilled" \
--sample_type "i2v" \
--image_path "example/wave.jpg" \
--prompt "A towering emerald wave surges forward, its crest curling with raw power and energy. Sunlight glints off the translucent water, illuminating the intricate textures and deep green hues within the wave’s body. A thick spray erupts from the breaking crest, casting a misty veil that dances above the churning surface. As the perspective widens, the immense scale of the wave becomes apparent, revealing the restless expanse of the ocean stretching beyond. The scene captures the ocean’s untamed beauty and relentless force, with every droplet and ripple shimmering in the light. The dynamic motion and vivid colors evoke both awe and respect for nature’s might." \
--num_frames 240 \
--guidance_scale 1.0 \
--is_enable_stage2 \
--pyramid_num_inference_steps_list 2 2 2 \
--is_amplify_first_chunk \
--enable_compile \
--output_folder "./output_helios/helios-distilled"
# --pyramid_num_inference_steps_list 1 1 1 \
+15
View File
@@ -0,0 +1,15 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Distilled" \
--transformer_path "BestWishYsh/Helios-Distilled" \
--sample_type "t2v" \
--prompt "A vibrant tropical fish swimming gracefully among colorful coral reefs in a clear, turquoise ocean. The fish has bright blue and yellow scales with a small, distinctive orange spot on its side, its fins moving fluidly. The coral reefs are alive with a variety of marine life, including small schools of colorful fish and sea turtles gliding by. The water is crystal clear, allowing for a view of the sandy ocean floor below. The reef itself is adorned with a mix of hard and soft corals in shades of red, orange, and green. The photo captures the fish from a slightly elevated angle, emphasizing its lively movements and the vivid colors of its surroundings. A close-up shot with dynamic movement." \
--num_frames 240 \
--guidance_scale 1.0 \
--is_enable_stage2 \
--pyramid_num_inference_steps_list 2 2 2 \
--is_amplify_first_chunk \
--enable_compile \
--output_folder "./output_helios/helios-distilled"
# --pyramid_num_inference_steps_list 1 1 1 \
+16
View File
@@ -0,0 +1,16 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Distilled" \
--transformer_path "BestWishYsh/Helios-Distilled" \
--sample_type "v2v" \
--video_path "example/car.mp4" \
--prompt "A bright yellow Lamborghini Huracn Tecnica speeds along a curving mountain road, surrounded by lush green trees under a partly cloudy sky. The car's sleek design and vibrant color stand out against the natural backdrop, emphasizing its dynamic movement. The road curves gently, with a guardrail visible on one side, adding depth to the scene. The motion blur captures the sense of speed and energy, creating a thrilling and exhilarating atmosphere. A front-facing shot from a slightly elevated angle, highlighting the car's aggressive stance and the surrounding greenery." \
--num_frames 240 \
--guidance_scale 1.0 \
--is_enable_stage2 \
--pyramid_num_inference_steps_list 2 2 2 \
--is_amplify_first_chunk \
--enable_compile \
--output_folder "./output_helios/helios-distilled"
# --pyramid_num_inference_steps_list 1 1 1 \
+16
View File
@@ -0,0 +1,16 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Mid" \
--transformer_path "BestWishYsh/Helios-Mid" \
--sample_type "i2v" \
--image_path "example/wave.jpg" \
--prompt "A towering emerald wave surges forward, its crest curling with raw power and energy. Sunlight glints off the translucent water, illuminating the intricate textures and deep green hues within the wave’s body. A thick spray erupts from the breaking crest, casting a misty veil that dances above the churning surface. As the perspective widens, the immense scale of the wave becomes apparent, revealing the restless expanse of the ocean stretching beyond. The scene captures the ocean’s untamed beauty and relentless force, with every droplet and ripple shimmering in the light. The dynamic motion and vivid colors evoke both awe and respect for nature’s might." \
--guidance_scale 5.0 \
--is_enable_stage2 \
--pyramid_num_inference_steps_list 20 20 20 \
--use_zero_init \
--zero_steps 1 \
--enable_compile \
--output_folder "./output_helios/helios-mid"
# --pyramid_num_inference_steps_list 17 17 17 \
+15
View File
@@ -0,0 +1,15 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Mid" \
--transformer_path "BestWishYsh/Helios-Mid" \
--sample_type "t2v" \
--prompt "A vibrant tropical fish swimming gracefully among colorful coral reefs in a clear, turquoise ocean. The fish has bright blue and yellow scales with a small, distinctive orange spot on its side, its fins moving fluidly. The coral reefs are alive with a variety of marine life, including small schools of colorful fish and sea turtles gliding by. The water is crystal clear, allowing for a view of the sandy ocean floor below. The reef itself is adorned with a mix of hard and soft corals in shades of red, orange, and green. The photo captures the fish from a slightly elevated angle, emphasizing its lively movements and the vivid colors of its surroundings. A close-up shot with dynamic movement." \
--guidance_scale 5.0 \
--is_enable_stage2 \
--pyramid_num_inference_steps_list 20 20 20 \
--use_zero_init \
--zero_steps 1 \
--enable_compile \
--output_folder "./output_helios/helios-mid"
# --pyramid_num_inference_steps_list 17 17 17 \
+16
View File
@@ -0,0 +1,16 @@
CUDA_VISIBLE_DEVICES=0 python infer_helios.py \
--base_model_path "BestWishYsh/Helios-Mid" \
--transformer_path "BestWishYsh/Helios-Mid" \
--sample_type "v2v" \
--video_path "example/car.mp4" \
--prompt "A bright yellow Lamborghini Huracn Tecnica speeds along a curving mountain road, surrounded by lush green trees under a partly cloudy sky. The car's sleek design and vibrant color stand out against the natural backdrop, emphasizing its dynamic movement. The road curves gently, with a guardrail visible on one side, adding depth to the scene. The motion blur captures the sense of speed and energy, creating a thrilling and exhilarating atmosphere. A front-facing shot from a slightly elevated angle, highlighting the car's aggressive stance and the surrounding greenery." \
--guidance_scale 5.0 \
--is_enable_stage2 \
--pyramid_num_inference_steps_list 20 20 20 \
--use_zero_init \
--zero_steps 1 \
--enable_compile \
--output_folder "./output_helios/helios-mid"
# --pyramid_num_inference_steps_list 17 17 17 \
+65
View File
@@ -0,0 +1,65 @@
import yaml
def compare_yaml(file1_path, file2_path):
with open(file1_path, "r") as f1:
yaml1 = yaml.safe_load(f1)
with open(file2_path, "r") as f2:
yaml2 = yaml.safe_load(f2)
missing_keys = []
different_values = []
compare_dict(yaml1, yaml2, "", missing_keys, different_values)
print("=" * 60)
print("Missing Keys")
print("=" * 60)
if missing_keys:
for diff in missing_keys:
print(diff)
else:
print("None")
print("\n" + "=" * 60)
print("Different Values")
print("=" * 60)
if different_values:
for diff in different_values:
print(diff)
else:
print("None")
print("\n" + "=" * 60)
print(f"Total: {len(missing_keys)} missing keys, {len(different_values)} different values")
print("=" * 60)
def compare_dict(dict1, dict2, path, missing_keys, different_values):
all_keys = set(dict1.keys()) | set(dict2.keys())
for key in all_keys:
current_path = f"{path}.{key}" if path else key
if key not in dict2:
missing_keys.append(f"[{current_path}] Only in file1: {dict1[key]}")
elif key not in dict1:
missing_keys.append(f"[{current_path}] Only in file2: {dict2[key]}")
else:
val1, val2 = dict1[key], dict2[key]
if isinstance(val1, dict) and isinstance(val2, dict):
compare_dict(val1, val2, current_path, missing_keys, different_values)
elif isinstance(val1, list) and isinstance(val2, list):
if val1 != val2:
different_values.append(f"[{current_path}]\n File1: {val1}\n File2: {val2}")
elif val1 != val2:
different_values.append(f"[{current_path}]\n File1: {val1}\n File2: {val2}")
if __name__ == "__main__":
compare_yaml(
"configs/stage_1_init.yaml",
"configs/stage_1_post.yaml",
)
+178
View File
@@ -0,0 +1,178 @@
output_dir: ablation_stage_1_init
logging_dir: logs
seed: 43
report_to:
tracker_name: Wan-Train
wandb_name: ablation_stage_1_init
report_to: wandb
data_config:
# ---- Base ----
use_shuffle: true
pin_memory: true
persistent_workers: true
force_rebuild: true
single_res: true
single_height: 384
single_width: 640
dataloader_num_workers: 8
prefetch_factor: 2
caption_dropout_p: 0
id_token: ""
instance_data_root:
- "demo_data/ultravideo-long"
# ---- Stage 1 ----
use_stage1_dataset: true
model_config:
# ---- Path ----
pretrained_model_name_or_path: "BestWishYsh/Helios-Base"
transformer_model_name_or_path: "Wan-AI/Wan2.1-T2V-14B-Diffusers"
load_checkpoints_custom: false
# load_model_path:
load_dcp: false
# load_dcp_path:
# ---- Vae ----
upcast_vae: true
enable_slicing: false
enable_tiling: false
# ---- Lora ----
lora_rank: 128
lora_alpha: 128.0
lora_dropout: 0.0
lora_layers: "all-linear"
# lora_target_modules:
# - to_k
# - to_q
# - to_v
# - to_out.0
# - ffn.net.0.proj
# - ffn.net.2
lora_exclude_modules:
- down
- up
# ---- Other ----
train_norm_layers: false
validation_config:
validation_steps: 500
validation_height: 384
validation_width: 640
validation_max_num_frames: 99
validation_prompts:
- "A stylish woman walks down a Tokyo street filled with warm glowing neon and animated city signage. She wears a black leather jacket, a long red dress, and black boots, and carries a black purse. She wears sunglasses and red lipstick. She walks confidently and casually. The street is damp and reflective, creating a mirror effect of the colorful lights. Many pedestrians walk about."
# - "Several giant wooly mammoths approach treading through a snowy meadow, their long wooly fur lightly blows in the wind as they walk, snow covered trees and dramatic snow capped mountains in the distance, mid afternoon light with wispy clouds and a sun high in the distance creates a warm glow, the low camera view is stunning capturing the large furry mammal with beautiful photography, depth of field."
# - "A movie trailer featuring the adventures of the 30 year old space man wearing a red wool knitted motorcycle helmet, blue sky, salt desert, cinematic style, shot on 35mm film, vivid colors."
validation_guidance_scale: 5.0
validation_latent_window_size:
- 9
num_validation_videos: 1
num_inference_steps: 50
training_config:
# ---- Environment ----
allow_tf32: false
gradient_checkpointing: true
enable_xformers_memory_efficient_attention: false
enable_npu_flash_attention: false
upcast_before_saving: false
offload: false
mixed_precision: "bf16"
# ---- Training Resource ----
max_train_steps: 1000000
train_batch_size: 2
gradient_accumulation_steps: 1
checkpointing_steps: 500
resume_from_checkpoint: "latest"
save_checkpoints_custom: false
# ---- Optimizer ----
learning_rate: 5e-5
lr_scheduler: "constant"
lr_warmup_steps: 500
optimizer: "adamw"
adam_beta1: 0.9
adam_beta2: 0.999
adam_weight_decay: 1e-04
adam_epsilon: 1e-08
max_grad_norm: 1.0
weighting_scheme: "logit_normal" # ["sigma_sqrt", "logit_normal", "mode", "cosmap", "none"]
logit_mean: 0.0
logit_std: 1.0
mode_scale: 1.29
# ---- Dynamic Shifting Parameters ----
use_dynamic_shifting: false
base_seq_len: 256
max_seq_len: 4096
base_shift: 0.5
max_shift: 1.15
# ---- VAE Decode Parameters ----
vae_decode_type: "default"
# ---- EMA Parameters ----
use_ema: false
use_ema_validation: false
ema_decay: 0.999
ema_start_step: 250
ema_zero3_port: 10543
ema_deepspeed_config_file: "scripts/accelerate_configs/zero3.json"
# ---- Stage 1 Parameters ----
is_enable_stage1: true
history_sizes:
- 16
- 2
- 1
latent_window_size:
# - 12
# - 10
- 9
# - 8
# - 6
# - 5
# - 4
# - 3
# - 2
# - 1
is_random_drop: true
random_drop_v2v_ratio: 0.4
random_drop_t2v_ratio: 0.4
#
corrupt_model_input: false
corrupt_mode_model_input: "noise"
corrupt_mode_prob_model_input: 0.9
is_frame_independent_corrupt_model_input: true
is_chunk_independent_corrupt_model_input: false
noise_corrupt_ratio_model_input: 0.33333333333333
noise_corrupt_clean_prob_model_input: 0.1
downsample_min_corrupt_ratio_model_input: 0.9
downsample_max_corrupt_ratio_model_input: 1.0
corrupt_history: true
corrupt_mode_history: "noise"
corrupt_mode_prob_history: 0.9
is_frame_independent_corrupt_history: true
is_chunk_independent_corrupt_history: false
noise_corrupt_ratio_history_short: 0.33333333333333
noise_corrupt_ratio_history_mid: 0.33333333333333
noise_corrupt_ratio_history_long: 0.33333333333333
noise_corrupt_clean_prob_history: 0.1
downsample_min_corrupt_ratio_history: 0.9
downsample_max_corrupt_ratio_history: 1.0
#
is_amplify_history: false
history_scale_mode: "per_head"
#
is_train_full_patch_embedding: false
is_train_lora_patch_embedding: true
has_multi_term_memory_patch: true
is_train_full_clean_patch_embedding: true
is_train_lora_clean_patch_embedding: false
zero_history_timestep: true
guidance_cross_attn: true
restrict_self_attn: false
is_train_restrict_lora: false
restrict_lora: false
restrict_lora_rank: 128
+179
View File
@@ -0,0 +1,179 @@
output_dir: ablation_stage_1_post
logging_dir: logs
seed: 44
report_to:
tracker_name: Wan-Train
wandb_name: ablation_stage_1_post
report_to: wandb
data_config:
# ---- Base ----
use_shuffle: true
pin_memory: true
persistent_workers: true
force_rebuild: true
single_res: true
single_height: 384
single_width: 640
dataloader_num_workers: 8
prefetch_factor: 2
caption_dropout_p: 0
id_token: ""
instance_data_root:
- "demo_data/ultravideo-long"
# ---- Stage 1 ----
use_stage1_dataset: true
model_config:
# ---- Path ----
pretrained_model_name_or_path: "BestWishYsh/Helios-Base"
transformer_model_name_or_path: "BestWishYsh/Helios-Base"
subfolder: "transformer_init"
load_checkpoints_custom: false
# load_model_path:
load_dcp: false
# load_dcp_path:
# ---- Vae ----
upcast_vae: true
enable_slicing: false
enable_tiling: false
# ---- Lora ----
lora_rank: 128
lora_alpha: 128.0
lora_dropout: 0.0
lora_layers: "all-linear"
# lora_target_modules:
# - to_k
# - to_q
# - to_v
# - to_out.0
# - ffn.net.0.proj
# - ffn.net.2
lora_exclude_modules:
- down
- up
# ---- Other ----
train_norm_layers: false
validation_config:
validation_steps: 500
validation_height: 384
validation_width: 640
validation_max_num_frames: 99
validation_prompts:
- "A stylish woman walks down a Tokyo street filled with warm glowing neon and animated city signage. She wears a black leather jacket, a long red dress, and black boots, and carries a black purse. She wears sunglasses and red lipstick. She walks confidently and casually. The street is damp and reflective, creating a mirror effect of the colorful lights. Many pedestrians walk about."
# - "Several giant wooly mammoths approach treading through a snowy meadow, their long wooly fur lightly blows in the wind as they walk, snow covered trees and dramatic snow capped mountains in the distance, mid afternoon light with wispy clouds and a sun high in the distance creates a warm glow, the low camera view is stunning capturing the large furry mammal with beautiful photography, depth of field."
# - "A movie trailer featuring the adventures of the 30 year old space man wearing a red wool knitted motorcycle helmet, blue sky, salt desert, cinematic style, shot on 35mm film, vivid colors."
validation_guidance_scale: 5.0
validation_latent_window_size:
- 9
num_validation_videos: 1
num_inference_steps: 50
training_config:
# ---- Environment ----
allow_tf32: false
gradient_checkpointing: true
enable_xformers_memory_efficient_attention: false
enable_npu_flash_attention: false
upcast_before_saving: false
offload: false
mixed_precision: "bf16"
# ---- Training Resource ----
max_train_steps: 1000000
train_batch_size: 2
gradient_accumulation_steps: 1
checkpointing_steps: 500
resume_from_checkpoint: "latest"
save_checkpoints_custom: false
# ---- Optimizer ----
learning_rate: 3e-5
lr_scheduler: "constant"
lr_warmup_steps: 500
optimizer: "adamw"
adam_beta1: 0.9
adam_beta2: 0.999
adam_weight_decay: 1e-04
adam_epsilon: 1e-08
max_grad_norm: 1.0
weighting_scheme: "logit_normal" # ["sigma_sqrt", "logit_normal", "mode", "cosmap", "none"]
logit_mean: 0.0
logit_std: 1.0
mode_scale: 1.29
# ---- Dynamic Shifting Parameters ----
use_dynamic_shifting: false
base_seq_len: 256
max_seq_len: 4096
base_shift: 0.5
max_shift: 1.15
# ---- VAE Decode Parameters ----
vae_decode_type: "default"
# ---- EMA Parameters ----
use_ema: false
use_ema_validation: false
ema_decay: 0.999
ema_start_step: 250
ema_zero3_port: 10543
ema_deepspeed_config_file: "scripts/accelerate_configs/zero3.json"
# ---- Stage 1 Parameters ----
is_enable_stage1: true
history_sizes:
- 16
- 2
- 1
latent_window_size:
# - 12
# - 10
- 9
# - 8
# - 6
# - 5
# - 4
# - 3
# - 2
# - 1
is_random_drop: true
random_drop_v2v_ratio: 0.4
random_drop_t2v_ratio: 0.4
#
corrupt_model_input: false
corrupt_mode_model_input: "noise"
corrupt_mode_prob_model_input: 0.9
is_frame_independent_corrupt_model_input: true
is_chunk_independent_corrupt_model_input: false
noise_corrupt_ratio_model_input: 0.33333333333333
noise_corrupt_clean_prob_model_input: 0.1
downsample_min_corrupt_ratio_model_input: 0.9
downsample_max_corrupt_ratio_model_input: 1.0
corrupt_history: true
corrupt_mode_history: "noise"
corrupt_mode_prob_history: 0.9
is_frame_independent_corrupt_history: true
is_chunk_independent_corrupt_history: false
noise_corrupt_ratio_history_short: 0.33333333333333
noise_corrupt_ratio_history_mid: 0.33333333333333
noise_corrupt_ratio_history_long: 0.33333333333333
noise_corrupt_clean_prob_history: 0.1
downsample_min_corrupt_ratio_history: 0.9
downsample_max_corrupt_ratio_history: 1.0
#
is_amplify_history: false
history_scale_mode: "per_head"
#
is_train_full_patch_embedding: false
is_train_lora_patch_embedding: true
has_multi_term_memory_patch: true
is_train_full_clean_patch_embedding: true
is_train_lora_clean_patch_embedding: false
zero_history_timestep: true
guidance_cross_attn: true
restrict_self_attn: false
is_train_restrict_lora: false
restrict_lora: false
restrict_lora_rank: 128
+198
View File
@@ -0,0 +1,198 @@
output_dir: ablation_stage_2_init
logging_dir: logs
seed: 45
report_to:
tracker_name: Wan-Train
wandb_name: ablation_stage_2_init
report_to: wandb
data_config:
# ---- Base ----
use_shuffle: true
pin_memory: true
persistent_workers: true
force_rebuild: true
single_res: true
single_height: 384
single_width: 640
dataloader_num_workers: 8
prefetch_factor: 2
caption_dropout_p: 0
id_token: ""
instance_data_root:
- "demo_data/ultravideo-long"
# ---- Stage 1 ----
use_stage1_dataset: true
model_config:
# ---- Path ----
pretrained_model_name_or_path: "BestWishYsh/Helios-Base"
transformer_model_name_or_path: "BestWishYsh/Helios-Base"
load_checkpoints_custom: false
# load_model_path:
load_dcp: false
# load_dcp_path:
# ---- Vae ----
upcast_vae: true
enable_slicing: false
enable_tiling: false
# ---- Lora ----
lora_rank: 256
lora_alpha: 256.0
lora_dropout: 0.0
lora_layers: "all-linear"
# lora_target_modules:
# - to_k
# - to_q
# - to_v
# - to_out.0
# - ffn.net.0.proj
# - ffn.net.2
lora_exclude_modules:
- down
- up
# ---- Other ----
train_norm_layers: false
validation_config:
validation_steps: 500
validation_height: 384
validation_width: 640
validation_max_num_frames: 99
validation_prompts:
- "A stylish woman walks down a Tokyo street filled with warm glowing neon and animated city signage. She wears a black leather jacket, a long red dress, and black boots, and carries a black purse. She wears sunglasses and red lipstick. She walks confidently and casually. The street is damp and reflective, creating a mirror effect of the colorful lights. Many pedestrians walk about."
# - "Several giant wooly mammoths approach treading through a snowy meadow, their long wooly fur lightly blows in the wind as they walk, snow covered trees and dramatic snow capped mountains in the distance, mid afternoon light with wispy clouds and a sun high in the distance creates a warm glow, the low camera view is stunning capturing the large furry mammal with beautiful photography, depth of field."
# - "A movie trailer featuring the adventures of the 30 year old space man wearing a red wool knitted motorcycle helmet, blue sky, salt desert, cinematic style, shot on 35mm film, vivid colors."
validation_guidance_scale: 5.0
validation_latent_window_size:
- 9
num_validation_videos: 1
# ---- Stage 2 ----
stage2_simulated_inference_steps:
- 20
- 20
- 20
training_config:
# ---- Environment ----
allow_tf32: false
gradient_checkpointing: true
enable_xformers_memory_efficient_attention: false
enable_npu_flash_attention: false
upcast_before_saving: false
offload: false
mixed_precision: "bf16"
# ---- Training Resource ----
max_train_steps: 1000000
train_batch_size: 1
gradient_accumulation_steps: 1
checkpointing_steps: 500
resume_from_checkpoint: "latest"
save_checkpoints_custom: false
# ---- Optimizer ----
learning_rate: 1e-4
lr_scheduler: "constant_with_warmup"
lr_warmup_steps: 1000
optimizer: "adamw"
adam_beta1: 0.9
adam_beta2: 0.999
adam_weight_decay: 1e-04
adam_epsilon: 1e-08
max_grad_norm: 1.0
weighting_scheme: "none" # ["sigma_sqrt", "logit_normal", "mode", "cosmap", "none"]
logit_mean: 0.0
logit_std: 1.0
mode_scale: 1.29
# ---- Dynamic Shifting Parameters ----
use_dynamic_shifting: false
base_seq_len: 256
max_seq_len: 4096
base_shift: 0.5
max_shift: 1.15
# ---- VAE Decode Parameters ----
vae_decode_type: "default"
# ---- EMA Parameters ----
use_ema: false
use_ema_validation: false
ema_decay: 0.999
ema_start_step: 250
ema_zero3_port: 10543
ema_deepspeed_config_file: "scripts/accelerate_configs/zero3.json"
# ---- Stage 1 Parameters ----
is_enable_stage1: true
history_sizes:
- 16
- 2
- 1
latent_window_size:
# - 12
# - 10
- 9
# - 8
# - 6
# - 5
# - 4
# - 3
# - 2
# - 1
is_random_drop: true
random_drop_v2v_ratio: 0.4
random_drop_t2v_ratio: 0.4
#
corrupt_model_input: false
corrupt_mode_model_input: "noise"
corrupt_mode_prob_model_input: 0.9
is_frame_independent_corrupt_model_input: true
is_chunk_independent_corrupt_model_input: false
noise_corrupt_ratio_model_input: 0.33333333333333
noise_corrupt_clean_prob_model_input: 0.1
downsample_min_corrupt_ratio_model_input: 0.9
downsample_max_corrupt_ratio_model_input: 1.0
corrupt_history: true
corrupt_mode_history: "noise"
corrupt_mode_prob_history: 0.9
is_frame_independent_corrupt_history: true
is_chunk_independent_corrupt_history: false
noise_corrupt_ratio_history_short: 0.33333333333333
noise_corrupt_ratio_history_mid: 0.33333333333333
noise_corrupt_ratio_history_long: 0.33333333333333
noise_corrupt_clean_prob_history: 0.1
downsample_min_corrupt_ratio_history: 0.9
downsample_max_corrupt_ratio_history: 1.0
#
is_amplify_history: false
history_scale_mode: "per_head"
#
is_train_full_patch_embedding: false
is_train_lora_patch_embedding: false
has_multi_term_memory_patch: true
is_train_full_clean_patch_embedding: false
is_train_lora_clean_patch_embedding: false
zero_history_timestep: true
guidance_cross_attn: true
restrict_self_attn: false
is_train_restrict_lora: false
restrict_lora: false
restrict_lora_rank: 128
# ---- Stage 2 Parameters ----
is_enable_stage2: true
is_navit_pyramid: true
stage2_num_stages: 3
stage2_timestep_shift: 1.0
stage2_scheduler_gamma: 0.333333333333333333333333333333333 # Approximate value of 1/3
stage2_stage_range:
- 0
- 0.333333333333333333333333333333333 # Approximate value of 1/3
- 0.666666666666666666666666666666666 # Approximate value of 2/3
- 1
stage2_sample_ratios:
- 1
- 2
- 1
efficient_sample: false
+199
View File
@@ -0,0 +1,199 @@
output_dir: ablation_stage_2_post
logging_dir: logs
seed: 46
report_to:
tracker_name: Wan-Train
wandb_name: ablation_stage_2_post
report_to: wandb
data_config:
# ---- Base ----
use_shuffle: true
pin_memory: true
persistent_workers: true
force_rebuild: true
single_res: true
single_height: 384
single_width: 640
dataloader_num_workers: 8
prefetch_factor: 2
caption_dropout_p: 0
id_token: ""
instance_data_root:
- "demo_data/ultravideo-long"
# ---- Stage 1 ----
use_stage1_dataset: true
model_config:
# ---- Path ----
pretrained_model_name_or_path: "BestWishYsh/Helios-Base"
transformer_model_name_or_path: "BestWishYsh/Helios-Mid"
subfolder: "transformer_init"
load_checkpoints_custom: false
# load_model_path:
load_dcp: false
# load_dcp_path:
# ---- Vae ----
upcast_vae: true
enable_slicing: false
enable_tiling: false
# ---- Lora ----
lora_rank: 256
lora_alpha: 256.0
lora_dropout: 0.0
lora_layers: "all-linear"
# lora_target_modules:
# - to_k
# - to_q
# - to_v
# - to_out.0
# - ffn.net.0.proj
# - ffn.net.2
lora_exclude_modules:
- down
- up
# ---- Other ----
train_norm_layers: false
validation_config:
validation_steps: 500
validation_height: 384
validation_width: 640
validation_max_num_frames: 99
validation_prompts:
- "A stylish woman walks down a Tokyo street filled with warm glowing neon and animated city signage. She wears a black leather jacket, a long red dress, and black boots, and carries a black purse. She wears sunglasses and red lipstick. She walks confidently and casually. The street is damp and reflective, creating a mirror effect of the colorful lights. Many pedestrians walk about."
# - "Several giant wooly mammoths approach treading through a snowy meadow, their long wooly fur lightly blows in the wind as they walk, snow covered trees and dramatic snow capped mountains in the distance, mid afternoon light with wispy clouds and a sun high in the distance creates a warm glow, the low camera view is stunning capturing the large furry mammal with beautiful photography, depth of field."
# - "A movie trailer featuring the adventures of the 30 year old space man wearing a red wool knitted motorcycle helmet, blue sky, salt desert, cinematic style, shot on 35mm film, vivid colors."
validation_guidance_scale: 5.0
validation_latent_window_size:
- 9
num_validation_videos: 1
# ---- Stage 2 ----
stage2_simulated_inference_steps:
- 20
- 20
- 20
training_config:
# ---- Environment ----
allow_tf32: false
gradient_checkpointing: true
enable_xformers_memory_efficient_attention: false
enable_npu_flash_attention: false
upcast_before_saving: false
offload: false
mixed_precision: "bf16"
# ---- Training Resource ----
max_train_steps: 1000000
train_batch_size: 1
gradient_accumulation_steps: 1
checkpointing_steps: 500
resume_from_checkpoint: "latest"
save_checkpoints_custom: false
# ---- Optimizer ----
learning_rate: 3e-5
lr_scheduler: "constant_with_warmup"
lr_warmup_steps: 500
optimizer: "adamw"
adam_beta1: 0.9
adam_beta2: 0.999
adam_weight_decay: 1e-04
adam_epsilon: 1e-08
max_grad_norm: 1.0
weighting_scheme: "none" # ["sigma_sqrt", "logit_normal", "mode", "cosmap", "none"]
logit_mean: 0.0
logit_std: 1.0
mode_scale: 1.29
# ---- Dynamic Shifting Parameters ----
use_dynamic_shifting: false
base_seq_len: 256
max_seq_len: 4096
base_shift: 0.5
max_shift: 1.15
# ---- VAE Decode Parameters ----
vae_decode_type: "default"
# ---- EMA Parameters ----
use_ema: false
use_ema_validation: false
ema_decay: 0.999
ema_start_step: 250
ema_zero3_port: 10543
ema_deepspeed_config_file: "scripts/accelerate_configs/zero3.json"
# ---- Stage 1 Parameters ----
is_enable_stage1: true
history_sizes:
- 16
- 2
- 1
latent_window_size:
# - 12
# - 10
- 9
# - 8
# - 6
# - 5
# - 4
# - 3
# - 2
# - 1
is_random_drop: true
random_drop_v2v_ratio: 0.4
random_drop_t2v_ratio: 0.4
#
corrupt_model_input: false
corrupt_mode_model_input: "noise"
corrupt_mode_prob_model_input: 0.9
is_frame_independent_corrupt_model_input: true
is_chunk_independent_corrupt_model_input: false
noise_corrupt_ratio_model_input: 0.33333333333333
noise_corrupt_clean_prob_model_input: 0.1
downsample_min_corrupt_ratio_model_input: 0.9
downsample_max_corrupt_ratio_model_input: 1.0
corrupt_history: true
corrupt_mode_history: "noise"
corrupt_mode_prob_history: 0.9
is_frame_independent_corrupt_history: true
is_chunk_independent_corrupt_history: false
noise_corrupt_ratio_history_short: 0.33333333333333
noise_corrupt_ratio_history_mid: 0.33333333333333
noise_corrupt_ratio_history_long: 0.33333333333333
noise_corrupt_clean_prob_history: 0.1
downsample_min_corrupt_ratio_history: 0.9
downsample_max_corrupt_ratio_history: 1.0
#
is_amplify_history: false
history_scale_mode: "per_head"
#
is_train_full_patch_embedding: false
is_train_lora_patch_embedding: true
has_multi_term_memory_patch: true
is_train_full_clean_patch_embedding: false
is_train_lora_clean_patch_embedding: true
zero_history_timestep: true
guidance_cross_attn: true
restrict_self_attn: false
is_train_restrict_lora: false
restrict_lora: false
restrict_lora_rank: 128
# ---- Stage 2 Parameters ----
is_enable_stage2: true
is_navit_pyramid: true
stage2_num_stages: 3
stage2_timestep_shift: 1.0
stage2_scheduler_gamma: 0.333333333333333333333333333333333 # Approximate value of 1/3
stage2_stage_range:
- 0
- 0.333333333333333333333333333333333 # Approximate value of 1/3
- 0.666666666666666666666666666666666 # Approximate value of 2/3
- 1
stage2_sample_ratios:
- 1
- 1
- 1
efficient_sample: false
+225
View File
@@ -0,0 +1,225 @@
output_dir: ablation_stage_3_ode
logging_dir: logs
seed: 47
report_to:
tracker_name: Wan-Train
wandb_name: ablation_stage_3_ode
report_to: wandb
data_config:
# ---- Base ----
use_shuffle: true
pin_memory: true
persistent_workers: true
force_rebuild: true
single_res: true
single_height: 384
single_width: 640
dataloader_num_workers: 8
prefetch_factor: 1
caption_dropout_p: 0
id_token: ""
# ---- Stage 1 ----
use_stage1_dataset: false
# ---- Stage 3 ----
use_stage3_dataset: true
ode_data_root:
- "demo_data/vidprom_filtered_extended"
model_config:
# ---- Path ----
pretrained_model_name_or_path: "BestWishYsh/Helios-Base"
transformer_model_name_or_path: "BestWishYsh/Helios-Mid"
load_checkpoints_custom: false
# load_model_path:
load_dcp: false
# load_dcp_path:
# ---- Vae ----
upcast_vae: true
enable_slicing: false
enable_tiling: false
# ---- Lora ----
lora_rank: 256
lora_alpha: 256.0
lora_dropout: 0.0
lora_layers: "all-linear"
# lora_target_modules:
# - to_k
# - to_q
# - to_v
# - to_out.0
# - ffn.net.0.proj
# - ffn.net.2
lora_exclude_modules:
- down
- up
# ---- Other ----
train_norm_layers: false
validation_config:
validation_steps: 500
validation_height: 384
validation_width: 640
validation_max_num_frames: 99
validation_prompts:
- "A stylish woman walks down a Tokyo street filled with warm glowing neon and animated city signage. She wears a black leather jacket, a long red dress, and black boots, and carries a black purse. She wears sunglasses and red lipstick. She walks confidently and casually. The street is damp and reflective, creating a mirror effect of the colorful lights. Many pedestrians walk about."
# - "Several giant wooly mammoths approach treading through a snowy meadow, their long wooly fur lightly blows in the wind as they walk, snow covered trees and dramatic snow capped mountains in the distance, mid afternoon light with wispy clouds and a sun high in the distance creates a warm glow, the low camera view is stunning capturing the large furry mammal with beautiful photography, depth of field."
# - "A movie trailer featuring the adventures of the 30 year old space man wearing a red wool knitted motorcycle helmet, blue sky, salt desert, cinematic style, shot on 35mm film, vivid colors."
validation_guidance_scale: 1.0
validation_latent_window_size:
- 9
num_validation_videos: 1
num_inference_steps: 6
# ---- Pyramid ----
stage2_simulated_inference_steps:
- 2
- 2
- 2
training_config:
# ---- Environment ----
allow_tf32: false
gradient_checkpointing: true
enable_xformers_memory_efficient_attention: false
enable_npu_flash_attention: false
upcast_before_saving: false
offload: false
mixed_precision: "bf16"
# ---- Training Resource ----
max_train_steps: 1000000
train_batch_size: 1
gradient_accumulation_steps: 1
checkpointing_steps: 250
resume_from_checkpoint: "latest"
save_checkpoints_custom: true
# ---- Optimizer ----
learning_rate: 2.0e-06
lr_scheduler: "constant"
lr_warmup_steps: 500
optimizer: "adamw"
adam_beta1: 0.0
adam_beta2: 0.999
adam_weight_decay: 1e-03
adam_epsilon: 1e-08
max_grad_norm: 10.0
weighting_scheme: "none" # ["sigma_sqrt", "logit_normal", "mode", "cosmap", "none"]
logit_mean: 0.0
logit_std: 1.0
mode_scale: 1.29
# ---- Dynamic Shifting Parameters ----
use_dynamic_shifting: true
base_seq_len: 256
max_seq_len: 4096
base_shift: 0.5
max_shift: 1.15
# ---- VAE Decode Parameters ----
vae_decode_type: "default"
# ---- EMA Parameters ----
use_ema: true
use_ema_validation: false
ema_decay: 0.99
ema_start_step: 250
ema_zero3_port: 10543
ema_deepspeed_config_file: "scripts/accelerate_configs/zero3.json"
# ---- Stage 1 Parameters ----
is_enable_stage1: true
history_sizes:
- 16
- 2
- 1
latent_window_size:
# - 12
# - 10
- 9
# - 8
# - 6
# - 5
# - 4
# - 3
# - 2
# - 1
is_amplify_history: false
history_scale_mode: "per_head"
#
is_train_full_patch_embedding: false
is_train_lora_patch_embedding: false
has_multi_term_memory_patch: true
is_train_full_clean_patch_embedding: false
is_train_lora_clean_patch_embedding: true
zero_history_timestep: true
guidance_cross_attn: true
restrict_self_attn: false
is_train_restrict_lora: false
restrict_lora: false
restrict_lora_rank: 128
# ---- Stage 2 Parameters ----
is_enable_stage2: true
is_navit_pyramid: false
stage2_num_stages: 3
stage2_timestep_shift: 1.0
stage2_scheduler_gamma: 0.333333333333333333333333333333333 # Approximate value of 1/3
stage2_stage_range:
- 0
- 0.333333333333333333333333333333333 # Approximate value of 1/3
- 0.666666666666666666666666666666666 # Approximate value of 2/3
- 1
stage2_sample_ratios:
- 1
- 1
- 1
efficient_sample: false
# ---- Stage 3 VRAM Parameters ----
dmd_is_low_vram_mode: true
# ---- Stage 3 Parameters ----
log_iters: 250
no_visualize: false
is_train_dmd: false
max_grad_norm_critic: 10.0
dmd_generator_deepspeed_config: scripts/accelerate_configs/zero2.json
dmd_critic_deepspeed_config: scripts/accelerate_configs/zero2.json
critic_learning_rate: 4.0e-07
dfake_gen_update_ratio: 5
dmd_denoising_step_list:
- 1000
- 750
- 500
- 250
num_critic_input_frames: 9
dmd_timestep_shift: 5.0
dmd_last_step_only: false
dmd_last_section_grad_only: false
dmd_teacher_forcing: false
dmd_teacher_forcing_ratio: 0.2
fake_guidance_scale: 0.0
real_guidance_scale: 3.0
# ---- VAE Re-Encode ----
is_dmd_vae_decode: false
# ---- Multi Stage Backward Simulated ----
is_multi_pyramid_stage_backward_simulated: false
# ---- ODE Regression Parameters ----
is_use_ode_regression: true
is_only_ode_regression: true
ode_regression_weight: 80.0
# ---- Cold Start Parameters ----
is_enable_cold_start: false
cold_start_step: 2000
stage_cold_start_step: 2000
# ---- Dynamic Timestep ----
generator_is_forcing_low_renoise: false
generator_dynamic_alpha: 4.0
generator_dynamic_beta: 1.5
generator_dynamic_sample_type: "uniform"
generator_dynamic_step: 1000
# ---- Dynamic ODE Section ----
ode_num_latent_sections_min: 3
ode_num_latent_sections_max: 3
ode_dynamic_alpha: 1.5
ode_dynamic_beta: 4.0
ode_dynamic_sample_type: "uniform"
ode_dynamic_step: 2000
+296
View File
@@ -0,0 +1,296 @@
output_dir: ablation_stage_3_post
logging_dir: logs
seed: 49
report_to:
tracker_name: Wan-Train
wandb_name: ablation_stage_3_post
report_to: wandb
data_config:
# ---- Base ----
use_shuffle: true
pin_memory: true
persistent_workers: true
force_rebuild: true
single_res: true
single_height: 384
single_width: 640
dataloader_num_workers: 8
prefetch_factor: 1
caption_dropout_p: 0
id_token: ""
# ---- Stage 1 ----
use_stage1_dataset: false
# ---- Stage 3 ----
use_stage3_dataset: true
gan_data_root:
- "demo_data/ultravideo-long"
model_config:
# ---- Path ----
pretrained_model_name_or_path: "BestWishYsh/Helios-Base"
transformer_model_name_or_path: "BestWishYsh/Helios-Distilled"
subfolder: "transformer_ode"
real_score_model_name_or_path: "BestWishYsh/Helios-Base"
load_checkpoints_custom: false
# load_model_path:
load_dcp: false
# load_dcp_path:
# ---- Vae ----
upcast_vae: true
enable_slicing: false
enable_tiling: false
# ---- Lora ----
lora_rank: 256
lora_alpha: 256.0
lora_dropout: 0.0
lora_layers: "all-linear"
# lora_target_modules:
# - to_k
# - to_q
# - to_v
# - to_out.0
# - ffn.net.0.proj
# - ffn.net.2
lora_exclude_modules:
- down
- up
# ---- Other ----
train_norm_layers: false
# ---- DMD ----
critic_lora_rank: 256
critic_lora_alpha: 256.0
critic_lora_dropout: 0.0
# ---- Reward Parameters ----
reward_model_name_or_path: "/mnt/bn/yufan-dev-my/ysh_new/Ckpts/Videoreward"
validation_config:
validation_steps: 500
validation_height: 384
validation_width: 640
validation_max_num_frames: 99
validation_prompts:
- "A stylish woman walks down a Tokyo street filled with warm glowing neon and animated city signage. She wears a black leather jacket, a long red dress, and black boots, and carries a black purse. She wears sunglasses and red lipstick. She walks confidently and casually. The street is damp and reflective, creating a mirror effect of the colorful lights. Many pedestrians walk about."
# - "Several giant wooly mammoths approach treading through a snowy meadow, their long wooly fur lightly blows in the wind as they walk, snow covered trees and dramatic snow capped mountains in the distance, mid afternoon light with wispy clouds and a sun high in the distance creates a warm glow, the low camera view is stunning capturing the large furry mammal with beautiful photography, depth of field."
# - "A movie trailer featuring the adventures of the 30 year old space man wearing a red wool knitted motorcycle helmet, blue sky, salt desert, cinematic style, shot on 35mm film, vivid colors."
validation_guidance_scale: 1.0
validation_latent_window_size:
- 9
num_validation_videos: 1
num_inference_steps: 6
# ---- Pyramid ----
stage2_simulated_inference_steps:
- 2
- 2
- 2
training_config:
# ---- Environment ----
allow_tf32: false
gradient_checkpointing: true
enable_xformers_memory_efficient_attention: false
enable_npu_flash_attention: false
upcast_before_saving: false
offload: false
mixed_precision: "bf16"
# ---- Training Resource ----
max_train_steps: 1000000
train_batch_size: 1
gradient_accumulation_steps: 1
checkpointing_steps: 250
resume_from_checkpoint: "latest"
save_checkpoints_custom: false
# ---- Optimizer ----
learning_rate: 2.0e-06
lr_scheduler: "constant"
lr_warmup_steps: 500
optimizer: "adamw"
adam_beta1: 0.0
adam_beta2: 0.999
adam_weight_decay: 1e-03
adam_epsilon: 1e-08
max_grad_norm: 10.0
weighting_scheme: "none" # ["sigma_sqrt", "logit_normal", "mode", "cosmap", "none"]
logit_mean: 0.0
logit_std: 1.0
mode_scale: 1.29
# ---- Dynamic Shifting Parameters ----
use_dynamic_shifting: true
base_seq_len: 256
max_seq_len: 4096
base_shift: 0.5
max_shift: 1.15
# ---- VAE Decode Parameters ----
vae_decode_type: "default"
# ---- EMA Parameters ----
use_ema: true
use_ema_validation: false
ema_decay: 0.99
ema_start_step: 750
ema_zero3_port: 10543
ema_deepspeed_config_file: "scripts/accelerate_configs/zero3.json"
# ---- Stage 1 Parameters ----
is_enable_stage1: true
history_sizes:
- 16
- 2
- 1
latent_window_size:
# - 12
# - 10
- 9
# - 8
# - 6
# - 5
# - 4
# - 3
# - 2
# - 1
is_random_drop: true
random_drop_v2v_ratio: 0.5
random_drop_t2v_ratio: 0.4
#
corrupt_model_input: false
corrupt_mode_model_input: "noise"
corrupt_mode_prob_model_input: 0.9
is_frame_independent_corrupt_model_input: true
is_chunk_independent_corrupt_model_input: false
noise_corrupt_ratio_model_input: 0.33333333333333
noise_corrupt_clean_prob_model_input: 0.1
downsample_min_corrupt_ratio_model_input: 0.9
downsample_max_corrupt_ratio_model_input: 1.0
corrupt_history: true
corrupt_mode_history: "noise"
corrupt_mode_prob_history: 0.9
is_frame_independent_corrupt_history: true
is_chunk_independent_corrupt_history: false
noise_corrupt_ratio_history_short: 0.33333333333333
noise_corrupt_ratio_history_mid: 0.33333333333333
noise_corrupt_ratio_history_long: 0.33333333333333
noise_corrupt_clean_prob_history: 0.1
downsample_min_corrupt_ratio_history: 0.9
downsample_max_corrupt_ratio_history: 1.0
#
is_add_saturation: true
saturation_ratio_clean_prob: 0.1
saturation_ratio_min: 0.3
saturation_ratio_max: 1.7
#
is_amplify_history: false
history_scale_mode: "per_head"
#
is_train_full_patch_embedding: false
is_train_lora_patch_embedding: false
has_multi_term_memory_patch: true
is_train_full_clean_patch_embedding: false
is_train_lora_clean_patch_embedding: true
zero_history_timestep: true
guidance_cross_attn: true
restrict_self_attn: false
is_train_restrict_lora: false
restrict_lora: false
restrict_lora_rank: 128
# ---- Stage 2 Parameters ----
is_enable_stage2: true
is_navit_pyramid: false
stage2_num_stages: 3
stage2_timestep_shift: 1.0
stage2_scheduler_gamma: 0.333333333333333333333333333333333 # Approximate value of 1/3
stage2_stage_range:
- 0
- 0.333333333333333333333333333333333 # Approximate value of 1/3
- 0.666666666666666666666666666666666 # Approximate value of 2/3
- 1
stage2_sample_ratios:
- 1
- 1
- 1
efficient_sample: false
# ---- Stage 3 VRAM Parameters ----
dmd_is_low_vram_mode: true
is_gan_low_vram_mode: true
dmd_is_offload_grad: false
# ---- Stage 3 Parameters ----
log_iters: 125
no_visualize: false
is_train_dmd: true
max_grad_norm_critic: 10.0
dmd_generator_deepspeed_config: scripts/accelerate_configs/zero2.json
dmd_critic_deepspeed_config: scripts/accelerate_configs/zero2.json
critic_learning_rate: 4.0e-07
dfake_gen_update_ratio: 5
dmd_denoising_step_list:
- 1000
- 750
- 500
- 250
num_critic_input_frames: 9
dmd_timestep_shift: 5.0
dmd_last_step_only: false
dmd_last_section_grad_only: false
dmd_teacher_forcing: false
dmd_teacher_forcing_ratio: 0.2
fake_guidance_scale: 0.0
real_guidance_scale: 3.0
# ---- GT History Parameters ----
is_use_gt_history: true
use_gt_history_ratio: 1.0
# ---- VAE Re-Encode ----
is_dmd_vae_decode: false
# ---- Multi Stage Backward Simulated ----
is_multi_pyramid_stage_backward_simulated: false
is_amplify_first_chunk: true
# ---- GAN Parameters ----
is_use_gan: false
gan_start_step: 1000
is_separate_gan_grad: false
is_use_gan_hooks: true
is_use_gan_final: true
gan_cond_map_dim: 768
gan_hooks:
- 5
- 15
- 25
- 35
gan_g_weight: 5e-2
gan_d_weight: 1e-2
aprox_r1: true
aprox_r2: true
r1_weight: 100.0
r2_weight: 0.0
r1_sigma: 0.1
r2_sigma: 0.1
# ---- Cold Start Parameters ----
is_enable_cold_start: false
cold_start_step: 2000
stage_cold_start_step: 2000
# ---- Dynamic Timestep ----
generator_is_forcing_low_renoise: false
generator_dynamic_alpha: 4.0
generator_dynamic_beta: 1.5
generator_dynamic_sample_type: "beta"
generator_dynamic_step: 500
critic_dynamic_alpha: 4.0
critic_dynamic_beta: 1.5
critic_dynamic_sample_type: "uniform"
critic_dynamic_step: 500
# ---- Dynamic DMD Section ----
dmd_num_latent_sections_min: 1
dmd_num_latent_sections_max: 1
dmd_dynamic_alpha: 1.5
dmd_dynamic_beta: 4.0
dmd_dynamic_sample_type: "uniform"
dmd_dynamic_step: 500
# ---- Dynamic ODE Section ----
ode_num_latent_sections_min: 3
ode_num_latent_sections_max: 3
ode_dynamic_alpha: 1.5
ode_dynamic_beta: 4.0
ode_dynamic_sample_type: "uniform"
ode_dynamic_step: 500
+92
View File
@@ -0,0 +1,92 @@
#!/bin/bash
export WANDB_MODE="offline"
export WANDB_API_KEY=""
export TOKENIZERS_PARALLELISM=true
export OMNISTORE_LOAD_STRICT_MODE=0
export OMNISTORE_LOGGING_LEVEL=ERROR
#################################################################
## Torch
#################################################################
export TOKENIZERS_PARALLELISM=false
export TORCH_LOGS="+dynamo,recompiles,graph_breaks"
export TORCHDYNAMO_VERBOSE=1
export TORCH_NCCL_ENABLE_MONITORING=1
export PYTORCH_CUDA_ALLOC_CONF="expandable_segments:True,garbage_collection_threshold:0.9"
#################################################################
#################################################################
## NCCL
#################################################################
export NCCL_IB_GID_INDEX=3
export NCCL_IB_HCA=$ARNOLD_RDMA_DEVICE
export NCCL_SOCKET_IFNAME=eth0
export NCCL_SOCKET_TIMEOUT=3600000
export NCCL_DEBUG=WARN # disable the verbose NCCL logs
export NCCL_P2P_DISABLE=0
export NCCL_IB_DISABLE=0 # was 1
export NCCL_SHM_DISABLE=0 # was 1
export NCCL_P2P_LEVEL=NVL
export NCCL_PXN_DISABLE=0
export NCCL_NET_GDR_LEVEL=2
export NCCL_IB_QPS_PER_CONNECTION=4
export NCCL_IB_TC=160
export NCCL_IB_TIMEOUT=22
#################################################################
# #################################################################
# ## DIST
# #################################################################
# MASTER_ADDR=$ARNOLD_WORKER_0_HOST
# ports=(`echo $METIS_WORKER_0_PORT | tr ',' ' '`)
# export MASTER_PORT=${ports[0]}
# NNODES=$ARNOLD_WORKER_NUM
# NODE_RANK=$ARNOLD_ID
# GPUS_PER_NODE=$ARNOLD_WORKER_GPU
# # GPUS_PER_NODE=1
# # NNODES=1
# # NODE_RANK=0
# WORLD_SIZE=$(($GPUS_PER_NODE*$NNODES))
# DISTRIBUTED_ARGS="--nproc_per_node $GPUS_PER_NODE --nnodes $NNODES --node_rank $NODE_RANK --master_addr $MASTER_ADDR --master_port $MASTER_PORT"
# if [ ! -z $RDZV_BACKEND ]; then
# DISTRIBUTED_ARGS="${DISTRIBUTED_ARGS} --rdzv_endpoint $MASTER_ADDR:$MASTER_PORT --rdzv_id 9863 --rdzv_backend c10d"
# export NCCL_SHM_DISABLE=1
# fi
# echo -e "\033[31mDISTRIBUTED_ARGS: ${DISTRIBUTED_ARGS}\033[0m"
#################################################################
## ACCELERATE CONFIG
#################################################################
MASTER_ADDR=$ARNOLD_WORKER_0_HOST
ports=(`echo $METIS_WORKER_0_PORT | tr ',' ' '`)
export MASTER_PORT=${ports[0]}
NUM_MACHINES=$ARNOLD_WORKER_NUM
MACHINE_RANK=$ARNOLD_ID
NUM_PROCESSES_PER_MACHINE=$ARNOLD_WORKER_GPU
# export CUDA_VISIBLE_DEVICES=0
# NUM_PROCESSES_PER_MACHINE=1
# NUM_MACHINES=1
# MACHINE_RANK=0
ACCELERATE_ARGS="--num_machines $NUM_MACHINES --machine_rank $MACHINE_RANK --num_processes $((NUM_PROCESSES_PER_MACHINE*NUM_MACHINES)) --main_process_ip $MASTER_ADDR --main_process_port $MASTER_PORT"
echo -e "\033[31mACCELERATE_ARGS: ${ACCELERATE_ARGS}\033[0m"
accelerate launch \
$ACCELERATE_ARGS \
train_helios.py \
--config scripts/training/configs/stage_1_init.yaml \
2>&1 | tee ./train.log
# accelerate launch \
# $ACCELERATE_ARGS \
# --config_file scripts/accelerate_configs/multi_node_example_zero2.yaml \
# train_helios.py \
# --config scripts/training/configs/stage_1_init.yaml \
# 2>&1 | tee ./train.log
+92
View File
@@ -0,0 +1,92 @@
#!/bin/bash
export WANDB_MODE="offline"
export WANDB_API_KEY=""
export TOKENIZERS_PARALLELISM=true
export OMNISTORE_LOAD_STRICT_MODE=0
export OMNISTORE_LOGGING_LEVEL=ERROR
#################################################################
## Torch
#################################################################
export TOKENIZERS_PARALLELISM=false
export TORCH_LOGS="+dynamo,recompiles,graph_breaks"
export TORCHDYNAMO_VERBOSE=1
export TORCH_NCCL_ENABLE_MONITORING=1
export PYTORCH_CUDA_ALLOC_CONF="expandable_segments:True,garbage_collection_threshold:0.9"
#################################################################
#################################################################
## NCCL
#################################################################
export NCCL_IB_GID_INDEX=3
export NCCL_IB_HCA=$ARNOLD_RDMA_DEVICE
export NCCL_SOCKET_IFNAME=eth0
export NCCL_SOCKET_TIMEOUT=3600000
export NCCL_DEBUG=WARN # disable the verbose NCCL logs
export NCCL_P2P_DISABLE=0
export NCCL_IB_DISABLE=0 # was 1
export NCCL_SHM_DISABLE=0 # was 1
export NCCL_P2P_LEVEL=NVL
export NCCL_PXN_DISABLE=0
export NCCL_NET_GDR_LEVEL=2
export NCCL_IB_QPS_PER_CONNECTION=4
export NCCL_IB_TC=160
export NCCL_IB_TIMEOUT=22
#################################################################
# #################################################################
# ## DIST
# #################################################################
# MASTER_ADDR=$ARNOLD_WORKER_0_HOST
# ports=(`echo $METIS_WORKER_0_PORT | tr ',' ' '`)
# export MASTER_PORT=${ports[0]}
# NNODES=$ARNOLD_WORKER_NUM
# NODE_RANK=$ARNOLD_ID
# GPUS_PER_NODE=$ARNOLD_WORKER_GPU
# # GPUS_PER_NODE=1
# # NNODES=1
# # NODE_RANK=0
# WORLD_SIZE=$(($GPUS_PER_NODE*$NNODES))
# DISTRIBUTED_ARGS="--nproc_per_node $GPUS_PER_NODE --nnodes $NNODES --node_rank $NODE_RANK --master_addr $MASTER_ADDR --master_port $MASTER_PORT"
# if [ ! -z $RDZV_BACKEND ]; then
# DISTRIBUTED_ARGS="${DISTRIBUTED_ARGS} --rdzv_endpoint $MASTER_ADDR:$MASTER_PORT --rdzv_id 9863 --rdzv_backend c10d"
# export NCCL_SHM_DISABLE=1
# fi
# echo -e "\033[31mDISTRIBUTED_ARGS: ${DISTRIBUTED_ARGS}\033[0m"
#################################################################
## ACCELERATE CONFIG
#################################################################
MASTER_ADDR=$ARNOLD_WORKER_0_HOST
ports=(`echo $METIS_WORKER_0_PORT | tr ',' ' '`)
export MASTER_PORT=${ports[0]}
NUM_MACHINES=$ARNOLD_WORKER_NUM
MACHINE_RANK=$ARNOLD_ID
NUM_PROCESSES_PER_MACHINE=$ARNOLD_WORKER_GPU
# export CUDA_VISIBLE_DEVICES=0
# NUM_PROCESSES_PER_MACHINE=1
# NUM_MACHINES=1
# MACHINE_RANK=0
ACCELERATE_ARGS="--num_machines $NUM_MACHINES --machine_rank $MACHINE_RANK --num_processes $((NUM_PROCESSES_PER_MACHINE*NUM_MACHINES)) --main_process_ip $MASTER_ADDR --main_process_port $MASTER_PORT"
echo -e "\033[31mACCELERATE_ARGS: ${ACCELERATE_ARGS}\033[0m"
# accelerate launch \
# $ACCELERATE_ARGS \
# train_helios.py \
# --config scripts/training/configs/stage_3_post.yaml \
# 2>&1 | tee ./train.log
accelerate launch \
$ACCELERATE_ARGS \
--config_file scripts/accelerate_configs/multi_node_example_zero2.yaml \
train_helios.py \
--config scripts/training/configs/stage_3_post.yaml \
2>&1 | tee ./train.log