Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cadaa0646a |
+8
-8
@@ -1,8 +1,8 @@
|
||||
Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting.
|
||||
A lone hiker stands atop a towering cliff, silhouetted against the vast horizon. The rugged landscape stretches endlessly beneath, its earthy tones blending into the soft blues of the sky. The scene captures the spirit of exploration and human resilience. High angle, dynamic framing, with soft natural lighting emphasizing the grandeur of nature.
|
||||
A hand with delicate fingers picks up a bright yellow lemon from a wooden bowl filled with lemons and sprigs of mint against a peach-colored background. The hand gently tosses the lemon up and catches it, showcasing its smooth texture. A beige string bag sits beside the bowl, adding a rustic touch to the scene. Additional lemons, one halved, are scattered around the base of the bowl. The even lighting enhances the vibrant colors and creates a fresh, inviting atmosphere.
|
||||
A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest. The playful yet serene atmosphere is complemented by soft natural light filtering through the petals. Mid-shot, warm and cheerful tones.
|
||||
A superintelligent humanoid robot waking up. The robot has a sleek metallic body with futuristic design features. Its glowing red eyes are the focal point, emanating a sharp, intense light as it powers on. The scene is set in a dimly lit, high-tech laboratory filled with glowing control panels, robotic arms, and holographic screens. The setting emphasizes advanced technology and an atmosphere of mystery. The ambiance is eerie and dramatic, highlighting the moment of awakening and the robots immense intelligence. Photorealistic style with a cinematic, dark sci-fi aesthetic. Aspect ratio: 16:9 --v 6.1
|
||||
fox in the forest close-up quickly turned its head to the left
|
||||
Man walking his dog in the woods on a hot sunny day
|
||||
A majestic lion strides across the golden savanna, its powerful frame glistening under the warm afternoon sun. The tall grass ripples gently in the breeze, enhancing the lion's commanding presence. The tone is vibrant, embodying the raw energy of the wild. Low angle, steady tracking shot, cinematic.
|
||||
Panda playing the guitar
|
||||
A princess is riding a horse across a river, realistic
|
||||
A corgi wearing sunglasses walks on the beach of a tropical island
|
||||
A princess is brushing her long golden hair in the garden.
|
||||
Several giant wooly mammoths approach treading through a snowy meadow, their long wooly fur lightly blows in the wind as they walk, snow covered trees
|
||||
A litter of golden retriever puppies playing in the snow. Their heads pop out of the snow, covered in.
|
||||
A llama sits in a cozy reading nook, surrounded by plush pillows and soft blankets. Warm, golden lighting from a floor lamp creates a welcoming atmosp
|
||||
A close up view of a glass sphere that has a zen garden within it. There is a small dwarf in the sphere who is raking the zen garden and creating patterns in the sand.
|
||||
@@ -36,10 +36,8 @@ from diffusers.utils import (USE_PEFT_BACKEND, BaseOutput, deprecate, logging,
|
||||
replace_example_docstring, scale_lora_layers)
|
||||
from diffusers.utils.torch_utils import randn_tensor
|
||||
from einops import rearrange
|
||||
|
||||
from fastvideo.utils.communications import all_gather
|
||||
from fastvideo.utils.parallel_states import (get_sequence_parallel_state,
|
||||
nccl_info)
|
||||
import imageio
|
||||
import torchvision
|
||||
|
||||
from ...constants import PRECISION_TO_TYPE
|
||||
from ...modules import HYVideoDiffusionTransformer
|
||||
@@ -842,12 +840,6 @@ class HunyuanVideoPipeline(DiffusionPipeline):
|
||||
generator,
|
||||
latents,
|
||||
)
|
||||
world_size, rank = nccl_info.sp_size, nccl_info.rank_within_group
|
||||
if get_sequence_parallel_state():
|
||||
latents = rearrange(latents,
|
||||
"b t (n s) h w -> b t n s h w",
|
||||
n=world_size).contiguous()
|
||||
latents = latents[:, :, rank, :, :, :]
|
||||
|
||||
# 6. Prepare extra step kwargs. TODO: Logic should ideally just be moved out of the pipeline
|
||||
extra_step_kwargs = self.prepare_extra_func_kwargs(
|
||||
@@ -869,10 +861,22 @@ class HunyuanVideoPipeline(DiffusionPipeline):
|
||||
num_warmup_steps = len(
|
||||
timesteps) - num_inference_steps * self.scheduler.order
|
||||
self._num_timesteps = len(timesteps)
|
||||
|
||||
import os
|
||||
if not os.path.exists("./data/step_by_step"):
|
||||
os.makedirs("./data/step_by_step", exist_ok=True)
|
||||
|
||||
# create step by step folder for this prompt
|
||||
prompt_folder = f"./data/step_by_step/{prompt}"
|
||||
if not os.path.exists(prompt_folder):
|
||||
os.makedirs(prompt_folder, exist_ok=True)
|
||||
|
||||
# if is_progress_bar:
|
||||
with self.progress_bar(total=num_inference_steps) as progress_bar:
|
||||
for i, t in enumerate(timesteps):
|
||||
step_folder = f"{prompt_folder}/{i}"
|
||||
if not os.path.exists(step_folder):
|
||||
os.makedirs(step_folder, exist_ok=True)
|
||||
mask_param = [
|
||||
mask_strategy, i
|
||||
] # if mask_strategy is None, STA will not be used
|
||||
@@ -960,9 +964,64 @@ class HunyuanVideoPipeline(DiffusionPipeline):
|
||||
if callback is not None and i % callback_steps == 0:
|
||||
step_idx = i // getattr(self.scheduler, "order", 1)
|
||||
callback(step_idx, t, latents)
|
||||
import copy
|
||||
temp_latent = copy.deepcopy(latents)
|
||||
|
||||
expand_temporal_dim = False
|
||||
if len(temp_latent.shape) == 4:
|
||||
if isinstance(self.vae, AutoencoderKLCausal3D):
|
||||
temp_latent = temp_latent.unsqueeze(2)
|
||||
expand_temporal_dim = True
|
||||
elif len(temp_latent.shape) == 5:
|
||||
pass
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Only support latents with shape (b, c, h, w) or (b, c, f, h, w), but got {temp_latent.shape}."
|
||||
)
|
||||
|
||||
if (
|
||||
hasattr(self.vae.config, "shift_factor")
|
||||
and self.vae.config.shift_factor
|
||||
):
|
||||
temp_latent = (
|
||||
temp_latent / self.vae.config.scaling_factor
|
||||
+ self.vae.config.shift_factor
|
||||
)
|
||||
else:
|
||||
temp_latent = temp_latent / self.vae.config.scaling_factor
|
||||
|
||||
with torch.autocast(
|
||||
device_type="cuda", dtype=vae_dtype, enabled=vae_autocast_enabled
|
||||
):
|
||||
if enable_tiling:
|
||||
self.vae.enable_tiling()
|
||||
image = self.vae.decode(
|
||||
temp_latent, return_dict=False, generator=generator
|
||||
)[0]
|
||||
else:
|
||||
image = self.vae.decode(
|
||||
temp_latent, return_dict=False, generator=generator
|
||||
)[0]
|
||||
|
||||
if expand_temporal_dim or image.shape[2] == 1:
|
||||
image = image.squeeze(2)
|
||||
|
||||
image = (image / 2 + 0.5).clamp(0, 1)
|
||||
# we always cast to float32 as this does not cause significant overhead and is compatible with bfloa16
|
||||
image = image.cpu().float()
|
||||
|
||||
# save image
|
||||
output_path = f"{step_folder}/output.mp4"
|
||||
|
||||
|
||||
videos = rearrange(image, "b c t h w -> t b c h w")
|
||||
video_frames = []
|
||||
for x in videos:
|
||||
x = torchvision.utils.make_grid(x, nrow=6)
|
||||
x = x.transpose(0, 1).transpose(1, 2).squeeze(-1)
|
||||
video_frames.append((x * 255).numpy().astype(np.uint8))
|
||||
imageio.mimsave(output_path, video_frames, fps=24)
|
||||
|
||||
if get_sequence_parallel_state():
|
||||
latents = all_gather(latents, dim=2)
|
||||
|
||||
if not output_type == "latent":
|
||||
expand_temporal_dim = False
|
||||
|
||||
@@ -195,19 +195,18 @@ def teacache_forward(
|
||||
|
||||
|
||||
def initialize_distributed():
|
||||
local_rank = int(os.getenv("RANK", 0))
|
||||
world_size = int(os.getenv("WORLD_SIZE", 1))
|
||||
print("world_size", world_size)
|
||||
local_rank = int(os.getenv("LOCAL_RANK", 0)) # Fetch local rank
|
||||
# No need to remap `local_rank` because `CUDA_VISIBLE_DEVICES` already restricts visibility
|
||||
torch.cuda.set_device(local_rank)
|
||||
dist.init_process_group(backend="nccl",
|
||||
init_method="env://",
|
||||
world_size=world_size,
|
||||
rank=local_rank)
|
||||
initialize_sequence_parallel_state(world_size)
|
||||
dist.init_process_group(
|
||||
backend="nccl", init_method="env://", world_size=1, rank=local_rank
|
||||
)
|
||||
|
||||
|
||||
def main(args):
|
||||
initialize_distributed()
|
||||
local_rank = int(os.getenv('RANK', 0))
|
||||
world_size = int(os.getenv('WORLD_SIZE', 1))
|
||||
print(nccl_info.sp_size)
|
||||
|
||||
print(args)
|
||||
@@ -241,7 +240,10 @@ def main(args):
|
||||
with open(args.prompt) as f:
|
||||
prompts = [line.strip() for line in f.readlines()]
|
||||
|
||||
for prompt in prompts:
|
||||
for idx, prompt in enumerate(prompts):
|
||||
|
||||
if idx % world_size != local_rank:
|
||||
continue
|
||||
outputs = hunyuan_video_sampler.predict(
|
||||
prompt=prompt,
|
||||
height=args.height,
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
#!/bin/bash
|
||||
|
||||
num_gpus=1
|
||||
num_gpus=8
|
||||
mask_strategy_file_path=assets/mask_strategy.json
|
||||
export MODEL_BASE=data/hunyuan
|
||||
rel_l1_thresh=0.2
|
||||
CUDA_VISIBLE_DEVICES=1 torchrun --nnodes=1 --nproc_per_node=$num_gpus --master_port 29603 \
|
||||
rel_l1_thresh=0.15
|
||||
torchrun --nnodes=1 --nproc_per_node=$num_gpus --master_port 29603 \
|
||||
fastvideo/sample/sample_t2v_hunyuan_STA.py \
|
||||
--height 768 \
|
||||
--width 1280 \
|
||||
|
||||
Reference in New Issue
Block a user