Compare commits
1
Commits
wei/videox
...
kernels
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0d12c41fc8 |
@@ -0,0 +1,102 @@
|
||||
import triton
|
||||
import triton.language as tl
|
||||
|
||||
CONFIG_LIST = [
|
||||
triton.Config({"BLOCK_M": 256, "BLOCK_N": 32}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 128, "BLOCK_N": 64}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 128, "BLOCK_N": 32}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 64, "BLOCK_N": 128}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 64, "BLOCK_N": 64}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 64, "BLOCK_N": 32}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 32, "BLOCK_N": 64}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 32, "BLOCK_N": 128}, num_stages=2, num_warps=4),
|
||||
triton.Config({"BLOCK_M": 32, "BLOCK_N": 256}, num_stages=2, num_warps=4),
|
||||
]
|
||||
|
||||
|
||||
@triton.autotune(
|
||||
configs=CONFIG_LIST,
|
||||
key=["M", "N"],
|
||||
)
|
||||
@triton.jit
|
||||
def _modulate_fwd(
|
||||
x_ptr, # *Pointer* to first input vector.
|
||||
output_ptr, # *Pointer* to output vector.
|
||||
scale_ptr,
|
||||
shift_ptr,
|
||||
m_stride,
|
||||
s_stride,
|
||||
M,
|
||||
N,
|
||||
seq_len,
|
||||
BLOCK_M: tl.constexpr, # Number of elements each program should process.
|
||||
BLOCK_N: tl.constexpr,
|
||||
# NOTE: `constexpr` so it can be used as a shape value.
|
||||
):
|
||||
row_id = tl.program_id(axis=0) # We use a 1D launch grid so axis is 0.
|
||||
rows = row_id * BLOCK_M + tl.arange(0, BLOCK_M)
|
||||
s_rows = (row_id // seq_len) * BLOCK_M
|
||||
col_id = tl.program_id(axis=1)
|
||||
cols = col_id * BLOCK_N + tl.arange(0, BLOCK_N)
|
||||
|
||||
x_ptrs = x_ptr + rows[:, None] * m_stride + cols[None, :]
|
||||
scale_ptrs = scale_ptr + s_rows * s_stride + cols[None, :]
|
||||
shift_ptrs = shift_ptr + s_rows * s_stride + cols[None, :]
|
||||
|
||||
col_mask = cols[None, :] < N
|
||||
block_mask = (rows[:, None] < M) & col_mask
|
||||
s_block_mask = col_mask
|
||||
x = tl.load(x_ptrs, mask=block_mask, other=0.0)
|
||||
scale = tl.load(scale_ptrs, mask=s_block_mask, other=0.0)
|
||||
shift = tl.load(shift_ptrs, mask=s_block_mask, other=0.0)
|
||||
|
||||
output = x * (1 + scale) + shift
|
||||
# Write x + y back to DRAM.
|
||||
tl.store(output_ptr + rows[:, None] * m_stride + cols[None, :], output, mask=block_mask)
|
||||
|
||||
|
||||
@triton.autotune(
|
||||
configs=CONFIG_LIST,
|
||||
key=["M", "N"],
|
||||
)
|
||||
@triton.jit
|
||||
def _modulate_bwd(
|
||||
dx_ptr, # *Pointer* to first input vector.
|
||||
x_ptr,
|
||||
dy_ptr, # *Pointer* to output vector.
|
||||
scale_ptr,
|
||||
dscale_ptr,
|
||||
m_stride,
|
||||
s_stride,
|
||||
M,
|
||||
N,
|
||||
seq_len,
|
||||
BLOCK_M: tl.constexpr, # Number of elements each program should process.
|
||||
BLOCK_N: tl.constexpr,
|
||||
# NOTE: `constexpr` so it can be used as a shape value.
|
||||
):
|
||||
row_id = tl.program_id(axis=0) # We use a 1D launch grid so axis is 0.
|
||||
rows = row_id * BLOCK_M + tl.arange(0, BLOCK_M)
|
||||
s_rows = (row_id // seq_len) * BLOCK_M
|
||||
col_id = tl.program_id(axis=1)
|
||||
cols = col_id * BLOCK_N + tl.arange(0, BLOCK_N)
|
||||
|
||||
x_ptrs = x_ptr + rows[:, None] * m_stride + cols[None, :]
|
||||
dy_ptrs = dy_ptr + rows[:, None] * m_stride + cols[None, :]
|
||||
dx_ptrs = dx_ptr + rows[:, None] * m_stride + cols[None, :]
|
||||
dscale_ptrs = dscale_ptr + rows[:, None] * m_stride + cols[None, :]
|
||||
|
||||
scale_ptrs = scale_ptr + s_rows * s_stride + cols[None, :]
|
||||
|
||||
col_mask = cols[None, :] < N
|
||||
block_mask = (rows[:, None] < M) & col_mask
|
||||
s_block_mask = col_mask
|
||||
x = tl.load(x_ptrs, mask=block_mask, other=0.0)
|
||||
dy = tl.load(dy_ptrs, mask=block_mask, other=0.0)
|
||||
scale = tl.load(scale_ptrs, mask=s_block_mask, other=0.0)
|
||||
|
||||
dx = dy * (1 + scale)
|
||||
dscale = dy * x
|
||||
# Write x + y back to DRAM.
|
||||
tl.store(dx_ptrs, dx, mask=block_mask)
|
||||
tl.store(dscale_ptrs, dscale, mask=block_mask)
|
||||
@@ -0,0 +1,63 @@
|
||||
import torch
|
||||
import triton
|
||||
|
||||
from fastvideo.ops.modulate.k_modulate import _modulate_fwd, _modulate_bwd
|
||||
|
||||
|
||||
class _FusedModulate(torch.autograd.Function):
|
||||
@staticmethod
|
||||
def forward(ctx, x, scale, shift):
|
||||
y = torch.empty_like(x)
|
||||
batch, seq_len, dim = x.shape
|
||||
M = batch * seq_len
|
||||
N = dim
|
||||
x = x.view(-1, dim).contiguous()
|
||||
scale = scale.view(-1, dim).contiguous()
|
||||
shift = shift.view(-1, dim).contiguous()
|
||||
|
||||
def grid(meta):
|
||||
return (
|
||||
triton.cdiv(batch * seq_len, meta["BLOCK_M"]),
|
||||
triton.cdiv(dim, meta["BLOCK_N"]),
|
||||
)
|
||||
|
||||
_modulate_fwd[grid](x, y, scale, shift, x.stride(0), scale.stride(0), M, N, seq_len)
|
||||
|
||||
ctx.save_for_backward(x, scale)
|
||||
ctx.batch = batch
|
||||
ctx.seq_len = seq_len
|
||||
ctx.dim = dim
|
||||
return y
|
||||
|
||||
@staticmethod
|
||||
def backward(ctx, dy): # pragma: no cover # this is covered, but called directly from C++
|
||||
x, scale = ctx.saved_tensors
|
||||
|
||||
batch, seq_len, dim = ctx.batch, ctx.seq_len, ctx.dim
|
||||
M = batch * seq_len
|
||||
N = dim
|
||||
|
||||
# allocate output
|
||||
dy = dy.contiguous()
|
||||
dx = torch.empty_like(dy)
|
||||
dscale = torch.empty_like(dy)
|
||||
dshift = torch.sum(dy, dim=1)
|
||||
|
||||
def grid(meta):
|
||||
return (
|
||||
triton.cdiv(batch * seq_len, meta["BLOCK_M"]),
|
||||
triton.cdiv(dim, meta["BLOCK_N"]),
|
||||
)
|
||||
|
||||
_modulate_bwd[grid](dx, x, dy, scale, dscale, x.stride(0), scale.stride(0), M, N, seq_len)
|
||||
|
||||
dscale = torch.sum(dscale, dim=1)
|
||||
return dx, dscale, dshift
|
||||
|
||||
|
||||
def fused_modulate(
|
||||
x: torch.Tensor,
|
||||
scale: torch.Tensor,
|
||||
shift: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
return _FusedModulate.apply(x, scale, shift)
|
||||
@@ -0,0 +1,306 @@
|
||||
"""
|
||||
This script demonstrates how to generate a video using the CogVideoX model with the Hugging Face `diffusers` pipeline.
|
||||
The script supports different types of video generation, including text-to-video (t2v), image-to-video (i2v),
|
||||
and video-to-video (v2v), depending on the input data and different weight.
|
||||
|
||||
- text-to-video: THUDM/CogVideoX-5b, THUDM/CogVideoX-2b or THUDM/CogVideoX1.5-5b
|
||||
- video-to-video: THUDM/CogVideoX-5b, THUDM/CogVideoX-2b or THUDM/CogVideoX1.5-5b
|
||||
- image-to-video: THUDM/CogVideoX-5b-I2V or THUDM/CogVideoX1.5-5b-I2V
|
||||
|
||||
Running the Script:
|
||||
To run the script, use the following command with appropriate arguments:
|
||||
|
||||
```bash
|
||||
$ python cli_demo.py --prompt "A girl riding a bike." --model_path THUDM/CogVideoX1.5-5b --generate_type "t2v"
|
||||
```
|
||||
|
||||
Additional options are available to specify the model path, guidance scale, number of inference steps, video generation type, and output paths.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import os
|
||||
from typing import Literal, Optional
|
||||
|
||||
import torch
|
||||
from diffusers import (
|
||||
CogVideoXDPMScheduler,
|
||||
CogVideoXImageToVideoPipeline,
|
||||
CogVideoXPipeline,
|
||||
CogVideoXVideoToVideoPipeline,
|
||||
)
|
||||
from diffusers.utils import export_to_video, load_image, load_video
|
||||
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
|
||||
# Recommended resolution for each model (width, height)
|
||||
RESOLUTION_MAP = {
|
||||
# cogvideox1.5-*
|
||||
"cogvideox1.5-5b-i2v": (1360, 768),
|
||||
"cogvideox1.5-5b": (1360, 768),
|
||||
# cogvideox-*
|
||||
"cogvideox-5b-i2v": (720, 480),
|
||||
"cogvideox-5b": (720, 480),
|
||||
"cogvideox-2b": (720, 480),
|
||||
}
|
||||
|
||||
|
||||
def generate_video(
|
||||
prompt: str,
|
||||
model_path: str,
|
||||
lora_path: str = None,
|
||||
lora_rank: int = 128,
|
||||
num_frames: int = 81,
|
||||
width: Optional[int] = None,
|
||||
height: Optional[int] = None,
|
||||
output_path: str = "./output.mp4",
|
||||
image_or_video_path: str = "",
|
||||
num_inference_steps: int = 50,
|
||||
guidance_scale: float = 6.0,
|
||||
num_videos_per_prompt: int = 1,
|
||||
dtype: torch.dtype = torch.bfloat16,
|
||||
generate_type: str = Literal[
|
||||
"t2v", "i2v", "v2v"
|
||||
], # i2v: image to video, v2v: video to video
|
||||
seed: int = 42,
|
||||
fps: int = 16,
|
||||
):
|
||||
"""
|
||||
Generates a video based on the given prompt and saves it to the specified path.
|
||||
|
||||
Parameters:
|
||||
- prompt (str): The description of the video to be generated.
|
||||
- model_path (str): The path of the pre-trained model to be used.
|
||||
- lora_path (str): The path of the LoRA weights to be used.
|
||||
- lora_rank (int): The rank of the LoRA weights.
|
||||
- output_path (str): The path where the generated video will be saved.
|
||||
- num_inference_steps (int): Number of steps for the inference process. More steps can result in better quality.
|
||||
- num_frames (int): Number of frames to generate. CogVideoX1.0 generates 49 frames for 6 seconds at 8 fps, while CogVideoX1.5 produces either 81 or 161 frames, corresponding to 5 seconds or 10 seconds at 16 fps.
|
||||
- width (int): The width of the generated video, applicable only for CogVideoX1.5-5B-I2V
|
||||
- height (int): The height of the generated video, applicable only for CogVideoX1.5-5B-I2V
|
||||
- guidance_scale (float): The scale for classifier-free guidance. Higher values can lead to better alignment with the prompt.
|
||||
- num_videos_per_prompt (int): Number of videos to generate per prompt.
|
||||
- dtype (torch.dtype): The data type for computation (default is torch.bfloat16).
|
||||
- generate_type (str): The type of video generation (e.g., 't2v', 'i2v', 'v2v').·
|
||||
- seed (int): The seed for reproducibility.
|
||||
- fps (int): The frames per second for the generated video.
|
||||
"""
|
||||
|
||||
# 1. Load the pre-trained CogVideoX pipeline with the specified precision (bfloat16).
|
||||
# add device_map="balanced" in the from_pretrained function and remove the enable_model_cpu_offload()
|
||||
# function to use Multi GPUs.
|
||||
|
||||
image = None
|
||||
video = None
|
||||
|
||||
model_name = model_path.split("/")[-1].lower()
|
||||
desired_resolution = RESOLUTION_MAP[model_name]
|
||||
if width is None or height is None:
|
||||
width, height = desired_resolution
|
||||
logging.info(
|
||||
f"\033[1mUsing default resolution {desired_resolution} for {model_name}\033[0m"
|
||||
)
|
||||
elif (width, height) != desired_resolution:
|
||||
if generate_type == "i2v":
|
||||
# For i2v models, use user-defined width and height
|
||||
logging.warning(
|
||||
f"\033[1;31mThe width({width}) and height({height}) are not recommended for {model_name}. The best resolution is {desired_resolution}.\033[0m"
|
||||
)
|
||||
else:
|
||||
# Otherwise, use the recommended width and height
|
||||
logging.warning(
|
||||
f"\033[1;31m{model_name} is not supported for custom resolution. Setting back to default resolution {desired_resolution}.\033[0m"
|
||||
)
|
||||
width, height = desired_resolution
|
||||
|
||||
if generate_type == "i2v":
|
||||
pipe = CogVideoXImageToVideoPipeline.from_pretrained(
|
||||
model_path, torch_dtype=dtype
|
||||
)
|
||||
image = load_image(image=image_or_video_path)
|
||||
elif generate_type == "t2v":
|
||||
pipe = CogVideoXPipeline.from_pretrained(model_path, torch_dtype=dtype)
|
||||
else:
|
||||
pipe = CogVideoXVideoToVideoPipeline.from_pretrained(
|
||||
model_path, torch_dtype=dtype
|
||||
)
|
||||
video = load_video(image_or_video_path)
|
||||
|
||||
# If you're using with lora, add this code
|
||||
if lora_path:
|
||||
pipe.load_lora_weights(
|
||||
lora_path,
|
||||
weight_name="pytorch_lora_weights.safetensors",
|
||||
adapter_name="test_1",
|
||||
)
|
||||
pipe.fuse_lora(lora_scale=1 / lora_rank)
|
||||
|
||||
# 2. Set Scheduler.
|
||||
# Can be changed to `CogVideoXDPMScheduler` or `CogVideoXDDIMScheduler`.
|
||||
# We recommend using `CogVideoXDDIMScheduler` for CogVideoX-2B.
|
||||
# using `CogVideoXDPMScheduler` for CogVideoX-5B / CogVideoX-5B-I2V.
|
||||
|
||||
# pipe.scheduler = CogVideoXDDIMScheduler.from_config(pipe.scheduler.config, timestep_spacing="trailing")
|
||||
pipe.scheduler = CogVideoXDPMScheduler.from_config(
|
||||
pipe.scheduler.config, timestep_spacing="trailing"
|
||||
)
|
||||
|
||||
# 3. Enable CPU offload for the model.
|
||||
# turn off if you have multiple GPUs or enough GPU memory(such as H100) and it will cost less time in inference
|
||||
# and enable to("cuda")
|
||||
|
||||
# pipe.to("cuda")
|
||||
pipe.enable_sequential_cpu_offload()
|
||||
pipe.vae.enable_slicing()
|
||||
pipe.vae.enable_tiling()
|
||||
|
||||
# 4. Generate the video frames based on the prompt.
|
||||
# `num_frames` is the Number of frames to generate.
|
||||
if generate_type == "i2v":
|
||||
video_generate = pipe(
|
||||
height=height,
|
||||
width=width,
|
||||
prompt=prompt,
|
||||
image=image,
|
||||
# The path of the image, the resolution of video will be the same as the image for CogVideoX1.5-5B-I2V, otherwise it will be 720 * 480
|
||||
num_videos_per_prompt=num_videos_per_prompt, # Number of videos to generate per prompt
|
||||
num_inference_steps=num_inference_steps, # Number of inference steps
|
||||
num_frames=num_frames, # Number of frames to generate
|
||||
use_dynamic_cfg=True, # This id used for DPM scheduler, for DDIM scheduler, it should be False
|
||||
guidance_scale=guidance_scale,
|
||||
generator=torch.Generator().manual_seed(
|
||||
seed
|
||||
), # Set the seed for reproducibility
|
||||
).frames[0]
|
||||
elif generate_type == "t2v":
|
||||
video_generate = pipe(
|
||||
height=height,
|
||||
width=width,
|
||||
prompt=prompt,
|
||||
num_videos_per_prompt=num_videos_per_prompt,
|
||||
num_inference_steps=num_inference_steps,
|
||||
num_frames=num_frames,
|
||||
use_dynamic_cfg=True,
|
||||
guidance_scale=guidance_scale,
|
||||
generator=torch.Generator().manual_seed(seed),
|
||||
).frames[0]
|
||||
else:
|
||||
video_generate = pipe(
|
||||
height=height,
|
||||
width=width,
|
||||
prompt=prompt,
|
||||
video=video, # The path of the video to be used as the background of the video
|
||||
num_videos_per_prompt=num_videos_per_prompt,
|
||||
num_inference_steps=num_inference_steps,
|
||||
num_frames=num_frames,
|
||||
use_dynamic_cfg=True,
|
||||
guidance_scale=guidance_scale,
|
||||
generator=torch.Generator().manual_seed(
|
||||
seed
|
||||
), # Set the seed for reproducibility
|
||||
).frames[0]
|
||||
export_to_video(video_generate, output_path, fps=fps)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Generate a video from a text prompt using CogVideoX"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--prompt",
|
||||
type=str,
|
||||
required=True,
|
||||
help="The description of the video to be generated",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--image_or_video_path",
|
||||
type=str,
|
||||
default=None,
|
||||
help="The path of the image to be used as the background of the video",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--model_path",
|
||||
type=str,
|
||||
default="THUDM/CogVideoX1.5-5B",
|
||||
help="Path of the pre-trained model use",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--lora_path",
|
||||
type=str,
|
||||
default=None,
|
||||
help="The path of the LoRA weights to be used",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--lora_rank", type=int, default=128, help="The rank of the LoRA weights"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output_path",
|
||||
type=str,
|
||||
default="./output.mp4",
|
||||
help="The path save generated video",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--guidance_scale",
|
||||
type=float,
|
||||
default=6.0,
|
||||
help="The scale for classifier-free guidance",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--num_inference_steps", type=int, default=50, help="Inference steps"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--num_frames",
|
||||
type=int,
|
||||
default=81,
|
||||
help="Number of steps for the inference process",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--width", type=int, default=None, help="The width of the generated video"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--height", type=int, default=None, help="The height of the generated video"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--fps",
|
||||
type=int,
|
||||
default=16,
|
||||
help="The frames per second for the generated video",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--num_videos_per_prompt",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Number of videos to generate per prompt",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--generate_type", type=str, default="t2v", help="The type of video generation"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--dtype", type=str, default="bfloat16", help="The data type for computation"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--seed", type=int, default=42, help="The seed for reproducibility"
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
dtype = torch.float16 if args.dtype == "float16" else torch.bfloat16
|
||||
os.makedirs(os.path.dirname(args.output_path), exist_ok=True)
|
||||
generate_video(
|
||||
prompt=args.prompt,
|
||||
model_path=args.model_path,
|
||||
lora_path=args.lora_path,
|
||||
lora_rank=args.lora_rank,
|
||||
output_path=args.output_path,
|
||||
num_frames=args.num_frames,
|
||||
width=args.width,
|
||||
height=args.height,
|
||||
image_or_video_path=args.image_or_video_path,
|
||||
num_inference_steps=args.num_inference_steps,
|
||||
guidance_scale=args.guidance_scale,
|
||||
num_videos_per_prompt=args.num_videos_per_prompt,
|
||||
dtype=dtype,
|
||||
generate_type=args.generate_type,
|
||||
seed=args.seed,
|
||||
fps=args.fps,
|
||||
)
|
||||
@@ -0,0 +1,297 @@
|
||||
# Copyright (c) 2023, Tri Dao.
|
||||
"""Useful functions for writing test code."""
|
||||
|
||||
import torch
|
||||
import torch.utils.benchmark as benchmark
|
||||
|
||||
|
||||
def get_abs_err(x, y):
|
||||
return (x - y).flatten().abs().max().item()
|
||||
|
||||
|
||||
def get_err_ratio(x, y):
|
||||
err = (x - y).flatten().square().mean().sqrt().item()
|
||||
base = x.flatten().square().mean().sqrt().item()
|
||||
return err / base
|
||||
|
||||
|
||||
def assert_close(prefix, ref, tri, ratio):
|
||||
msg = f"{prefix} diff: {get_abs_err(ref, tri):.6f} ratio: {get_err_ratio(ref, tri):.6f}"
|
||||
print(msg)
|
||||
assert get_err_ratio(ref, tri) < ratio, msg
|
||||
|
||||
|
||||
def benchmark_forward(
|
||||
fn,
|
||||
*inputs,
|
||||
repeats=10,
|
||||
desc="",
|
||||
verbose=True,
|
||||
amp=False,
|
||||
amp_dtype=torch.float16,
|
||||
**kwinputs,
|
||||
):
|
||||
"""Use Pytorch Benchmark on the forward pass of an arbitrary function."""
|
||||
if verbose:
|
||||
print(desc, "- Forward pass")
|
||||
|
||||
def amp_wrapper(*inputs, **kwinputs):
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
|
||||
fn(*inputs, **kwinputs)
|
||||
|
||||
t = benchmark.Timer(
|
||||
stmt="fn_amp(*inputs, **kwinputs)",
|
||||
globals={"fn_amp": amp_wrapper, "inputs": inputs, "kwinputs": kwinputs},
|
||||
num_threads=torch.get_num_threads(),
|
||||
)
|
||||
m = t.timeit(repeats)
|
||||
if verbose:
|
||||
print(m)
|
||||
return t, m
|
||||
|
||||
|
||||
def benchmark_backward(
|
||||
fn,
|
||||
*inputs,
|
||||
grad=None,
|
||||
repeats=10,
|
||||
desc="",
|
||||
verbose=True,
|
||||
amp=False,
|
||||
amp_dtype=torch.float16,
|
||||
**kwinputs,
|
||||
):
|
||||
"""Use Pytorch Benchmark on the backward pass of an arbitrary function."""
|
||||
if verbose:
|
||||
print(desc, "- Backward pass")
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
|
||||
y = fn(*inputs, **kwinputs)
|
||||
if type(y) is tuple:
|
||||
y = y[0]
|
||||
if grad is None:
|
||||
grad = torch.randn_like(y)
|
||||
else:
|
||||
if grad.shape != y.shape:
|
||||
raise RuntimeError("Grad shape does not match output shape")
|
||||
|
||||
def f(*inputs, y, grad):
|
||||
# Set .grad to None to avoid extra operation of gradient accumulation
|
||||
for x in inputs:
|
||||
if isinstance(x, torch.Tensor):
|
||||
x.grad = None
|
||||
y.backward(grad, retain_graph=True)
|
||||
|
||||
t = benchmark.Timer(
|
||||
stmt="f(*inputs, y=y, grad=grad)",
|
||||
globals={"f": f, "inputs": inputs, "y": y, "grad": grad},
|
||||
num_threads=torch.get_num_threads(),
|
||||
)
|
||||
m = t.timeit(repeats)
|
||||
if verbose:
|
||||
print(m)
|
||||
return t, m
|
||||
|
||||
|
||||
def benchmark_combined(
|
||||
fn,
|
||||
*inputs,
|
||||
grad=None,
|
||||
repeats=10,
|
||||
desc="",
|
||||
verbose=True,
|
||||
amp=False,
|
||||
amp_dtype=torch.float16,
|
||||
**kwinputs,
|
||||
):
|
||||
"""Use Pytorch Benchmark on the forward+backward pass of an arbitrary function."""
|
||||
if verbose:
|
||||
print(desc, "- Forward + Backward pass")
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
|
||||
y = fn(*inputs, **kwinputs)
|
||||
if type(y) is tuple:
|
||||
y = y[0]
|
||||
if grad is None:
|
||||
grad = torch.randn_like(y)
|
||||
else:
|
||||
if grad.shape != y.shape:
|
||||
raise RuntimeError("Grad shape does not match output shape")
|
||||
|
||||
def f(grad, *inputs, **kwinputs):
|
||||
for x in inputs:
|
||||
if isinstance(x, torch.Tensor):
|
||||
x.grad = None
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
|
||||
y = fn(*inputs, **kwinputs)
|
||||
if type(y) is tuple:
|
||||
y = y[0]
|
||||
y.backward(grad, retain_graph=True)
|
||||
|
||||
t = benchmark.Timer(
|
||||
stmt="f(grad, *inputs, **kwinputs)",
|
||||
globals={
|
||||
"f": f,
|
||||
"fn": fn,
|
||||
"inputs": inputs,
|
||||
"grad": grad,
|
||||
"kwinputs": kwinputs,
|
||||
},
|
||||
num_threads=torch.get_num_threads(),
|
||||
)
|
||||
m = t.timeit(repeats)
|
||||
if verbose:
|
||||
print(m)
|
||||
return t, m
|
||||
|
||||
|
||||
def benchmark_fwd_bwd(
|
||||
fn,
|
||||
*inputs,
|
||||
grad=None,
|
||||
repeats=10,
|
||||
desc="",
|
||||
verbose=True,
|
||||
amp=False,
|
||||
amp_dtype=torch.float16,
|
||||
**kwinputs,
|
||||
):
|
||||
"""Use Pytorch Benchmark on the forward+backward pass of an arbitrary function."""
|
||||
return (
|
||||
benchmark_forward(
|
||||
fn,
|
||||
*inputs,
|
||||
repeats=repeats,
|
||||
desc=desc,
|
||||
verbose=verbose,
|
||||
amp=amp,
|
||||
amp_dtype=amp_dtype,
|
||||
**kwinputs,
|
||||
),
|
||||
benchmark_backward(
|
||||
fn,
|
||||
*inputs,
|
||||
grad=grad,
|
||||
repeats=repeats,
|
||||
desc=desc,
|
||||
verbose=verbose,
|
||||
amp=amp,
|
||||
amp_dtype=amp_dtype,
|
||||
**kwinputs,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def benchmark_all(
|
||||
fn,
|
||||
*inputs,
|
||||
grad=None,
|
||||
repeats=10,
|
||||
desc="",
|
||||
verbose=True,
|
||||
amp=False,
|
||||
amp_dtype=torch.float16,
|
||||
**kwinputs,
|
||||
):
|
||||
"""Use Pytorch Benchmark on the forward+backward pass of an arbitrary function."""
|
||||
return (
|
||||
benchmark_forward(
|
||||
fn,
|
||||
*inputs,
|
||||
repeats=repeats,
|
||||
desc=desc,
|
||||
verbose=verbose,
|
||||
amp=amp,
|
||||
amp_dtype=amp_dtype,
|
||||
**kwinputs,
|
||||
),
|
||||
benchmark_backward(
|
||||
fn,
|
||||
*inputs,
|
||||
grad=grad,
|
||||
repeats=repeats,
|
||||
desc=desc,
|
||||
verbose=verbose,
|
||||
amp=amp,
|
||||
amp_dtype=amp_dtype,
|
||||
**kwinputs,
|
||||
),
|
||||
benchmark_combined(
|
||||
fn,
|
||||
*inputs,
|
||||
grad=grad,
|
||||
repeats=repeats,
|
||||
desc=desc,
|
||||
verbose=verbose,
|
||||
amp=amp,
|
||||
amp_dtype=amp_dtype,
|
||||
**kwinputs,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def pytorch_profiler(
|
||||
fn,
|
||||
*inputs,
|
||||
trace_filename=None,
|
||||
backward=False,
|
||||
amp=False,
|
||||
amp_dtype=torch.float16,
|
||||
cpu=False,
|
||||
verbose=True,
|
||||
**kwinputs,
|
||||
):
|
||||
"""Wrap benchmark functions in Pytorch profiler to see CUDA information."""
|
||||
if backward:
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
|
||||
out = fn(*inputs, **kwinputs)
|
||||
if type(out) is tuple:
|
||||
out = out[0]
|
||||
g = torch.randn_like(out)
|
||||
for _ in range(30): # Warm up
|
||||
if backward:
|
||||
for x in inputs:
|
||||
if isinstance(x, torch.Tensor):
|
||||
x.grad = None
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
|
||||
out = fn(*inputs, **kwinputs)
|
||||
if type(out) is tuple:
|
||||
out = out[0]
|
||||
# Backward should be done outside autocast
|
||||
if backward:
|
||||
out.backward(g, retain_graph=True)
|
||||
activities = ([torch.profiler.ProfilerActivity.CPU] if cpu else []) + [
|
||||
torch.profiler.ProfilerActivity.CUDA
|
||||
]
|
||||
with torch.profiler.profile(
|
||||
activities=activities,
|
||||
record_shapes=True,
|
||||
# profile_memory=True,
|
||||
with_stack=True,
|
||||
) as prof:
|
||||
if backward:
|
||||
for x in inputs:
|
||||
if isinstance(x, torch.Tensor):
|
||||
x.grad = None
|
||||
with torch.autocast(device_type="cuda", dtype=amp_dtype, enabled=amp):
|
||||
out = fn(*inputs, **kwinputs)
|
||||
if type(out) is tuple:
|
||||
out = out[0]
|
||||
if backward:
|
||||
out.backward(g, retain_graph=True)
|
||||
if verbose:
|
||||
# print(prof.key_averages().table(sort_by="self_cuda_time_total", row_limit=50))
|
||||
print(prof.key_averages().table(row_limit=50))
|
||||
if trace_filename is not None:
|
||||
prof.export_chrome_trace(trace_filename)
|
||||
|
||||
|
||||
def benchmark_memory(fn, *inputs, desc="", verbose=True, **kwinputs):
|
||||
torch.cuda.empty_cache()
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
torch.cuda.synchronize()
|
||||
fn(*inputs, **kwinputs)
|
||||
torch.cuda.synchronize()
|
||||
mem = torch.cuda.max_memory_allocated() / ((2**20) * 1000)
|
||||
if verbose:
|
||||
print(f"{desc} max memory: {mem}GB")
|
||||
torch.cuda.empty_cache()
|
||||
return mem
|
||||
@@ -0,0 +1,83 @@
|
||||
import torch
|
||||
from torch import nn
|
||||
|
||||
from benchmark import (
|
||||
benchmark_combined,
|
||||
benchmark_forward,
|
||||
benchmark_backward,
|
||||
assert_close,
|
||||
)
|
||||
from fastvideo.ops.modulate.modulate import fused_modulate
|
||||
|
||||
|
||||
def torch_modulate(x, scale, shift):
|
||||
return x * (scale.unsqueeze(1) + 1) + shift.unsqueeze(1)
|
||||
|
||||
|
||||
# from flash_attn import flash_attn_func
|
||||
|
||||
|
||||
def time_fwd(func, *args, **kwargs):
|
||||
time_fb = benchmark_forward(func, *args, **kwargs)
|
||||
return time_fb[1].mean
|
||||
|
||||
|
||||
def time_fwd_bwd(func, *args, **kwargs):
|
||||
time_fb = benchmark_combined(func, *args, **kwargs)
|
||||
return time_fb[1].mean
|
||||
|
||||
|
||||
def time_bwd(func, *args, **kwargs):
|
||||
time_fb = benchmark_backward(func, *args, **kwargs)
|
||||
return time_fb[1].mean
|
||||
|
||||
|
||||
device = "cuda"
|
||||
dtype = torch.bfloat16
|
||||
|
||||
|
||||
batch_sizes = [1, 8, 32]
|
||||
seq_lengths = [128, 512, 1024]
|
||||
hidden_dims = [768, 1024, 2048]
|
||||
|
||||
# methods = (["torch", "triton", "thunderkitten"])
|
||||
methods = ["torch", "triton"]
|
||||
time_f = {}
|
||||
time_b = {}
|
||||
time_f_b = {}
|
||||
speed_f = {}
|
||||
speed_b = {}
|
||||
speed_f_b = {}
|
||||
for B in batch_sizes:
|
||||
for T in seq_lengths:
|
||||
for D in hidden_dims:
|
||||
config = (B, T, D)
|
||||
|
||||
norm_func = nn.LayerNorm(D, elementwise_affine=False, eps=1e-6)
|
||||
x = torch.randn(B, T, D, device="cuda", requires_grad=True, dtype=dtype)
|
||||
x = norm_func(x.to(torch.float32)).to(dtype)
|
||||
shift = torch.randn(B, D, device="cuda", requires_grad=True, dtype=dtype)
|
||||
scale = torch.randn(B, D, device="cuda", requires_grad=True, dtype=dtype)
|
||||
# test torch
|
||||
o_ref = torch_modulate(x, scale, shift)
|
||||
o_ref.sum().backward(retain_graph=True)
|
||||
f_b = time_fwd_bwd(torch_modulate, x, scale, shift, verbose=False)
|
||||
time_f_b[config, "torch"] = f_b
|
||||
# test triton
|
||||
o2 = fused_modulate(x, scale, shift)
|
||||
o2.sum().backward(retain_graph=True)
|
||||
f_b = time_fwd_bwd(fused_modulate, x, scale, shift, verbose=False)
|
||||
time_f_b[config, "triton"] = f_b
|
||||
# test if the results are close
|
||||
assert_close(" o", o_ref, o2, 0.005)
|
||||
# time_f_b[config, "thunderkitten"] = f_b
|
||||
|
||||
print(f"### batch size={B}, seq length={T}, B={B}, hidden dim={D} ###")
|
||||
for method in methods:
|
||||
# time_f_b[config, method] = time_f[config, method] + time_b[config, method]
|
||||
print(
|
||||
f"{method:>50} fwd + bwd:\t {time_f_b[config, method]*1000:>6.4f} ms "
|
||||
)
|
||||
|
||||
# with open('flash2_attn_time.plk', 'wb') as fp:
|
||||
# pickle.dump((speed_f, speed_b, speed_f_b), fp, protocol=pickle.HIGHEST_PROTOCOL)
|
||||
Reference in New Issue
Block a user