Compare commits

...
6 changed files with 600 additions and 0 deletions
@@ -0,0 +1,51 @@
{
"benchmark_id": "fastwan-t2v-1.3b-vsa-dmd-gb10-1gpu",
"description": "FastWan2.1 T2V 1.3B DMD/VSA local Nsight profile for one DGX Spark GB10 GPU",
"model": {
"model_path": "FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
"model_short_name": "FastWan2.1-T2V-1.3B-DMD-VSA"
},
"init_kwargs": {
"num_gpus": 1,
"sp_size": 1,
"tp_size": 1,
"vae_sp": false,
"vae_tiling": true,
"VSA_sparsity": 0.8,
"dmd_denoising_steps": [1000, 757, 522],
"enable_torch_compile": false,
"dit_cpu_offload": false,
"dit_layerwise_offload": false,
"vae_cpu_offload": false,
"text_encoder_cpu_offload": true,
"pin_cpu_memory": false,
"text_encoder_precisions": ["fp32"]
},
"generation_kwargs": {
"height": 480,
"width": 832,
"num_frames": 81,
"num_inference_steps": 3,
"seed": 1024,
"fps": 16,
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
},
"test_prompts": [
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
],
"run_config": {
"num_warmup_runs": 2,
"num_measurement_runs": 5,
"required_gpus": 1
},
"thresholds": {
"GB10": {
"max_generation_time_s": 120.0,
"max_peak_memory_mb": 30000.0
},
"default": {
"max_generation_time_s": 120.0,
"max_peak_memory_mb": 30000.0
}
}
}
@@ -0,0 +1,46 @@
{
"benchmark_id": "wan-t2v-1.3b-gb10-1gpu",
"description": "Wan2.1 T2V 1.3B local Nsight profile for one DGX Spark GB10 GPU",
"model": {
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
"model_short_name": "Wan2.1-T2V-1.3B"
},
"init_kwargs": {
"num_gpus": 1,
"flow_shift": 7.0,
"sp_size": 1,
"tp_size": 1,
"vae_sp": false,
"vae_tiling": true,
"text_encoder_precisions": ["fp32"]
},
"generation_kwargs": {
"height": 480,
"width": 832,
"num_frames": 45,
"num_inference_steps": 4,
"guidance_scale": 3,
"embedded_cfg_scale": 6,
"seed": 1024,
"fps": 24,
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
},
"test_prompts": [
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
],
"run_config": {
"num_warmup_runs": 2,
"num_measurement_runs": 5,
"required_gpus": 1
},
"thresholds": {
"GB10": {
"max_generation_time_s": 120.0,
"max_peak_memory_mb": 30000.0
},
"default": {
"max_generation_time_s": 120.0,
"max_peak_memory_mb": 30000.0
}
}
}
@@ -0,0 +1,11 @@
#!/usr/bin/env bash
# Profile FastWan2.1 T2V 1.3B DMD/VSA on one DGX Spark GB10 GPU with Nsight Systems.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
export CONFIG_PATH="${CONFIG_PATH:-${SCRIPT_DIR}/configs/fastwan-t2v-1.3b-vsa-dmd-gb10-1gpu.json}"
export OUTPUT_ROOT="${OUTPUT_ROOT:-outputs/nsys/gb10-vsa}"
export FASTVIDEO_ATTENTION_BACKEND="${FASTVIDEO_ATTENTION_BACKEND:-VIDEO_SPARSE_ATTN}"
exec "${SCRIPT_DIR}/profile_wan_t2v_1_3b_nsys.sh" "$@"
+10
View File
@@ -0,0 +1,10 @@
#!/usr/bin/env bash
# Profile Wan2.1 T2V 1.3B on one DGX Spark GB10 GPU with Nsight Systems.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
export CONFIG_PATH="${CONFIG_PATH:-${SCRIPT_DIR}/configs/wan-t2v-1.3b-gb10-1gpu.json}"
export OUTPUT_ROOT="${OUTPUT_ROOT:-outputs/nsys/gb10}"
exec "${SCRIPT_DIR}/profile_wan_t2v_1_3b_nsys.sh" "$@"
+134
View File
@@ -0,0 +1,134 @@
#!/usr/bin/env bash
# Profile a Wan2.1 T2V 1.3B inference benchmark with Nsight Systems.
#
# Intended environment:
# - FastVideo Python 3.10-3.12 environment
# - CUDA-visible NVIDIA GPU(s)
# - Nsight Systems CLI available on PATH
#
# Usage:
# bash scripts/performance/profile_wan_t2v_1_3b_nsys.sh
#
# Optional overrides:
# PYTHON_BIN=python3.12
# NSYS_BIN=nsys
# CONFIG_PATH=.buildkite/performance-benchmarks/tests/wan-t2v-1.3b.json
# OUTPUT_ROOT=outputs/nsys
# NSYS_EXTRA_ARGS="--some-nsys-flag=value"
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
cd "${REPO_ROOT}"
PYTHON_BIN="${PYTHON_BIN:-python}"
NSYS_BIN="${NSYS_BIN:-nsys}"
CONFIG_PATH="${CONFIG_PATH:-.buildkite/performance-benchmarks/tests/wan-t2v-1.3b.json}"
OUTPUT_ROOT="${OUTPUT_ROOT:-outputs/nsys}"
if ! command -v "${PYTHON_BIN}" >/dev/null 2>&1; then
echo "Python executable not found: ${PYTHON_BIN}" >&2
exit 1
fi
"${PYTHON_BIN}" - <<'PY'
import sys
if not ((3, 10) <= sys.version_info[:2] <= (3, 12)):
raise SystemExit(f"Expected Python 3.10-3.12, got {sys.version.split()[0]}")
PY
if ! command -v "${NSYS_BIN}" >/dev/null 2>&1; then
echo "Nsight Systems CLI not found: ${NSYS_BIN}. Install Nsight Systems before profiling." >&2
exit 1
fi
if [[ ! -f "${CONFIG_PATH}" ]]; then
echo "Benchmark config not found: ${CONFIG_PATH}" >&2
exit 1
fi
BENCHMARK_ID="$("${PYTHON_BIN}" - <<'PY' "${CONFIG_PATH}"
import json
import sys
with open(sys.argv[1], encoding="utf-8") as f:
print(json.load(f)["benchmark_id"])
PY
)"
TIMESTAMP="$(date -u +%Y%m%dT%H%M%SZ)"
RUN_DIR="${OUTPUT_ROOT}/${BENCHMARK_ID}/${TIMESTAMP}"
mkdir -p "${RUN_DIR}"
NSYS_ARGS=(
profile
--force-overwrite=true
--capture-range=cudaProfilerApi
--capture-range-end=stop
--sample=none
--cpuctxsw=none
--stats=true
--output "${RUN_DIR}/${BENCHMARK_ID}"
)
NSYS_TRACE="${NSYS_TRACE:-cuda,nvtx,cudnn,cublas,osrt}"
if [[ "${NSYS_ENABLE_NCCL_TRACE:-0}" == "1" ]]; then
if "${NSYS_BIN}" profile --help 2>&1 | grep -Eq "(^|[[:space:],'])nccl([[:space:],']|$)"; then
NSYS_TRACE="${NSYS_TRACE},nccl"
else
echo "Skipping NCCL trace: ${NSYS_BIN} does not list nccl as a supported trace domain."
fi
fi
NSYS_ARGS+=(--trace="${NSYS_TRACE}")
if "${NSYS_BIN}" profile --help 2>&1 | grep -q -- "--cuda-trace-scope"; then
NSYS_ARGS+=(--cuda-trace-scope=process-tree)
fi
if "${NSYS_BIN}" profile --help 2>&1 | grep -q -- "--trace-fork-before-exec"; then
NSYS_ARGS+=(--trace-fork-before-exec=true)
fi
if [[ -n "${NSYS_EXTRA_ARGS:-}" ]]; then
# shellcheck disable=SC2206
EXTRA_ARGS=(${NSYS_EXTRA_ARGS})
NSYS_ARGS+=("${EXTRA_ARGS[@]}")
fi
export TOKENIZERS_PARALLELISM="${TOKENIZERS_PARALLELISM:-false}"
export FASTVIDEO_STAGE_LOGGING="${FASTVIDEO_STAGE_LOGGING:-1}"
echo "Repo root: ${REPO_ROOT}"
echo "Python: $("${PYTHON_BIN}" --version)"
echo "Nsight Systems: $("${NSYS_BIN}" --version | head -n 1)"
echo "Config: ${CONFIG_PATH}"
echo "Output: ${RUN_DIR}"
"${PYTHON_BIN}" - <<'PY'
try:
import torch
except ImportError:
print("Torch: unavailable")
else:
print(f"Torch: {torch.__version__}")
print(f"Torch CUDA runtime: {torch.version.cuda}")
print(f"CUDA available: {torch.cuda.is_available()}")
if torch.cuda.is_available():
print(f"CUDA device count: {torch.cuda.device_count()}")
for idx in range(torch.cuda.device_count()):
print(
f"CUDA device {idx}: {torch.cuda.get_device_name(idx)} "
f"capability={torch.cuda.get_device_capability(idx)}"
)
PY
echo "Starting nsys profile..."
"${NSYS_BIN}" "${NSYS_ARGS[@]}" \
"${PYTHON_BIN}" scripts/performance/run_inference_profile_from_config.py \
--config "${CONFIG_PATH}" \
--output-dir "${RUN_DIR}/generated_videos" \
--summary-path "${RUN_DIR}/profile_summary.json"
echo "Done. Profile outputs are under ${RUN_DIR}"
@@ -0,0 +1,348 @@
# SPDX-License-Identifier: Apache-2.0
"""Run one config-driven inference benchmark under an nsys capture range."""
from __future__ import annotations
import argparse
import ctypes
import json
import logging
import os
import time
from collections.abc import Mapping
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
logger = logging.getLogger(__name__)
STAGE_METRIC_MAP: dict[str, str] = {
"TextEncodingStage": "text_encoder_time_s",
"DenoisingStage": "dit_time_s",
"DmdDenoisingStage": "dit_time_s",
"DecodingStage": "vae_decode_time_s",
}
_CUDART: ctypes.CDLL | None = None
_CUDART_LIBRARY_NAMES: tuple[str, ...] = (
"libcudart.so",
"libcudart.so.13",
"libcudart.so.12",
"libcudart.so.11.0",
)
def _load_cudart() -> ctypes.CDLL:
global _CUDART
if _CUDART is not None:
return _CUDART
errors: list[str] = []
for name in _CUDART_LIBRARY_NAMES:
try:
lib = ctypes.CDLL(name)
except OSError as exc:
errors.append(f"{name}: {exc}")
continue
lib.cudaProfilerStart.restype = ctypes.c_int
lib.cudaProfilerStop.restype = ctypes.c_int
_CUDART = lib
return lib
raise RuntimeError("Unable to load CUDA runtime for cudaProfilerApi capture range: " + "; ".join(errors))
def _cuda_profiler_start() -> None:
err = _load_cudart().cudaProfilerStart()
if err != 0:
raise RuntimeError(f"cudaProfilerStart failed with CUDA error code {err}")
def _cuda_profiler_stop() -> None:
err = _load_cudart().cudaProfilerStop()
if err != 0:
raise RuntimeError(f"cudaProfilerStop failed with CUDA error code {err}")
def _make_worker_cuda_profiler_start_payload() -> bytes:
import cloudpickle
library_names = _CUDART_LIBRARY_NAMES
def worker_cuda_profiler_start(worker_wrapper: Any) -> dict[str, Any]:
import ctypes
import os
for name in library_names:
try:
lib = ctypes.CDLL(name)
break
except OSError:
lib = None
if lib is None:
raise RuntimeError("Unable to load CUDA runtime in worker")
lib.cudaProfilerStart.restype = ctypes.c_int
err = lib.cudaProfilerStart()
if err != 0:
raise RuntimeError(f"cudaProfilerStart failed with CUDA error code {err}")
return {
"pid": os.getpid(),
"rank": getattr(worker_wrapper, "rpc_rank", None),
"event": "cudaProfilerStart",
}
return cloudpickle.dumps(worker_cuda_profiler_start)
def _make_worker_cuda_profiler_stop_payload() -> bytes:
import cloudpickle
library_names = _CUDART_LIBRARY_NAMES
def worker_cuda_profiler_stop(worker_wrapper: Any) -> dict[str, Any]:
import ctypes
import os
import torch
torch.cuda.synchronize()
for name in library_names:
try:
lib = ctypes.CDLL(name)
break
except OSError:
lib = None
if lib is None:
raise RuntimeError("Unable to load CUDA runtime in worker")
lib.cudaProfilerStop.restype = ctypes.c_int
err = lib.cudaProfilerStop()
if err != 0:
raise RuntimeError(f"cudaProfilerStop failed with CUDA error code {err}")
return {
"pid": os.getpid(),
"rank": getattr(worker_wrapper, "rpc_rank", None),
"event": "cudaProfilerStop",
}
return cloudpickle.dumps(worker_cuda_profiler_stop)
def _load_config(path: Path) -> dict[str, Any]:
with path.open(encoding="utf-8") as f:
return json.load(f)
def _remap_init_kwargs(init_kwargs: dict[str, Any]) -> dict[str, Any]:
remapped = dict(init_kwargs)
text_enc_prec = remapped.pop("text_encoder_precisions", None)
if text_enc_prec is not None:
remapped["text_encoder_precisions"] = tuple(text_enc_prec)
return remapped
def _extract_component_times(result: dict[str, Any]) -> dict[str, float | None]:
component_times: dict[str, float | None] = {
"text_encoder_time_s": None,
"dit_time_s": None,
"vae_decode_time_s": None,
}
logging_info = result.get("logging_info")
if logging_info is None:
return component_times
if isinstance(logging_info, Mapping):
stages: dict[str, Any] = logging_info.get("stages", {}) or {}
else:
stages = getattr(logging_info, "stages", {}) or {}
for stage_name, stage_data in stages.items():
metric_key = STAGE_METRIC_MAP.get(stage_name)
if metric_key is None:
continue
elapsed = stage_data.get("execution_time")
if elapsed is None:
continue
existing = component_times[metric_key]
component_times[metric_key] = elapsed if existing is None else existing + elapsed
return component_times
def _run_generation(
generator: Any,
prompt: str,
generation_kwargs: dict[str, Any],
) -> tuple[float, float, dict[str, float | None]]:
import torch
torch.cuda.synchronize()
start = time.perf_counter()
result = generator.generate_video(prompt, **generation_kwargs)
torch.cuda.synchronize()
elapsed = time.perf_counter() - start
peak_memory_mb = result.get("peak_memory_mb", 0.0) or 0.0
return elapsed, peak_memory_mb, _extract_component_times(result)
def _shutdown_executor(generator: Any | None) -> None:
from fastvideo.worker.multiproc_executor import MultiprocExecutor
if generator is None:
return
if isinstance(generator.executor, MultiprocExecutor):
generator.executor.shutdown()
def _start_capture(generator: Any) -> list[dict[str, Any]]:
from fastvideo.worker.multiproc_executor import MultiprocExecutor
worker_payload = _make_worker_cuda_profiler_start_payload()
_cuda_profiler_start()
responses: list[dict[str, Any]] = []
if isinstance(generator.executor, MultiprocExecutor):
responses = generator.executor.collective_rpc(worker_payload)
return responses
def _stop_capture(generator: Any) -> list[dict[str, Any]]:
import torch
from fastvideo.worker.multiproc_executor import MultiprocExecutor
worker_payload = _make_worker_cuda_profiler_stop_payload()
responses: list[dict[str, Any]] = []
if isinstance(generator.executor, MultiprocExecutor):
responses = generator.executor.collective_rpc(worker_payload)
torch.cuda.synchronize()
_cuda_profiler_stop()
return responses
def _write_summary(path: Path, summary: dict[str, Any]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8") as f:
json.dump(summary, f, indent=2)
f.write("\n")
def _collect_torch_runtime_info(torch_module: Any, num_devices: int) -> dict[str, Any]:
return {
"torch_version": torch_module.__version__,
"torch_cuda_version": torch_module.version.cuda,
"cuda_device_count": torch_module.cuda.device_count(),
"cuda_arch_list": torch_module.cuda.get_arch_list(),
"device_capabilities": [
list(torch_module.cuda.get_device_capability(i))
for i in range(num_devices)
],
"fastvideo_attention_backend": os.environ.get("FASTVIDEO_ATTENTION_BACKEND"),
}
def run_profile(args: argparse.Namespace) -> None:
global logger
import torch
from fastvideo import VideoGenerator
from fastvideo.logger import init_logger
logger = init_logger(__name__)
cfg = _load_config(args.config)
run_config = cfg.get("run_config", {})
required_gpus = int(run_config.get("required_gpus", 1))
available_gpus = torch.cuda.device_count()
if available_gpus < required_gpus:
raise RuntimeError(f"Need {required_gpus} CUDA GPUs for {cfg['benchmark_id']}, found {available_gpus}")
model_info = cfg["model"]
init_kwargs = _remap_init_kwargs(cfg.get("init_kwargs", {}))
generation_kwargs = dict(cfg.get("generation_kwargs", {}))
generation_kwargs["output_path"] = str(args.output_dir)
prompt = cfg.get("test_prompts", ["A cinematic video."])[0]
num_warmup = int(run_config.get("num_warmup_runs", 1))
args.output_dir.mkdir(parents=True, exist_ok=True)
os.environ["FASTVIDEO_STAGE_LOGGING"] = "1"
generator: VideoGenerator | None = None
capture_started = False
worker_start_events: list[dict[str, Any]] = []
worker_stop_events: list[dict[str, Any]] = []
try:
logger.info("Loading model %s with init kwargs: %s", model_info["model_path"], init_kwargs)
generator = VideoGenerator.from_pretrained(
model_path=model_info["model_path"],
**init_kwargs,
)
for i in range(num_warmup):
logger.info("Warmup run %d/%d", i + 1, num_warmup)
_run_generation(generator, prompt, generation_kwargs)
logger.info("Starting cudaProfilerApi capture range")
worker_start_events = _start_capture(generator)
capture_started = True
elapsed, peak_mb, component_times = _run_generation(generator, prompt, generation_kwargs)
finally:
if capture_started and generator is not None:
logger.info("Stopping cudaProfilerApi capture range")
worker_stop_events = _stop_capture(generator)
_shutdown_executor(generator)
device_names = [torch.cuda.get_device_name(i) for i in range(required_gpus)]
runtime_info = _collect_torch_runtime_info(torch, required_gpus)
throughput_fps = None
num_frames = generation_kwargs.get("num_frames")
if isinstance(num_frames, (int, float)) and elapsed > 0:
throughput_fps = num_frames / elapsed
summary = {
"benchmark_id": cfg["benchmark_id"],
"model_short_name": model_info.get("model_short_name", ""),
"model_path": model_info["model_path"],
"config_path": str(args.config),
"timestamp": datetime.now(timezone.utc).isoformat(),
"device_names": device_names,
"device_capabilities": runtime_info["device_capabilities"],
"available_gpus": available_gpus,
"required_gpus": required_gpus,
"torch_version": runtime_info["torch_version"],
"torch_cuda_version": runtime_info["torch_cuda_version"],
"cuda_arch_list": runtime_info["cuda_arch_list"],
"fastvideo_attention_backend": runtime_info["fastvideo_attention_backend"],
"num_warmup_runs": num_warmup,
"num_profiled_runs": 1,
"profiled_generation_time_s": round(elapsed, 3),
"throughput_fps": round(throughput_fps, 3) if throughput_fps is not None else None,
"peak_memory_mb": round(peak_mb, 1),
"component_times_s": component_times,
"generation_kwargs": generation_kwargs,
"worker_start_events": worker_start_events,
"worker_stop_events": worker_stop_events,
}
_write_summary(args.summary_path, summary)
logger.info("Profile summary written to %s", args.summary_path)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--config",
type=Path,
default=Path(".buildkite/performance-benchmarks/tests/wan-t2v-1.3b.json"),
help="Benchmark JSON config to profile.",
)
parser.add_argument(
"--output-dir",
type=Path,
required=True,
help="Directory for generated videos.",
)
parser.add_argument(
"--summary-path",
type=Path,
required=True,
help="Path for the profile summary JSON.",
)
return parser.parse_args()
if __name__ == "__main__":
run_profile(parse_args())