Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5b90064d15 | ||
|
|
718ed1537a | ||
|
|
842d26657b | ||
|
|
da53f27258 |
@@ -0,0 +1,51 @@
|
||||
{
|
||||
"benchmark_id": "fastwan-t2v-1.3b-vsa-dmd-gb10-1gpu",
|
||||
"description": "FastWan2.1 T2V 1.3B DMD/VSA local Nsight profile for one DGX Spark GB10 GPU",
|
||||
"model": {
|
||||
"model_path": "FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "FastWan2.1-T2V-1.3B-DMD-VSA"
|
||||
},
|
||||
"init_kwargs": {
|
||||
"num_gpus": 1,
|
||||
"sp_size": 1,
|
||||
"tp_size": 1,
|
||||
"vae_sp": false,
|
||||
"vae_tiling": true,
|
||||
"VSA_sparsity": 0.8,
|
||||
"dmd_denoising_steps": [1000, 757, 522],
|
||||
"enable_torch_compile": false,
|
||||
"dit_cpu_offload": false,
|
||||
"dit_layerwise_offload": false,
|
||||
"vae_cpu_offload": false,
|
||||
"text_encoder_cpu_offload": true,
|
||||
"pin_cpu_memory": false,
|
||||
"text_encoder_precisions": ["fp32"]
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 81,
|
||||
"num_inference_steps": 3,
|
||||
"seed": 1024,
|
||||
"fps": 16,
|
||||
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
|
||||
},
|
||||
"test_prompts": [
|
||||
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
|
||||
],
|
||||
"run_config": {
|
||||
"num_warmup_runs": 2,
|
||||
"num_measurement_runs": 5,
|
||||
"required_gpus": 1
|
||||
},
|
||||
"thresholds": {
|
||||
"GB10": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 30000.0
|
||||
},
|
||||
"default": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 30000.0
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-gb10-1gpu",
|
||||
"description": "Wan2.1 T2V 1.3B local Nsight profile for one DGX Spark GB10 GPU",
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "Wan2.1-T2V-1.3B"
|
||||
},
|
||||
"init_kwargs": {
|
||||
"num_gpus": 1,
|
||||
"flow_shift": 7.0,
|
||||
"sp_size": 1,
|
||||
"tp_size": 1,
|
||||
"vae_sp": false,
|
||||
"vae_tiling": true,
|
||||
"text_encoder_precisions": ["fp32"]
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 45,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 3,
|
||||
"embedded_cfg_scale": 6,
|
||||
"seed": 1024,
|
||||
"fps": 24,
|
||||
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
|
||||
},
|
||||
"test_prompts": [
|
||||
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
|
||||
],
|
||||
"run_config": {
|
||||
"num_warmup_runs": 2,
|
||||
"num_measurement_runs": 5,
|
||||
"required_gpus": 1
|
||||
},
|
||||
"thresholds": {
|
||||
"GB10": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 30000.0
|
||||
},
|
||||
"default": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 30000.0
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
#!/usr/bin/env bash
|
||||
# Profile FastWan2.1 T2V 1.3B DMD/VSA on one DGX Spark GB10 GPU with Nsight Systems.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
export CONFIG_PATH="${CONFIG_PATH:-${SCRIPT_DIR}/configs/fastwan-t2v-1.3b-vsa-dmd-gb10-1gpu.json}"
|
||||
export OUTPUT_ROOT="${OUTPUT_ROOT:-outputs/nsys/gb10-vsa}"
|
||||
export FASTVIDEO_ATTENTION_BACKEND="${FASTVIDEO_ATTENTION_BACKEND:-VIDEO_SPARSE_ATTN}"
|
||||
|
||||
exec "${SCRIPT_DIR}/profile_wan_t2v_1_3b_nsys.sh" "$@"
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
#!/usr/bin/env bash
|
||||
# Profile Wan2.1 T2V 1.3B on one DGX Spark GB10 GPU with Nsight Systems.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
export CONFIG_PATH="${CONFIG_PATH:-${SCRIPT_DIR}/configs/wan-t2v-1.3b-gb10-1gpu.json}"
|
||||
export OUTPUT_ROOT="${OUTPUT_ROOT:-outputs/nsys/gb10}"
|
||||
|
||||
exec "${SCRIPT_DIR}/profile_wan_t2v_1_3b_nsys.sh" "$@"
|
||||
+134
@@ -0,0 +1,134 @@
|
||||
#!/usr/bin/env bash
|
||||
# Profile a Wan2.1 T2V 1.3B inference benchmark with Nsight Systems.
|
||||
#
|
||||
# Intended environment:
|
||||
# - FastVideo Python 3.10-3.12 environment
|
||||
# - CUDA-visible NVIDIA GPU(s)
|
||||
# - Nsight Systems CLI available on PATH
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/performance/profile_wan_t2v_1_3b_nsys.sh
|
||||
#
|
||||
# Optional overrides:
|
||||
# PYTHON_BIN=python3.12
|
||||
# NSYS_BIN=nsys
|
||||
# CONFIG_PATH=.buildkite/performance-benchmarks/tests/wan-t2v-1.3b.json
|
||||
# OUTPUT_ROOT=outputs/nsys
|
||||
# NSYS_EXTRA_ARGS="--some-nsys-flag=value"
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
|
||||
cd "${REPO_ROOT}"
|
||||
|
||||
PYTHON_BIN="${PYTHON_BIN:-python}"
|
||||
NSYS_BIN="${NSYS_BIN:-nsys}"
|
||||
CONFIG_PATH="${CONFIG_PATH:-.buildkite/performance-benchmarks/tests/wan-t2v-1.3b.json}"
|
||||
OUTPUT_ROOT="${OUTPUT_ROOT:-outputs/nsys}"
|
||||
|
||||
if ! command -v "${PYTHON_BIN}" >/dev/null 2>&1; then
|
||||
echo "Python executable not found: ${PYTHON_BIN}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
"${PYTHON_BIN}" - <<'PY'
|
||||
import sys
|
||||
|
||||
if not ((3, 10) <= sys.version_info[:2] <= (3, 12)):
|
||||
raise SystemExit(f"Expected Python 3.10-3.12, got {sys.version.split()[0]}")
|
||||
PY
|
||||
|
||||
if ! command -v "${NSYS_BIN}" >/dev/null 2>&1; then
|
||||
echo "Nsight Systems CLI not found: ${NSYS_BIN}. Install Nsight Systems before profiling." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ ! -f "${CONFIG_PATH}" ]]; then
|
||||
echo "Benchmark config not found: ${CONFIG_PATH}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
BENCHMARK_ID="$("${PYTHON_BIN}" - <<'PY' "${CONFIG_PATH}"
|
||||
import json
|
||||
import sys
|
||||
|
||||
with open(sys.argv[1], encoding="utf-8") as f:
|
||||
print(json.load(f)["benchmark_id"])
|
||||
PY
|
||||
)"
|
||||
TIMESTAMP="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||
RUN_DIR="${OUTPUT_ROOT}/${BENCHMARK_ID}/${TIMESTAMP}"
|
||||
|
||||
mkdir -p "${RUN_DIR}"
|
||||
|
||||
NSYS_ARGS=(
|
||||
profile
|
||||
--force-overwrite=true
|
||||
--capture-range=cudaProfilerApi
|
||||
--capture-range-end=stop
|
||||
--sample=none
|
||||
--cpuctxsw=none
|
||||
--stats=true
|
||||
--output "${RUN_DIR}/${BENCHMARK_ID}"
|
||||
)
|
||||
|
||||
NSYS_TRACE="${NSYS_TRACE:-cuda,nvtx,cudnn,cublas,osrt}"
|
||||
|
||||
if [[ "${NSYS_ENABLE_NCCL_TRACE:-0}" == "1" ]]; then
|
||||
if "${NSYS_BIN}" profile --help 2>&1 | grep -Eq "(^|[[:space:],'])nccl([[:space:],']|$)"; then
|
||||
NSYS_TRACE="${NSYS_TRACE},nccl"
|
||||
else
|
||||
echo "Skipping NCCL trace: ${NSYS_BIN} does not list nccl as a supported trace domain."
|
||||
fi
|
||||
fi
|
||||
|
||||
NSYS_ARGS+=(--trace="${NSYS_TRACE}")
|
||||
|
||||
if "${NSYS_BIN}" profile --help 2>&1 | grep -q -- "--cuda-trace-scope"; then
|
||||
NSYS_ARGS+=(--cuda-trace-scope=process-tree)
|
||||
fi
|
||||
|
||||
if "${NSYS_BIN}" profile --help 2>&1 | grep -q -- "--trace-fork-before-exec"; then
|
||||
NSYS_ARGS+=(--trace-fork-before-exec=true)
|
||||
fi
|
||||
|
||||
if [[ -n "${NSYS_EXTRA_ARGS:-}" ]]; then
|
||||
# shellcheck disable=SC2206
|
||||
EXTRA_ARGS=(${NSYS_EXTRA_ARGS})
|
||||
NSYS_ARGS+=("${EXTRA_ARGS[@]}")
|
||||
fi
|
||||
|
||||
export TOKENIZERS_PARALLELISM="${TOKENIZERS_PARALLELISM:-false}"
|
||||
export FASTVIDEO_STAGE_LOGGING="${FASTVIDEO_STAGE_LOGGING:-1}"
|
||||
|
||||
echo "Repo root: ${REPO_ROOT}"
|
||||
echo "Python: $("${PYTHON_BIN}" --version)"
|
||||
echo "Nsight Systems: $("${NSYS_BIN}" --version | head -n 1)"
|
||||
echo "Config: ${CONFIG_PATH}"
|
||||
echo "Output: ${RUN_DIR}"
|
||||
"${PYTHON_BIN}" - <<'PY'
|
||||
try:
|
||||
import torch
|
||||
except ImportError:
|
||||
print("Torch: unavailable")
|
||||
else:
|
||||
print(f"Torch: {torch.__version__}")
|
||||
print(f"Torch CUDA runtime: {torch.version.cuda}")
|
||||
print(f"CUDA available: {torch.cuda.is_available()}")
|
||||
if torch.cuda.is_available():
|
||||
print(f"CUDA device count: {torch.cuda.device_count()}")
|
||||
for idx in range(torch.cuda.device_count()):
|
||||
print(
|
||||
f"CUDA device {idx}: {torch.cuda.get_device_name(idx)} "
|
||||
f"capability={torch.cuda.get_device_capability(idx)}"
|
||||
)
|
||||
PY
|
||||
echo "Starting nsys profile..."
|
||||
|
||||
"${NSYS_BIN}" "${NSYS_ARGS[@]}" \
|
||||
"${PYTHON_BIN}" scripts/performance/run_inference_profile_from_config.py \
|
||||
--config "${CONFIG_PATH}" \
|
||||
--output-dir "${RUN_DIR}/generated_videos" \
|
||||
--summary-path "${RUN_DIR}/profile_summary.json"
|
||||
|
||||
echo "Done. Profile outputs are under ${RUN_DIR}"
|
||||
@@ -0,0 +1,348 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Run one config-driven inference benchmark under an nsys capture range."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import ctypes
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from collections.abc import Mapping
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
STAGE_METRIC_MAP: dict[str, str] = {
|
||||
"TextEncodingStage": "text_encoder_time_s",
|
||||
"DenoisingStage": "dit_time_s",
|
||||
"DmdDenoisingStage": "dit_time_s",
|
||||
"DecodingStage": "vae_decode_time_s",
|
||||
}
|
||||
|
||||
_CUDART: ctypes.CDLL | None = None
|
||||
_CUDART_LIBRARY_NAMES: tuple[str, ...] = (
|
||||
"libcudart.so",
|
||||
"libcudart.so.13",
|
||||
"libcudart.so.12",
|
||||
"libcudart.so.11.0",
|
||||
)
|
||||
|
||||
|
||||
def _load_cudart() -> ctypes.CDLL:
|
||||
global _CUDART
|
||||
if _CUDART is not None:
|
||||
return _CUDART
|
||||
|
||||
errors: list[str] = []
|
||||
for name in _CUDART_LIBRARY_NAMES:
|
||||
try:
|
||||
lib = ctypes.CDLL(name)
|
||||
except OSError as exc:
|
||||
errors.append(f"{name}: {exc}")
|
||||
continue
|
||||
lib.cudaProfilerStart.restype = ctypes.c_int
|
||||
lib.cudaProfilerStop.restype = ctypes.c_int
|
||||
_CUDART = lib
|
||||
return lib
|
||||
|
||||
raise RuntimeError("Unable to load CUDA runtime for cudaProfilerApi capture range: " + "; ".join(errors))
|
||||
|
||||
|
||||
def _cuda_profiler_start() -> None:
|
||||
err = _load_cudart().cudaProfilerStart()
|
||||
if err != 0:
|
||||
raise RuntimeError(f"cudaProfilerStart failed with CUDA error code {err}")
|
||||
|
||||
|
||||
def _cuda_profiler_stop() -> None:
|
||||
err = _load_cudart().cudaProfilerStop()
|
||||
if err != 0:
|
||||
raise RuntimeError(f"cudaProfilerStop failed with CUDA error code {err}")
|
||||
|
||||
|
||||
def _make_worker_cuda_profiler_start_payload() -> bytes:
|
||||
import cloudpickle
|
||||
|
||||
library_names = _CUDART_LIBRARY_NAMES
|
||||
|
||||
def worker_cuda_profiler_start(worker_wrapper: Any) -> dict[str, Any]:
|
||||
import ctypes
|
||||
import os
|
||||
|
||||
for name in library_names:
|
||||
try:
|
||||
lib = ctypes.CDLL(name)
|
||||
break
|
||||
except OSError:
|
||||
lib = None
|
||||
if lib is None:
|
||||
raise RuntimeError("Unable to load CUDA runtime in worker")
|
||||
lib.cudaProfilerStart.restype = ctypes.c_int
|
||||
err = lib.cudaProfilerStart()
|
||||
if err != 0:
|
||||
raise RuntimeError(f"cudaProfilerStart failed with CUDA error code {err}")
|
||||
return {
|
||||
"pid": os.getpid(),
|
||||
"rank": getattr(worker_wrapper, "rpc_rank", None),
|
||||
"event": "cudaProfilerStart",
|
||||
}
|
||||
|
||||
return cloudpickle.dumps(worker_cuda_profiler_start)
|
||||
|
||||
|
||||
def _make_worker_cuda_profiler_stop_payload() -> bytes:
|
||||
import cloudpickle
|
||||
|
||||
library_names = _CUDART_LIBRARY_NAMES
|
||||
|
||||
def worker_cuda_profiler_stop(worker_wrapper: Any) -> dict[str, Any]:
|
||||
import ctypes
|
||||
import os
|
||||
import torch
|
||||
|
||||
torch.cuda.synchronize()
|
||||
for name in library_names:
|
||||
try:
|
||||
lib = ctypes.CDLL(name)
|
||||
break
|
||||
except OSError:
|
||||
lib = None
|
||||
if lib is None:
|
||||
raise RuntimeError("Unable to load CUDA runtime in worker")
|
||||
lib.cudaProfilerStop.restype = ctypes.c_int
|
||||
err = lib.cudaProfilerStop()
|
||||
if err != 0:
|
||||
raise RuntimeError(f"cudaProfilerStop failed with CUDA error code {err}")
|
||||
return {
|
||||
"pid": os.getpid(),
|
||||
"rank": getattr(worker_wrapper, "rpc_rank", None),
|
||||
"event": "cudaProfilerStop",
|
||||
}
|
||||
|
||||
return cloudpickle.dumps(worker_cuda_profiler_stop)
|
||||
|
||||
|
||||
def _load_config(path: Path) -> dict[str, Any]:
|
||||
with path.open(encoding="utf-8") as f:
|
||||
return json.load(f)
|
||||
|
||||
|
||||
def _remap_init_kwargs(init_kwargs: dict[str, Any]) -> dict[str, Any]:
|
||||
remapped = dict(init_kwargs)
|
||||
text_enc_prec = remapped.pop("text_encoder_precisions", None)
|
||||
if text_enc_prec is not None:
|
||||
remapped["text_encoder_precisions"] = tuple(text_enc_prec)
|
||||
return remapped
|
||||
|
||||
|
||||
def _extract_component_times(result: dict[str, Any]) -> dict[str, float | None]:
|
||||
component_times: dict[str, float | None] = {
|
||||
"text_encoder_time_s": None,
|
||||
"dit_time_s": None,
|
||||
"vae_decode_time_s": None,
|
||||
}
|
||||
logging_info = result.get("logging_info")
|
||||
if logging_info is None:
|
||||
return component_times
|
||||
if isinstance(logging_info, Mapping):
|
||||
stages: dict[str, Any] = logging_info.get("stages", {}) or {}
|
||||
else:
|
||||
stages = getattr(logging_info, "stages", {}) or {}
|
||||
for stage_name, stage_data in stages.items():
|
||||
metric_key = STAGE_METRIC_MAP.get(stage_name)
|
||||
if metric_key is None:
|
||||
continue
|
||||
elapsed = stage_data.get("execution_time")
|
||||
if elapsed is None:
|
||||
continue
|
||||
existing = component_times[metric_key]
|
||||
component_times[metric_key] = elapsed if existing is None else existing + elapsed
|
||||
return component_times
|
||||
|
||||
|
||||
def _run_generation(
|
||||
generator: Any,
|
||||
prompt: str,
|
||||
generation_kwargs: dict[str, Any],
|
||||
) -> tuple[float, float, dict[str, float | None]]:
|
||||
import torch
|
||||
|
||||
torch.cuda.synchronize()
|
||||
start = time.perf_counter()
|
||||
result = generator.generate_video(prompt, **generation_kwargs)
|
||||
torch.cuda.synchronize()
|
||||
elapsed = time.perf_counter() - start
|
||||
peak_memory_mb = result.get("peak_memory_mb", 0.0) or 0.0
|
||||
return elapsed, peak_memory_mb, _extract_component_times(result)
|
||||
|
||||
|
||||
def _shutdown_executor(generator: Any | None) -> None:
|
||||
from fastvideo.worker.multiproc_executor import MultiprocExecutor
|
||||
|
||||
if generator is None:
|
||||
return
|
||||
if isinstance(generator.executor, MultiprocExecutor):
|
||||
generator.executor.shutdown()
|
||||
|
||||
|
||||
def _start_capture(generator: Any) -> list[dict[str, Any]]:
|
||||
from fastvideo.worker.multiproc_executor import MultiprocExecutor
|
||||
|
||||
worker_payload = _make_worker_cuda_profiler_start_payload()
|
||||
_cuda_profiler_start()
|
||||
responses: list[dict[str, Any]] = []
|
||||
if isinstance(generator.executor, MultiprocExecutor):
|
||||
responses = generator.executor.collective_rpc(worker_payload)
|
||||
return responses
|
||||
|
||||
|
||||
def _stop_capture(generator: Any) -> list[dict[str, Any]]:
|
||||
import torch
|
||||
from fastvideo.worker.multiproc_executor import MultiprocExecutor
|
||||
|
||||
worker_payload = _make_worker_cuda_profiler_stop_payload()
|
||||
responses: list[dict[str, Any]] = []
|
||||
if isinstance(generator.executor, MultiprocExecutor):
|
||||
responses = generator.executor.collective_rpc(worker_payload)
|
||||
torch.cuda.synchronize()
|
||||
_cuda_profiler_stop()
|
||||
return responses
|
||||
|
||||
|
||||
def _write_summary(path: Path, summary: dict[str, Any]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8") as f:
|
||||
json.dump(summary, f, indent=2)
|
||||
f.write("\n")
|
||||
|
||||
|
||||
def _collect_torch_runtime_info(torch_module: Any, num_devices: int) -> dict[str, Any]:
|
||||
return {
|
||||
"torch_version": torch_module.__version__,
|
||||
"torch_cuda_version": torch_module.version.cuda,
|
||||
"cuda_device_count": torch_module.cuda.device_count(),
|
||||
"cuda_arch_list": torch_module.cuda.get_arch_list(),
|
||||
"device_capabilities": [
|
||||
list(torch_module.cuda.get_device_capability(i))
|
||||
for i in range(num_devices)
|
||||
],
|
||||
"fastvideo_attention_backend": os.environ.get("FASTVIDEO_ATTENTION_BACKEND"),
|
||||
}
|
||||
|
||||
|
||||
def run_profile(args: argparse.Namespace) -> None:
|
||||
global logger
|
||||
|
||||
import torch
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.logger import init_logger
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
cfg = _load_config(args.config)
|
||||
run_config = cfg.get("run_config", {})
|
||||
required_gpus = int(run_config.get("required_gpus", 1))
|
||||
available_gpus = torch.cuda.device_count()
|
||||
if available_gpus < required_gpus:
|
||||
raise RuntimeError(f"Need {required_gpus} CUDA GPUs for {cfg['benchmark_id']}, found {available_gpus}")
|
||||
|
||||
model_info = cfg["model"]
|
||||
init_kwargs = _remap_init_kwargs(cfg.get("init_kwargs", {}))
|
||||
generation_kwargs = dict(cfg.get("generation_kwargs", {}))
|
||||
generation_kwargs["output_path"] = str(args.output_dir)
|
||||
prompt = cfg.get("test_prompts", ["A cinematic video."])[0]
|
||||
num_warmup = int(run_config.get("num_warmup_runs", 1))
|
||||
|
||||
args.output_dir.mkdir(parents=True, exist_ok=True)
|
||||
os.environ["FASTVIDEO_STAGE_LOGGING"] = "1"
|
||||
|
||||
generator: VideoGenerator | None = None
|
||||
capture_started = False
|
||||
worker_start_events: list[dict[str, Any]] = []
|
||||
worker_stop_events: list[dict[str, Any]] = []
|
||||
try:
|
||||
logger.info("Loading model %s with init kwargs: %s", model_info["model_path"], init_kwargs)
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_path=model_info["model_path"],
|
||||
**init_kwargs,
|
||||
)
|
||||
|
||||
for i in range(num_warmup):
|
||||
logger.info("Warmup run %d/%d", i + 1, num_warmup)
|
||||
_run_generation(generator, prompt, generation_kwargs)
|
||||
|
||||
logger.info("Starting cudaProfilerApi capture range")
|
||||
worker_start_events = _start_capture(generator)
|
||||
capture_started = True
|
||||
elapsed, peak_mb, component_times = _run_generation(generator, prompt, generation_kwargs)
|
||||
finally:
|
||||
if capture_started and generator is not None:
|
||||
logger.info("Stopping cudaProfilerApi capture range")
|
||||
worker_stop_events = _stop_capture(generator)
|
||||
_shutdown_executor(generator)
|
||||
|
||||
device_names = [torch.cuda.get_device_name(i) for i in range(required_gpus)]
|
||||
runtime_info = _collect_torch_runtime_info(torch, required_gpus)
|
||||
throughput_fps = None
|
||||
num_frames = generation_kwargs.get("num_frames")
|
||||
if isinstance(num_frames, (int, float)) and elapsed > 0:
|
||||
throughput_fps = num_frames / elapsed
|
||||
|
||||
summary = {
|
||||
"benchmark_id": cfg["benchmark_id"],
|
||||
"model_short_name": model_info.get("model_short_name", ""),
|
||||
"model_path": model_info["model_path"],
|
||||
"config_path": str(args.config),
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"device_names": device_names,
|
||||
"device_capabilities": runtime_info["device_capabilities"],
|
||||
"available_gpus": available_gpus,
|
||||
"required_gpus": required_gpus,
|
||||
"torch_version": runtime_info["torch_version"],
|
||||
"torch_cuda_version": runtime_info["torch_cuda_version"],
|
||||
"cuda_arch_list": runtime_info["cuda_arch_list"],
|
||||
"fastvideo_attention_backend": runtime_info["fastvideo_attention_backend"],
|
||||
"num_warmup_runs": num_warmup,
|
||||
"num_profiled_runs": 1,
|
||||
"profiled_generation_time_s": round(elapsed, 3),
|
||||
"throughput_fps": round(throughput_fps, 3) if throughput_fps is not None else None,
|
||||
"peak_memory_mb": round(peak_mb, 1),
|
||||
"component_times_s": component_times,
|
||||
"generation_kwargs": generation_kwargs,
|
||||
"worker_start_events": worker_start_events,
|
||||
"worker_stop_events": worker_stop_events,
|
||||
}
|
||||
_write_summary(args.summary_path, summary)
|
||||
logger.info("Profile summary written to %s", args.summary_path)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument(
|
||||
"--config",
|
||||
type=Path,
|
||||
default=Path(".buildkite/performance-benchmarks/tests/wan-t2v-1.3b.json"),
|
||||
help="Benchmark JSON config to profile.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output-dir",
|
||||
type=Path,
|
||||
required=True,
|
||||
help="Directory for generated videos.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--summary-path",
|
||||
type=Path,
|
||||
required=True,
|
||||
help="Path for the profile summary JSON.",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
run_profile(parse_args())
|
||||
Reference in New Issue
Block a user