Compare commits
14
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
19d9ee06fe | ||
|
|
eb75e08f6e | ||
|
|
9bba72edab | ||
|
|
95a775b14a | ||
|
|
bc51cd6d37 | ||
|
|
f43bcdefae | ||
|
|
2e0dba1a4f | ||
|
|
dc8a661696 | ||
|
|
e03864a852 | ||
|
|
5902ab2669 | ||
|
|
53ce043b32 | ||
|
|
7b74b94ad4 | ||
|
|
752248e466 | ||
|
|
cc7848ab46 |
@@ -1,9 +1,9 @@
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v-1.3b",
|
||||
"variant_id": "canonical",
|
||||
"benchmark_version": 1,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"description": "Wan2.1 T2V 1.3B inference performance",
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { useEffect, useMemo, useState } from "react";
|
||||
|
||||
import { fetchSummary, fetchTrends, refreshData, RunSource, SummaryResponse, TrendGroup, TrendPoint } from "./api";
|
||||
import { fetchSummary, fetchTrends, refreshData } from "./api";
|
||||
import type { CohortValue, RunSource, SummaryResponse, TrendGroup, TrendPoint } from "./api";
|
||||
|
||||
const METRIC_KEYS = ["latency", "throughput", "memory", "text_encoder_time_s", "dit_time_s", "vae_decode_time_s"];
|
||||
const RUN_SOURCES: Array<{ value: "" | RunSource; label: string }> = [
|
||||
@@ -109,6 +110,61 @@ function metricLabel(metricKey: string) {
|
||||
return METRIC_DEFINITIONS[metricKey]?.label ?? metricKey;
|
||||
}
|
||||
|
||||
type CohortFields = {
|
||||
model_id: string;
|
||||
gpu_type: string;
|
||||
workload_id: CohortValue;
|
||||
variant_id: CohortValue;
|
||||
benchmark_version: CohortValue;
|
||||
recipe_fingerprint: CohortValue;
|
||||
hardware_profile_id: CohortValue;
|
||||
software_profile_id: CohortValue;
|
||||
};
|
||||
|
||||
function cohortValue(value: CohortValue) {
|
||||
if (value === null || value === undefined || value === "") {
|
||||
return "legacy";
|
||||
}
|
||||
return String(value);
|
||||
}
|
||||
|
||||
function shortCohortValue(value: CohortValue) {
|
||||
const text = cohortValue(value);
|
||||
if (text === "legacy" || text.length <= 14) {
|
||||
return text;
|
||||
}
|
||||
return text.slice(0, 12);
|
||||
}
|
||||
|
||||
function cohortKey(cohort: CohortFields) {
|
||||
return [
|
||||
cohort.model_id,
|
||||
cohort.gpu_type,
|
||||
cohortValue(cohort.workload_id),
|
||||
cohortValue(cohort.variant_id),
|
||||
cohortValue(cohort.benchmark_version),
|
||||
cohortValue(cohort.recipe_fingerprint),
|
||||
cohortValue(cohort.hardware_profile_id),
|
||||
cohortValue(cohort.software_profile_id)
|
||||
].join("|");
|
||||
}
|
||||
|
||||
function cohortTitle(cohort: CohortFields) {
|
||||
const workload = cohortValue(cohort.workload_id);
|
||||
const variant = cohortValue(cohort.variant_id);
|
||||
const version = cohortValue(cohort.benchmark_version);
|
||||
const versionLabel = version === "legacy" ? version : `v${version}`;
|
||||
return `${workload} / ${variant} / ${versionLabel}`;
|
||||
}
|
||||
|
||||
function cohortDetail(cohort: CohortFields) {
|
||||
return [
|
||||
`recipe ${shortCohortValue(cohort.recipe_fingerprint)}`,
|
||||
shortCohortValue(cohort.hardware_profile_id),
|
||||
shortCohortValue(cohort.software_profile_id)
|
||||
].join(" | ");
|
||||
}
|
||||
|
||||
function formatMetricValue(metricKey: string, value: number | null | undefined, tooltip = false) {
|
||||
const definition = METRIC_DEFINITIONS[metricKey];
|
||||
if (!definition) {
|
||||
@@ -171,7 +227,9 @@ function TrendChart({ group, metricKey }: { group: TrendGroup; metricKey: string
|
||||
top: `${(activePoint.y / height) * 100}%`
|
||||
}
|
||||
: undefined;
|
||||
const ariaLabel = `${metricLabel(metricKey)} trend for ${group.model_id} on ${group.gpu_type}`;
|
||||
const ariaLabel = `${metricLabel(metricKey)} trend for ${group.model_id} on ${group.gpu_type}, ${cohortTitle(
|
||||
group
|
||||
)}`;
|
||||
|
||||
return (
|
||||
<div className="chart-shell">
|
||||
@@ -419,7 +477,7 @@ export default function App() {
|
||||
<section className="panel">
|
||||
<div className="panel-header">
|
||||
<h2>Latest Status</h2>
|
||||
<span>{latestRows.length} model/GPU groups</span>
|
||||
<span>{latestRows.length} comparison cohorts</span>
|
||||
</div>
|
||||
{latestRows.length === 0 ? (
|
||||
<div className="empty">No records match the selected filters.</div>
|
||||
@@ -432,6 +490,7 @@ export default function App() {
|
||||
<th>Recomputed</th>
|
||||
<th>Model</th>
|
||||
<th>GPU</th>
|
||||
<th>Cohort</th>
|
||||
<th>Commit</th>
|
||||
<th>Source</th>
|
||||
<th>Baseline</th>
|
||||
@@ -446,7 +505,7 @@ export default function App() {
|
||||
</thead>
|
||||
<tbody>
|
||||
{latestRows.map((row) => (
|
||||
<tr key={`${row.model_id}-${row.gpu_type}`}>
|
||||
<tr key={cohortKey(row)}>
|
||||
<td>
|
||||
<span className={`badge ${row.status}`}>{row.status}</span>
|
||||
</td>
|
||||
@@ -457,6 +516,12 @@ export default function App() {
|
||||
</td>
|
||||
<td>{row.model_id}</td>
|
||||
<td>{row.gpu_type}</td>
|
||||
<td>
|
||||
<div className="cohort-cell">
|
||||
<strong>{cohortTitle(row)}</strong>
|
||||
<span>{cohortDetail(row)}</span>
|
||||
</div>
|
||||
</td>
|
||||
<td>{shortSha(row.commit_sha)}</td>
|
||||
<td>
|
||||
<span className={`source-badge source-${row.run_source}`}>{runSourceLabel(row.run_source)}</span>
|
||||
@@ -495,11 +560,13 @@ export default function App() {
|
||||
) : (
|
||||
trends.map((group) =>
|
||||
METRIC_KEYS.map((metricKey) => (
|
||||
<article className="trend-card" key={`${group.model_id}-${group.gpu_type}-${metricKey}`}>
|
||||
<article className="trend-card" key={`${cohortKey(group)}-${metricKey}`}>
|
||||
<div>
|
||||
<h3>{metricLabel(metricKey)}</h3>
|
||||
<p>
|
||||
{group.model_id} | {group.gpu_type}
|
||||
<span>{cohortTitle(group)}</span>
|
||||
<span>{cohortDetail(group)}</span>
|
||||
</p>
|
||||
</div>
|
||||
<TrendChart group={group} metricKey={metricKey} />
|
||||
|
||||
@@ -13,6 +13,17 @@ export type MetricValue = {
|
||||
precision: number;
|
||||
};
|
||||
|
||||
export type CohortValue = string | number | null;
|
||||
|
||||
export type ComparisonCohort = {
|
||||
workload_id: CohortValue;
|
||||
variant_id: CohortValue;
|
||||
benchmark_version: CohortValue;
|
||||
recipe_fingerprint: CohortValue;
|
||||
hardware_profile_id: CohortValue;
|
||||
software_profile_id: CohortValue;
|
||||
};
|
||||
|
||||
export type SummaryRow = {
|
||||
model_id: string;
|
||||
gpu_type: string;
|
||||
@@ -34,7 +45,7 @@ export type SummaryRow = {
|
||||
build_id: string;
|
||||
job_id: string;
|
||||
metrics: Record<string, MetricValue>;
|
||||
};
|
||||
} & ComparisonCohort;
|
||||
|
||||
export type RunSource = "pr" | "local" | "scheduled_main" | "unknown";
|
||||
|
||||
@@ -68,13 +79,13 @@ export type TrendPoint = {
|
||||
build_id: string;
|
||||
job_id: string;
|
||||
metrics: Record<string, number | null>;
|
||||
};
|
||||
} & ComparisonCohort;
|
||||
|
||||
export type TrendGroup = {
|
||||
model_id: string;
|
||||
gpu_type: string;
|
||||
points: TrendPoint[];
|
||||
};
|
||||
} & ComparisonCohort;
|
||||
|
||||
export type TrendsResponse = {
|
||||
groups: TrendGroup[];
|
||||
|
||||
@@ -149,6 +149,11 @@ h3 {
|
||||
font-size: 0.82rem;
|
||||
}
|
||||
|
||||
.trend-card p {
|
||||
display: grid;
|
||||
gap: 2px;
|
||||
}
|
||||
|
||||
.stat strong {
|
||||
display: block;
|
||||
margin-top: 8px;
|
||||
@@ -186,7 +191,7 @@ h3 {
|
||||
|
||||
table {
|
||||
width: 100%;
|
||||
min-width: 1120px;
|
||||
min-width: 1260px;
|
||||
border-collapse: collapse;
|
||||
}
|
||||
|
||||
@@ -209,6 +214,25 @@ td {
|
||||
font-size: 0.9rem;
|
||||
}
|
||||
|
||||
.cohort-cell {
|
||||
display: grid;
|
||||
gap: 2px;
|
||||
}
|
||||
|
||||
.cohort-cell strong,
|
||||
.trend-card p span {
|
||||
color: #1b2836;
|
||||
font-size: 0.78rem;
|
||||
font-weight: 700;
|
||||
}
|
||||
|
||||
.cohort-cell span,
|
||||
.trend-card p span + span {
|
||||
color: #607080;
|
||||
font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, "Liberation Mono", monospace;
|
||||
font-size: 0.72rem;
|
||||
}
|
||||
|
||||
.badge {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
|
||||
@@ -79,10 +79,11 @@ fastvideo/performance/
|
||||
```
|
||||
|
||||
The HF dataset (`FastVideo/performance-tracking` by default) holds one
|
||||
normalized JSON per `(model_id, gpu_type, run)` tuple. The rolling baseline is
|
||||
the median of the last 5 successful, baseline-eligible records for that
|
||||
model+GPU. PR and local records are visible in the dashboard but are not
|
||||
baseline eligible.
|
||||
normalized JSON per run. For v2 records, the rolling baseline is the median of
|
||||
the last 5 successful, baseline-eligible records in the same comparison cohort:
|
||||
`model_id`, `gpu_type`, `workload_id`, `variant_id`, `benchmark_version`,
|
||||
`recipe_fingerprint`, `hardware_profile_id`, and `software_profile_id`. PR and
|
||||
local records are visible in the dashboard but are not baseline eligible.
|
||||
|
||||
## Planned Coverage
|
||||
|
||||
@@ -158,13 +159,16 @@ unrealistic memory growth, and optionally large component-specific slowdowns
|
||||
even when the rolling baseline is empty. They are hand-set with generous
|
||||
headroom and almost never need touching.
|
||||
|
||||
### Rolling baseline (per `(model_id, gpu_type)`)
|
||||
### Rolling baseline (per comparison cohort)
|
||||
|
||||
`compare_baseline.py` loads the last 5 successful, baseline-eligible records
|
||||
for the same `(model_id, gpu_type)` from the HF dataset, computes the median
|
||||
for each available metric, and evaluates the current run with the metric's
|
||||
rolling regression policy. For latency, memory, and component times, higher
|
||||
values are regressions. For throughput, lower values are regressions.
|
||||
for the same comparison cohort from the HF dataset, computes the median for
|
||||
each available metric, and evaluates the current run with the metric's
|
||||
rolling regression policy. For v2 records, that cohort is `model_id`,
|
||||
`gpu_type`, `workload_id`, `variant_id`, `benchmark_version`,
|
||||
`recipe_fingerprint`, `hardware_profile_id`, and `software_profile_id`. For
|
||||
latency, memory, and component times, higher values are regressions. For
|
||||
throughput, lower values are regressions.
|
||||
|
||||
A metric exceeds its rolling threshold when both of these are true:
|
||||
|
||||
@@ -201,9 +205,9 @@ configs and remain loadable. New or migrated configs should use
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v-1.3b",
|
||||
"variant_id": "canonical",
|
||||
"benchmark_version": 1
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2
|
||||
}
|
||||
```
|
||||
|
||||
@@ -214,21 +218,27 @@ metadata that make the measured workload explicit:
|
||||
|
||||
| Field | Purpose |
|
||||
|---|---|
|
||||
| `workload_id` | Stable benchmark family, such as `wan-t2v-1.3b`. |
|
||||
| `variant_id` | Intentional recipe family, such as `canonical`. |
|
||||
| `workload_id` | Stable benchmark family, such as `wan-t2v`. |
|
||||
| `variant_id` | Intentional recipe family, including model size and parallelism config, such as `1.3b-sp2`. |
|
||||
| `benchmark_version` | Version of the measurement protocol and comparison policy. |
|
||||
|
||||
If a config declares `config_schema_version: 2`, loading fails clearly when any
|
||||
required v2 identity field is missing. If v2 identity or metadata fields are
|
||||
added without `config_schema_version: 2`, loading also fails so partial
|
||||
migrations do not silently run as v1 configs. Optional v2 metadata fields
|
||||
reserved for follow-up work, such as `recipe`, `metric_threshold_policy`, and
|
||||
`quality_metadata`, must be JSON objects when present.
|
||||
reserved for follow-up work, such as `metric_threshold_policy` and
|
||||
`quality_metadata`, must be JSON objects when present. (`recipe` is emitted
|
||||
by the harness and is not config-declarable.)
|
||||
|
||||
Recipe fingerprinting, hardware/software profile IDs, exact-identity
|
||||
comparison, metric-specific threshold policy behavior, promoted baselines, and
|
||||
dashboard regrouping are separate follow-up changes. Until those land, rolling
|
||||
baseline comparison remains keyed by `(model_id, gpu_type)`.
|
||||
comparison, and dashboard cohort grouping land with this change: v2 records
|
||||
compare only within their identity cohort, and a record that opens a NEW
|
||||
cohort is marked `baseline_status: "initialized_new_cohort"` (regression
|
||||
gating starts once that cohort accumulates history). Legacy v1 configs still
|
||||
run and are normalized for reporting, but their records skip rolling-baseline
|
||||
comparison entirely (`baseline_status: "skipped_missing_identity"`, never
|
||||
baseline eligible); only static thresholds gate them. Metric-specific
|
||||
threshold policies and promoted baselines remain separate follow-ups.
|
||||
|
||||
### Raw record (`results/perf_*.json`)
|
||||
|
||||
@@ -237,10 +247,10 @@ Written by `test_inference_performance.py`. One file per benchmark run.
|
||||
```jsonc
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v-1.3b",
|
||||
"variant_id": "canonical",
|
||||
"benchmark_version": 1,
|
||||
"result_schema_version": 2,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"model_short_name": "Wan2.1-T2V-1.3B-Diffusers",
|
||||
"device": "NVIDIA L40S",
|
||||
"num_gpus": 2,
|
||||
@@ -266,11 +276,54 @@ Written by `test_inference_performance.py`. One file per benchmark run.
|
||||
}
|
||||
},
|
||||
"commit": "<full sha>",
|
||||
"run_source": "pr",
|
||||
"branch": "feature/perf-change",
|
||||
"pr_number": "1234",
|
||||
"test_scope": "direct",
|
||||
"build_url": "https://buildkite.example/build",
|
||||
"build_id": "<buildkite-build-id>",
|
||||
"job_id": "<buildkite-job-id>",
|
||||
"timestamp": "2026-05-08T22:00:00+00:00",
|
||||
"quality_metadata": { "quality_status": "canonical" },
|
||||
"text_encoder_time_s": 2.141,
|
||||
"dit_time_s": 8.437,
|
||||
"vae_decode_time_s": 3.208
|
||||
"vae_decode_time_s": 3.208,
|
||||
"recipe": {
|
||||
"recipe_schema_version": 1,
|
||||
"benchmark": {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2
|
||||
},
|
||||
"model": { "model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" },
|
||||
"init_kwargs": { "num_gpus": 2, "sp_size": 2, "tp_size": 1 },
|
||||
"generation_kwargs": { "height": 480, "width": 832, "num_frames": 45 },
|
||||
"inputs": { "prompt_count": 1, "prompt_sha256": ["<measured-prompt-sha256>"] },
|
||||
"attention": { "requested_backend": "FLASH_ATTN", "resolved_backend": "FLASH_ATTN" }
|
||||
},
|
||||
"recipe_fingerprint": "<sha256>",
|
||||
"hardware_profile": {
|
||||
"device_type": "cuda",
|
||||
"gpu_count": 2,
|
||||
"gpus": [{ "name": "NVIDIA L40S", "memory_gb": 48, "compute_capability": "8.9" }],
|
||||
"interconnect": "none_or_partial"
|
||||
},
|
||||
"hardware_profile_id": "hw-<sha256-prefix>",
|
||||
"software_profile": {
|
||||
"python": "3.12",
|
||||
"pytorch": "2.12",
|
||||
"cuda": "13.0",
|
||||
"packages": {
|
||||
"fastvideo_kernel": "0.3.2",
|
||||
"flashinfer": "0.2.11",
|
||||
"nvidia_cutlass_dsl": "4.5.0",
|
||||
"triton": "3.4.1"
|
||||
}
|
||||
},
|
||||
"software_profile_id": "sw-<sha256-prefix>",
|
||||
"environment_metadata": { "env": { "IMAGE_VERSION": "py3.12-cuda13.0.0" } },
|
||||
"environment_fingerprint": "env-<sha256-prefix>"
|
||||
}
|
||||
```
|
||||
|
||||
@@ -282,6 +335,10 @@ result, used as the rolling-baseline source of truth.
|
||||
```jsonc
|
||||
{
|
||||
"model_id": "wan-t2v-1.3b-2gpu",
|
||||
"result_schema_version": 2,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"timestamp": "2026-05-08T22:00:00+00:00",
|
||||
"commit_sha": "<full sha>",
|
||||
"gpu_type": "NVIDIA L40S",
|
||||
@@ -298,17 +355,46 @@ result, used as the rolling-baseline source of truth.
|
||||
"gated": true
|
||||
}
|
||||
},
|
||||
"recipe_fingerprint": "<sha256>",
|
||||
"hardware_profile_id": "hw-<sha256-prefix>",
|
||||
"software_profile_id": "sw-<sha256-prefix>",
|
||||
"environment_fingerprint": "env-<sha256-prefix>",
|
||||
"run_source": "pr",
|
||||
"branch": "feature/perf-change",
|
||||
"pr_number": "1234",
|
||||
"test_scope": "direct",
|
||||
"build_url": "https://buildkite.example/build",
|
||||
"build_id": "<buildkite-build-id>",
|
||||
"job_id": "<buildkite-job-id>",
|
||||
"quality_metadata": { "quality_status": "canonical" },
|
||||
"success": true
|
||||
}
|
||||
```
|
||||
|
||||
### Compatibility with legacy records
|
||||
|
||||
Older records in the HF dataset may not have component timing fields. The
|
||||
comparator ignores missing or `null` metrics when computing a median, and the
|
||||
dashboard lists skipped plots for metric series that have no non-null values.
|
||||
Records missing both `run_source` and `baseline_eligible` are treated as legacy
|
||||
successful main/full-suite uploads and remain eligible for rolling baselines.
|
||||
Older records in the HF dataset may not have `result_schema_version`,
|
||||
component timing fields, or v2 identity/profile fields. Records without
|
||||
`result_schema_version` are treated as v1. The comparator ignores missing or
|
||||
`null` metrics when computing a median, and the dashboard lists skipped plots
|
||||
for metric series that have no non-null values. Records missing both
|
||||
`run_source` and `baseline_eligible` are treated as legacy successful
|
||||
main/full-suite uploads and remain eligible for rolling baselines.
|
||||
Current `perf_*.json` artifacts that lack the v2 comparison identity are
|
||||
normalized for reporting but skip rolling-baseline comparison and are not marked
|
||||
baseline eligible.
|
||||
|
||||
New records compare only against the same `model_id`, `gpu_type`,
|
||||
`workload_id`, `variant_id`, `benchmark_version`, `recipe_fingerprint`,
|
||||
`hardware_profile_id`, and `software_profile_id` cohort.
|
||||
`environment_metadata` and `environment_fingerprint` are audit data and are not
|
||||
part of the comparison key.
|
||||
The recipe prompt digests describe the prompts actually measured by the
|
||||
benchmark run; extra configured prompts are ignored unless the benchmark runner
|
||||
executes them.
|
||||
Software profile package cohorts keep exact versions for relevant
|
||||
attention/kernel packages, including FastVideo kernels, FlashAttention,
|
||||
FlashInfer, Cutlass DSL, SageAttention, Triton, and xFormers when installed.
|
||||
|
||||
## Environment variable reference
|
||||
|
||||
@@ -318,7 +404,7 @@ successful main/full-suite uploads and remain eligible for rolling baselines.
|
||||
| `PERF_REPORTS_DIR` | `/root/data/perf_reports` | `compare_baseline.py`, `dashboard.py` | Where the Markdown summary and Plotly HTML get written for Buildkite to pick up. |
|
||||
| `HF_REPO_ID` | `FastVideo/performance-tracking` | `fastvideo/performance/hf_store.py` | HF dataset repo holding rolling-baseline records. |
|
||||
| `HF_API_KEY`, `HUGGINGFACE_HUB_TOKEN`, `HF_TOKEN` | unset | `fastvideo/performance/hf_store.py` | Required for upload or private dataset reads. |
|
||||
| `PERF_RUN_SOURCE` | inferred | `compare_baseline.py` | Source metadata for uploaded records: `pr`, `local`, `scheduled_main`, or `unknown`. |
|
||||
| `PERF_RUN_SOURCE` | inferred | `compare_baseline.py`, `test_inference_performance.py` | Source metadata for uploaded records: `pr`, `local`, `scheduled_main`, or `unknown`. |
|
||||
| `PERF_UPLOAD_POLICY` | `never` | `compare_baseline.py` | Upload policy: `never`, `pass`, or `always`. |
|
||||
| `PERF_PYTEST_RC` | unset | `compare_baseline.py` | Static-threshold pytest exit code, used so scheduled-main failures can be uploaded with `success=false`. |
|
||||
| `TEST_SCOPE` | unset | `compare_baseline.py` | CI context used to infer scheduled-main runs together with `BUILDKITE_BRANCH=main`. |
|
||||
@@ -335,11 +421,14 @@ point is `fastvideo/tests/modal/pr_test.py:run_performance_tests` and the
|
||||
Buildkite artifact upload is in
|
||||
`.buildkite/scripts/pr_test.sh:upload_performance_artifacts`.
|
||||
|
||||
Each performance build runs pytest first. If that fixed-threshold phase fails,
|
||||
PR/direct runs skip `compare_baseline.py` because they only upload passing
|
||||
records. Scheduled-main runs still execute `compare_baseline.py` with
|
||||
`PERF_PYTEST_RC` set so the failed canonical attempt is visible in normalized
|
||||
JSON and dashboard history. The dashboard runs best-effort for observability.
|
||||
Each performance build runs pytest first. PR and direct runs only continue to
|
||||
`compare_baseline.py` when that fixed-threshold phase passes; if pytest fails,
|
||||
Markdown summaries and normalized JSON artifacts are not emitted. Scheduled
|
||||
main runs set `PERF_UPLOAD_POLICY=always`, so they still run
|
||||
`compare_baseline.py` (with `PERF_PYTEST_RC` set) after a fixed-threshold
|
||||
failure. Those failed scheduled main runs emit summaries and normalized
|
||||
records, upload records with `success=false`, and are excluded from future
|
||||
rolling baselines. The dashboard still runs best-effort for observability.
|
||||
When the rolling-baseline phase runs, it emits:
|
||||
|
||||
* **Markdown summary** — appended to `$GITHUB_STEP_SUMMARY` when that variable
|
||||
@@ -347,7 +436,7 @@ When the rolling-baseline phase runs, it emits:
|
||||
per-benchmark row with current vs. baseline values for latency, throughput,
|
||||
memory, text encoder time, DiT time, and VAE decode time.
|
||||
* **Plotly dashboard** — `dashboard_<sha>_<ts>.html` showing time-series for
|
||||
each metric grouped by `(model_id, gpu_type)`.
|
||||
each metric grouped by comparison cohort.
|
||||
* **Normalized records** — `normalized_perf_*.json`, one per benchmark.
|
||||
Useful as input to the
|
||||
[`reseed-performance-baseline`](https://github.com/hao-ai-lab/FastVideo/blob/main/.agents/skills/reseed-performance-baseline/SKILL.md)
|
||||
@@ -364,7 +453,7 @@ When the rolling-baseline phase runs, it emits:
|
||||
"benchmark_id": "<unique-id>",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "<stable-workload-id>",
|
||||
"variant_id": "canonical",
|
||||
"variant_id": "<variant, e.g. 1.3b-sp2>",
|
||||
"benchmark_version": 1,
|
||||
"model": { "model_path": "...", "model_short_name": "..." },
|
||||
"init_kwargs": { "num_gpus": 1, ... },
|
||||
@@ -391,7 +480,9 @@ When the rolling-baseline phase runs, it emits:
|
||||
|
||||
Legacy v1 configs without `config_schema_version` still load, but should not
|
||||
gain v2 identity or metadata fields until they are migrated to
|
||||
`config_schema_version: 2`.
|
||||
`config_schema_version: 2`. For v2 configs, `workload_id`, `variant_id`,
|
||||
and `benchmark_version` are part of the comparison key; benchmark runs
|
||||
fail if any of these identity fields are missing.
|
||||
|
||||
2. The pytest test auto-discovers all configs — no test code needed. CI
|
||||
picks it up on the next `/test performance` run.
|
||||
@@ -419,8 +510,8 @@ When the rolling-baseline phase runs, it emits:
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**"No baseline for ... Initializing"** — first run for this `(model_id,
|
||||
gpu_type)`. Run will pass and (if persisting) seed the first record.
|
||||
**"No baseline for ... Initializing"** — first run for this comparison cohort.
|
||||
Run will pass and (if persisting) seed the first record.
|
||||
|
||||
**Persistent failure right after a torch / kernel / image upgrade** —
|
||||
genuine regression *or* baseline drift. Compare the failing normalized record
|
||||
|
||||
@@ -268,16 +268,31 @@ def load_records_for_model(
|
||||
model_id: str,
|
||||
gpu_type: str | None = None,
|
||||
*,
|
||||
workload_id: str | None = None,
|
||||
variant_id: str | None = None,
|
||||
benchmark_version: str | None = None,
|
||||
recipe_fingerprint: str | None = None,
|
||||
hardware_profile_id: str | None = None,
|
||||
software_profile_id: str | None = None,
|
||||
last_n: int | None = None,
|
||||
successful_only: bool = True,
|
||||
baseline_eligible_only: bool = False,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return records for a specific *model_id*, optionally filtered by GPU.
|
||||
"""Return records for a specific *model_id*, optionally filtered by cohort.
|
||||
|
||||
Args:
|
||||
local_dir: Root directory previously populated by :func:`sync_from_hf`.
|
||||
model_id: Matches the ``model_id`` field inside each JSON record.
|
||||
gpu_type: When set, only records whose ``gpu_type`` matches are returned.
|
||||
workload_id: When set, only records from the same workload are returned.
|
||||
variant_id: When set, only records from the same workload variant are returned.
|
||||
benchmark_version: When set, only records from the same benchmark version are returned.
|
||||
recipe_fingerprint: When set, only records from the same benchmark
|
||||
recipe are returned.
|
||||
hardware_profile_id: When set, only records from the same hardware
|
||||
cohort are returned.
|
||||
software_profile_id: When set, only records from the same software
|
||||
cohort are returned.
|
||||
last_n: When set, return only the most recent *n* records (after all
|
||||
other filters). Useful for sliding-window baseline calculations.
|
||||
successful_only: Passed through to :func:`load_records`.
|
||||
@@ -299,6 +314,18 @@ def load_records_for_model(
|
||||
if gpu_type is not None:
|
||||
records = [r for r in records if r.get("gpu_type") == gpu_type]
|
||||
|
||||
identity_filters = {
|
||||
"workload_id": workload_id,
|
||||
"variant_id": variant_id,
|
||||
"benchmark_version": benchmark_version,
|
||||
"recipe_fingerprint": recipe_fingerprint,
|
||||
"hardware_profile_id": hardware_profile_id,
|
||||
"software_profile_id": software_profile_id,
|
||||
}
|
||||
for key, expected in identity_filters.items():
|
||||
if expected is not None:
|
||||
records = [r for r in records if str(r.get(key)) == str(expected)]
|
||||
|
||||
if last_n is not None:
|
||||
records = records[-last_n:]
|
||||
|
||||
|
||||
@@ -17,6 +17,15 @@ from fastvideo.performance.hf_store import is_baseline_eligible_record, safe_flo
|
||||
from fastvideo.performance.metric_policy import regression_delta, resolve_metric_policies
|
||||
|
||||
Record = dict[str, Any]
|
||||
CohortKey = tuple[str, str, str, str, str, str, str, str]
|
||||
COMPARISON_COHORT_KEYS = (
|
||||
"workload_id",
|
||||
"variant_id",
|
||||
"benchmark_version",
|
||||
"recipe_fingerprint",
|
||||
"hardware_profile_id",
|
||||
"software_profile_id",
|
||||
)
|
||||
|
||||
|
||||
def parse_timestamp(value: Any) -> datetime | None:
|
||||
@@ -77,15 +86,56 @@ def record_metadata(record: Record) -> Record:
|
||||
}
|
||||
|
||||
|
||||
def group_by_model_gpu(records: list[Record]) -> dict[tuple[str, str], list[Record]]:
|
||||
groups: dict[tuple[str, str], list[Record]] = defaultdict(list)
|
||||
def _cohort_metadata_value(value: Any) -> Any:
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, str) and not value.strip():
|
||||
return ""
|
||||
return value
|
||||
|
||||
|
||||
def _cohort_key_value(value: Any) -> str:
|
||||
return str(_cohort_metadata_value(value))
|
||||
|
||||
|
||||
def record_comparison_metadata(record: Record) -> Record:
|
||||
return {key: _cohort_metadata_value(record.get(key)) for key in COMPARISON_COHORT_KEYS}
|
||||
|
||||
|
||||
def comparison_cohort_key(record: Record) -> CohortKey:
|
||||
return (
|
||||
str(record.get("model_id") or "unknown"),
|
||||
str(record.get("gpu_type") or "unknown"),
|
||||
_cohort_key_value(record.get("workload_id")),
|
||||
_cohort_key_value(record.get("variant_id")),
|
||||
_cohort_key_value(record.get("benchmark_version")),
|
||||
_cohort_key_value(record.get("recipe_fingerprint")),
|
||||
_cohort_key_value(record.get("hardware_profile_id")),
|
||||
_cohort_key_value(record.get("software_profile_id")),
|
||||
)
|
||||
|
||||
|
||||
def group_by_comparison_cohort(records: list[Record]) -> dict[CohortKey, list[Record]]:
|
||||
groups: dict[CohortKey, list[Record]] = defaultdict(list)
|
||||
for record in records:
|
||||
model_id = str(record.get("model_id") or "unknown")
|
||||
gpu_type = str(record.get("gpu_type") or "unknown")
|
||||
groups[(model_id, gpu_type)].append(record)
|
||||
groups[comparison_cohort_key(record)].append(record)
|
||||
return {key: sorted(value, key=record_sort_key) for key, value in groups.items()}
|
||||
|
||||
|
||||
def comparison_sort_key(record: Record) -> CohortKey:
|
||||
return comparison_cohort_key(record)
|
||||
|
||||
|
||||
def latest_row_sort_key(row: Record) -> tuple[Any, ...]:
|
||||
return (row["status"] != "fail", *comparison_sort_key(row))
|
||||
|
||||
|
||||
def group_identity(key: CohortKey) -> tuple[str, str]:
|
||||
model_id = key[0]
|
||||
gpu_type = key[1]
|
||||
return model_id, gpu_type
|
||||
|
||||
|
||||
def baseline_value(records: list[Record], metric_key: str) -> float | None:
|
||||
values = [safe_float(record.get(metric_key)) for record in records]
|
||||
values = [value for value in values if value is not None]
|
||||
@@ -99,7 +149,8 @@ def build_latest_summary(records: list[Record],
|
||||
baseline_window: int = 5,
|
||||
run_source: str | None = None) -> list[Record]:
|
||||
rows: list[Record] = []
|
||||
for (model_id, gpu_type), group in group_by_model_gpu(records).items():
|
||||
for key, group in group_by_comparison_cohort(records).items():
|
||||
model_id, gpu_type = group_identity(key)
|
||||
latest_candidates = group
|
||||
if run_source:
|
||||
latest_candidates = [record for record in group if record_run_source(record) == run_source]
|
||||
@@ -107,9 +158,10 @@ def build_latest_summary(records: list[Record],
|
||||
continue
|
||||
|
||||
latest = latest_candidates[-1]
|
||||
latest_index = next(index for index, record in enumerate(group) if record is latest)
|
||||
baseline_pool = [
|
||||
record for record in group
|
||||
if record is not latest and record.get("success", True) and is_baseline_eligible_record(record)
|
||||
record for record in group[:latest_index]
|
||||
if record.get("success", True) and is_baseline_eligible_record(record)
|
||||
]
|
||||
baseline_records = baseline_pool[-baseline_window:]
|
||||
metric_policies = resolve_metric_policies(latest.get("regression_thresholds"))
|
||||
@@ -156,6 +208,7 @@ def build_latest_summary(records: list[Record],
|
||||
"timestamp": latest.get("timestamp"),
|
||||
"commit_sha": latest.get("commit_sha"),
|
||||
**record_metadata(latest),
|
||||
**record_comparison_metadata(latest),
|
||||
"success": success,
|
||||
"baseline_n": len(baseline_records),
|
||||
"worst_regression_pct": worst_regression,
|
||||
@@ -166,12 +219,13 @@ def build_latest_summary(records: list[Record],
|
||||
"metrics": metrics,
|
||||
})
|
||||
|
||||
return sorted(rows, key=lambda row: (row["status"] != "fail", row["model_id"], row["gpu_type"]))
|
||||
return sorted(rows, key=latest_row_sort_key)
|
||||
|
||||
|
||||
def build_trends(records: list[Record]) -> list[Record]:
|
||||
trends: list[Record] = []
|
||||
for (model_id, gpu_type), group in group_by_model_gpu(records).items():
|
||||
for key, group in group_by_comparison_cohort(records).items():
|
||||
model_id, gpu_type = group_identity(key)
|
||||
points = []
|
||||
for record in group:
|
||||
metric_policies = resolve_metric_policies(record.get("regression_thresholds"))
|
||||
@@ -179,6 +233,7 @@ def build_trends(records: list[Record]) -> list[Record]:
|
||||
"timestamp": record.get("timestamp"),
|
||||
"commit_sha": record.get("commit_sha"),
|
||||
**record_metadata(record),
|
||||
**record_comparison_metadata(record),
|
||||
"success": bool(record.get("success", True)),
|
||||
"metrics": {
|
||||
policy.key: safe_float(record.get(policy.key))
|
||||
@@ -189,6 +244,7 @@ def build_trends(records: list[Record]) -> list[Record]:
|
||||
trends.append({
|
||||
"model_id": model_id,
|
||||
"gpu_type": gpu_type,
|
||||
**record_comparison_metadata(group[-1]),
|
||||
"points": points,
|
||||
})
|
||||
return sorted(trends, key=lambda trend: (trend["model_id"], trend["gpu_type"]))
|
||||
return sorted(trends, key=comparison_sort_key)
|
||||
|
||||
@@ -5,7 +5,8 @@ This script:
|
||||
1) reads current benchmark results from fastvideo/tests/performance/results,
|
||||
2) syncs the canonical baseline from the configured HF dataset repo,
|
||||
3) compares each current record against the median of up to 5 prior
|
||||
baseline-eligible successful records (filtered by gpu_type),
|
||||
baseline-eligible successful records in the same workload/variant/version,
|
||||
GPU, recipe, hardware, and software cohort,
|
||||
4) writes normalized records back to the HF dataset repo according to
|
||||
PERF_UPLOAD_POLICY,
|
||||
5) exits non-zero if any gated metric exceeds both its percent and absolute
|
||||
@@ -64,6 +65,30 @@ PERF_REPORTS_DIR = os.environ.get("PERF_REPORTS_DIR", "/root/data/perf_reports")
|
||||
UPLOAD_POLICY = os.environ.get("PERF_UPLOAD_POLICY", "never").strip().lower()
|
||||
VALID_UPLOAD_POLICIES = {"never", "pass", "always"}
|
||||
VALID_RUN_SOURCES = {"pr", "local", "scheduled_main", "unknown"}
|
||||
IDENTITY_KEYS = (
|
||||
"result_schema_version",
|
||||
"workload_id",
|
||||
"variant_id",
|
||||
"benchmark_version",
|
||||
"recipe",
|
||||
"recipe_fingerprint",
|
||||
"hardware_profile",
|
||||
"hardware_profile_id",
|
||||
"software_profile",
|
||||
"software_profile_id",
|
||||
"environment_metadata",
|
||||
"environment_fingerprint",
|
||||
"quality_metadata",
|
||||
"variant_metadata",
|
||||
)
|
||||
COMPARISON_IDENTITY_KEYS = (
|
||||
"workload_id",
|
||||
"variant_id",
|
||||
"benchmark_version",
|
||||
"recipe_fingerprint",
|
||||
"hardware_profile_id",
|
||||
"software_profile_id",
|
||||
)
|
||||
|
||||
|
||||
def _should_persist_tracking() -> bool:
|
||||
@@ -121,21 +146,72 @@ def _result_failed_static_thresholds() -> bool:
|
||||
|
||||
|
||||
def _record_metadata(run_source: str, result: dict[str, Any]) -> dict[str, Any]:
|
||||
raw_run_source = str(result.get("run_source") or "").strip().lower()
|
||||
if raw_run_source in VALID_RUN_SOURCES:
|
||||
run_source = raw_run_source
|
||||
|
||||
pr_number = result.get("pr_number") or os.environ.get("BUILDKITE_PULL_REQUEST", "")
|
||||
if not _truthy_pr_number(str(pr_number)):
|
||||
pr_number = ""
|
||||
return {
|
||||
"run_source": run_source,
|
||||
"baseline_eligible": False,
|
||||
"branch": os.environ.get("BUILDKITE_BRANCH", ""),
|
||||
"branch": result.get("branch") or os.environ.get("BUILDKITE_BRANCH", ""),
|
||||
"pr_number": pr_number,
|
||||
"test_scope": os.environ.get("TEST_SCOPE", ""),
|
||||
"build_url": os.environ.get("BUILDKITE_BUILD_URL", ""),
|
||||
"build_id": os.environ.get("BUILDKITE_BUILD_ID", ""),
|
||||
"job_id": os.environ.get("BUILDKITE_JOB_ID", ""),
|
||||
"test_scope": result.get("test_scope") or os.environ.get("TEST_SCOPE", ""),
|
||||
"build_url": result.get("build_url") or os.environ.get("BUILDKITE_BUILD_URL", ""),
|
||||
"build_id": result.get("build_id") or os.environ.get("BUILDKITE_BUILD_ID", ""),
|
||||
"job_id": result.get("job_id") or os.environ.get("BUILDKITE_JOB_ID", ""),
|
||||
}
|
||||
|
||||
|
||||
def _identity_metadata(result: dict[str, Any]) -> dict[str, Any]:
|
||||
metadata = {
|
||||
key: result[key]
|
||||
for key in IDENTITY_KEYS
|
||||
if key in result and result[key] is not None
|
||||
}
|
||||
recipe = result.get("recipe")
|
||||
benchmark = recipe.get("benchmark") if isinstance(recipe, dict) else None
|
||||
if isinstance(benchmark, dict):
|
||||
for key in ("workload_id", "variant_id", "benchmark_version"):
|
||||
if key not in metadata and benchmark.get(key) is not None:
|
||||
metadata[key] = benchmark[key]
|
||||
return metadata
|
||||
|
||||
|
||||
def _comparison_identity_filters(record: dict[str, Any]) -> dict[str, str]:
|
||||
missing = [
|
||||
key for key in COMPARISON_IDENTITY_KEYS
|
||||
if key not in record or record[key] is None or (isinstance(record[key], str) and not record[key].strip())
|
||||
]
|
||||
if missing:
|
||||
raise ValueError("Performance record missing required comparison identity fields: " + ", ".join(missing))
|
||||
return {
|
||||
key: str(record[key])
|
||||
for key in COMPARISON_IDENTITY_KEYS
|
||||
}
|
||||
|
||||
|
||||
def _comparison_identity_filters_or_none(record: dict[str, Any]) -> dict[str, str] | None:
|
||||
try:
|
||||
return _comparison_identity_filters(record)
|
||||
except ValueError as exc:
|
||||
print("Skipping rolling baseline comparison for "
|
||||
f"{record.get('model_id', 'unknown')} on "
|
||||
f"{record.get('gpu_type', 'unknown')}: {exc}. "
|
||||
"The normalized artifact will still be written, but this record "
|
||||
"will not be baseline eligible.")
|
||||
return None
|
||||
|
||||
|
||||
def _format_identity_filters(filters: dict[str, str]) -> str:
|
||||
if not filters:
|
||||
return ""
|
||||
compact = ", ".join(f"{key}={value}" for key, value in filters.items())
|
||||
return f" ({compact})"
|
||||
|
||||
|
||||
def _load_current_results() -> list[dict[str, Any]]:
|
||||
pattern = os.path.join(RESULTS_DIR, "perf_*.json")
|
||||
records: list[dict[str, Any]] = []
|
||||
@@ -169,7 +245,7 @@ def normalize_performance_result(result: dict[str, Any]) -> dict[str, Any]:
|
||||
vae_decode_time = safe_float(result.get("vae_decode_time_s"))
|
||||
metric_policies = resolve_metric_policies(result.get("regression_thresholds"))
|
||||
|
||||
return {
|
||||
record = {
|
||||
"model_id": model_id,
|
||||
"timestamp": timestamp,
|
||||
"commit_sha": commit_sha,
|
||||
@@ -184,6 +260,8 @@ def normalize_performance_result(result: dict[str, Any]) -> dict[str, Any]:
|
||||
"success": True,
|
||||
**_record_metadata(_detect_run_source(), result),
|
||||
}
|
||||
record.update(_identity_metadata(result))
|
||||
return record
|
||||
|
||||
|
||||
def _normalize_record(result: dict[str, Any]) -> dict[str, Any]:
|
||||
@@ -417,29 +495,53 @@ def main() -> int:
|
||||
for raw in current_results:
|
||||
record = _normalize_record(raw)
|
||||
metric_policies = resolve_metric_policies(record.get("regression_thresholds"))
|
||||
identity_filters = _comparison_identity_filters_or_none(record)
|
||||
|
||||
baseline_records = load_records_for_model(
|
||||
TRACKING_ROOT,
|
||||
record["model_id"],
|
||||
record["gpu_type"],
|
||||
last_n=5,
|
||||
successful_only=True,
|
||||
baseline_eligible_only=True,
|
||||
)
|
||||
|
||||
if not baseline_records:
|
||||
print(f"No baseline for {record['model_id']} on "
|
||||
f"{record['gpu_type']}. Initializing...")
|
||||
baseline_records: list[dict[str, Any]] = []
|
||||
if identity_filters is None:
|
||||
# Records without the full v2 identity block skip rolling-baseline
|
||||
# comparison entirely: only the static thresholds gate them and
|
||||
# they never become baseline eligible.
|
||||
record["baseline_status"] = "skipped_missing_identity"
|
||||
failures: list[str] = []
|
||||
record["success"] = True
|
||||
else:
|
||||
failures = _check_regressions(record, baseline_records, metric_policies)
|
||||
baseline_records = load_records_for_model(
|
||||
TRACKING_ROOT,
|
||||
record["model_id"],
|
||||
record["gpu_type"],
|
||||
**identity_filters,
|
||||
last_n=5,
|
||||
successful_only=True,
|
||||
baseline_eligible_only=True,
|
||||
)
|
||||
|
||||
if not baseline_records:
|
||||
# A brand-new cohort has NO regression gating until history
|
||||
# accumulates — make that loud and machine-readable instead of
|
||||
# an indistinguishable pass, so a cohort shift (intended or
|
||||
# accidental, e.g. an identity-field change) never silently
|
||||
# blinds the comparison.
|
||||
record["baseline_status"] = "initialized_new_cohort"
|
||||
print("=" * 72)
|
||||
print(f"WARNING: NO BASELINE — initializing a NEW cohort for "
|
||||
f"{record['model_id']} on {record['gpu_type']}"
|
||||
f"{_format_identity_filters(identity_filters)}")
|
||||
print("Regression gating is INACTIVE for this cohort until "
|
||||
"baseline history accumulates. If this cohort shift is "
|
||||
"unexpected, check the identity fields above.")
|
||||
print("=" * 72)
|
||||
failures = []
|
||||
else:
|
||||
record["baseline_status"] = "compared"
|
||||
failures = _check_regressions(record, baseline_records, metric_policies)
|
||||
if static_threshold_failed:
|
||||
failures.append(f"{record['model_id']} fixed-threshold phase failed "
|
||||
f"(PERF_PYTEST_RC={os.environ.get('PERF_PYTEST_RC')})")
|
||||
|
||||
record["success"] = not failures
|
||||
record["baseline_eligible"] = _is_baseline_eligible(record["run_source"], record["success"])
|
||||
record["baseline_eligible"] = (
|
||||
identity_filters is not None and _is_baseline_eligible(record["run_source"], record["success"])
|
||||
)
|
||||
all_failures.extend(failures)
|
||||
|
||||
_write_normalized_artifact(record)
|
||||
|
||||
@@ -29,15 +29,76 @@ METRICS = (
|
||||
"dit_time_s",
|
||||
"vae_decode_time_s",
|
||||
)
|
||||
COMPARISON_COHORT_KEYS = (
|
||||
"workload_id",
|
||||
"variant_id",
|
||||
"benchmark_version",
|
||||
"recipe_fingerprint",
|
||||
"hardware_profile_id",
|
||||
"software_profile_id",
|
||||
)
|
||||
GROUP_KEYS = ("model_id", "gpu_type", *COMPARISON_COHORT_KEYS)
|
||||
|
||||
|
||||
def _cohort_value(value: object) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
try:
|
||||
if pd.isna(value):
|
||||
return ""
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
return str(value)
|
||||
|
||||
|
||||
def _display_value(value: object) -> str:
|
||||
return _cohort_value(value) or "legacy"
|
||||
|
||||
|
||||
def _short_value(value: object) -> str:
|
||||
text = _display_value(value)
|
||||
if text == "legacy" or len(text) <= 14:
|
||||
return text
|
||||
return text[:12]
|
||||
|
||||
|
||||
def _cohort_title(group_key: tuple[object, ...]) -> str:
|
||||
workload = _display_value(group_key[2])
|
||||
variant = _display_value(group_key[3])
|
||||
version = _display_value(group_key[4])
|
||||
version_label = version if version == "legacy" else f"v{version}"
|
||||
return f"{workload} / {variant} / {version_label}"
|
||||
|
||||
|
||||
def _cohort_detail(group_key: tuple[object, ...]) -> str:
|
||||
return " | ".join((
|
||||
f"recipe {_short_value(group_key[5])}",
|
||||
_short_value(group_key[6]),
|
||||
_short_value(group_key[7]),
|
||||
))
|
||||
|
||||
|
||||
def _group_record(group_key: tuple[object, ...]) -> dict[str, str]:
|
||||
return {
|
||||
key: _cohort_value(value)
|
||||
for key, value in zip(GROUP_KEYS, group_key)
|
||||
}
|
||||
|
||||
|
||||
def _dashboard_frame(df: pd.DataFrame) -> pd.DataFrame:
|
||||
dashboard_df = df.copy()
|
||||
for key in GROUP_KEYS:
|
||||
if key not in dashboard_df.columns:
|
||||
dashboard_df[key] = ""
|
||||
dashboard_df[key] = dashboard_df[key].map(_cohort_value)
|
||||
return dashboard_df
|
||||
|
||||
# -----------------------------
|
||||
# 1. Grouping
|
||||
# -----------------------------
|
||||
def group_data(df: pd.DataFrame):
|
||||
# Group only by model+GPU so each group produces a time-series line.
|
||||
# config_id (commit SHA) is carried as a column for hover/color use.
|
||||
keys = ["model_id", "gpu_type"]
|
||||
return df.groupby(keys, dropna=False)
|
||||
# Group by the same comparison cohort used by baseline gating.
|
||||
return _dashboard_frame(df).groupby(list(GROUP_KEYS), dropna=False)
|
||||
|
||||
# -----------------------------
|
||||
# 2. Plot builder
|
||||
@@ -46,15 +107,20 @@ def build_plots(df: pd.DataFrame) -> tuple[list, list[dict[str, object]]]:
|
||||
figs = []
|
||||
skipped_metrics: list[dict[str, object]] = []
|
||||
|
||||
for (model_id, gpu_type), g in group_data(df):
|
||||
for group_key, g in group_data(df):
|
||||
model_id, gpu_type = group_key[:2]
|
||||
group_record = _group_record(group_key)
|
||||
cohort_title = _cohort_title(group_key)
|
||||
cohort_detail = _cohort_detail(group_key)
|
||||
g = g.sort_values("timestamp")
|
||||
|
||||
# One chart per metric so the y-axes aren't on wildly different scales
|
||||
for metric in METRICS:
|
||||
if metric not in g.columns:
|
||||
skipped_metrics.append({
|
||||
"model_id": model_id,
|
||||
"gpu_type": gpu_type,
|
||||
**group_record,
|
||||
"cohort": cohort_title,
|
||||
"cohort_detail": cohort_detail,
|
||||
"metric": metric,
|
||||
"reason": "column missing from loaded records",
|
||||
"records": len(g),
|
||||
@@ -65,8 +131,9 @@ def build_plots(df: pd.DataFrame) -> tuple[list, list[dict[str, object]]]:
|
||||
non_null = int(g[metric].notna().sum())
|
||||
if non_null == 0:
|
||||
skipped_metrics.append({
|
||||
"model_id": model_id,
|
||||
"gpu_type": gpu_type,
|
||||
**group_record,
|
||||
"cohort": cohort_title,
|
||||
"cohort_detail": cohort_detail,
|
||||
"metric": metric,
|
||||
"reason": "no non-null values in loaded records",
|
||||
"records": len(g),
|
||||
@@ -79,8 +146,8 @@ def build_plots(df: pd.DataFrame) -> tuple[list, list[dict[str, object]]]:
|
||||
x="timestamp",
|
||||
y=metric,
|
||||
markers=True,
|
||||
hover_data=["config_id", "commit_sha"],
|
||||
title=f"{model_id} | {gpu_type} | {metric}",
|
||||
hover_data=["config_id", "commit_sha", *COMPARISON_COHORT_KEYS],
|
||||
title=f"{model_id} | {gpu_type} | {cohort_title} | {cohort_detail} | {metric}",
|
||||
labels={"timestamp": "Time", metric: metric},
|
||||
)
|
||||
figs.append(fig)
|
||||
@@ -95,7 +162,7 @@ def render_skipped_metrics(skipped_metrics: list[dict[str, object]]) -> str:
|
||||
rows = [
|
||||
"<h3>Skipped Metric Plots</h3>",
|
||||
"<table>",
|
||||
("<thead><tr><th>Model</th><th>GPU</th><th>Metric</th>"
|
||||
("<thead><tr><th>Model</th><th>GPU</th><th>Cohort</th><th>Metric</th>"
|
||||
"<th>Records</th><th>Non-null</th><th>Reason</th></tr></thead>"),
|
||||
"<tbody>",
|
||||
]
|
||||
@@ -104,6 +171,7 @@ def render_skipped_metrics(skipped_metrics: list[dict[str, object]]) -> str:
|
||||
"<tr>"
|
||||
f"<td>{escape(str(item['model_id']))}</td>"
|
||||
f"<td>{escape(str(item['gpu_type']))}</td>"
|
||||
f"<td>{escape(str(item['cohort']))}<br><code>{escape(str(item['cohort_detail']))}</code></td>"
|
||||
f"<td>{escape(str(item['metric']))}</td>"
|
||||
f"<td>{item['records']}</td>"
|
||||
f"<td>{item['non_null']}</td>"
|
||||
@@ -165,6 +233,7 @@ def main() -> None:
|
||||
print("Skipped metric plots:")
|
||||
for item in skipped_metrics:
|
||||
print(f" - {item['model_id']} | {item['gpu_type']} | "
|
||||
f"{item['cohort']} | {item['cohort_detail']} | "
|
||||
f"{item['metric']}: {item['reason']} "
|
||||
f"({item['non_null']}/{item['records']} non-null)")
|
||||
print(f"Generated {len(figs)} metric plot(s)")
|
||||
|
||||
@@ -0,0 +1,435 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Comparable identity helpers for performance benchmark records."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
import enum
|
||||
import hashlib
|
||||
import importlib.metadata
|
||||
import json
|
||||
import os
|
||||
import platform
|
||||
import re
|
||||
from collections.abc import Mapping, Sequence
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
|
||||
RECIPE_SCHEMA_VERSION = 1
|
||||
PROFILE_ID_LENGTH = 16
|
||||
_PATH_EXCLUDED_GENERATION_KEYS = {"output_path", "output_video_name"}
|
||||
_PROMPT_KEYS = {"prompt", "negative_prompt", "neg_prompt"}
|
||||
REQUIRED_BENCHMARK_IDENTITY_KEYS = (
|
||||
"benchmark_id",
|
||||
"workload_id",
|
||||
"variant_id",
|
||||
"benchmark_version",
|
||||
)
|
||||
_PROFILE_ENV_VARS = (
|
||||
"CUDA_VISIBLE_DEVICES",
|
||||
"FASTVIDEO_ATTENTION_BACKEND",
|
||||
"IMAGE_VERSION",
|
||||
"UV_TORCH_BACKEND",
|
||||
)
|
||||
_PACKAGE_DISTRIBUTIONS = {
|
||||
"fastvideo_kernel": ("fastvideo-kernel", "fastvideo_kernel"),
|
||||
"flash_attn": ("flash-attn", "flash_attn"),
|
||||
"flash_attn_4": ("flash-attn-4",),
|
||||
"flash_attention_fp4": ("flash-attention-fp4",),
|
||||
"flashinfer": ("flashinfer-python", "flashinfer"),
|
||||
"nvidia_cutlass_dsl": ("nvidia-cutlass-dsl", "cutlass-dsl"),
|
||||
"sageattention": ("sageattention",),
|
||||
"sageattn3": ("sageattn3",),
|
||||
"triton": ("triton",),
|
||||
"xformers": ("xformers",),
|
||||
}
|
||||
|
||||
|
||||
def canonical_json(payload: Any) -> str:
|
||||
"""Return deterministic, compact JSON for a recipe/profile payload."""
|
||||
|
||||
return json.dumps(
|
||||
_canonicalize(payload),
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=True,
|
||||
)
|
||||
|
||||
|
||||
def sha256_hexdigest(payload: str | bytes) -> str:
|
||||
if isinstance(payload, str):
|
||||
payload = payload.encode("utf-8")
|
||||
return hashlib.sha256(payload).hexdigest()
|
||||
|
||||
|
||||
def recipe_fingerprint(recipe: Mapping[str, Any]) -> str:
|
||||
# Runtime-resolved values are audit metadata, not identity: hashing the
|
||||
# resolved HF snapshot revision would churn the cohort (and via the
|
||||
# no-baseline init path, silently reset the rolling baseline) on ANY
|
||||
# upstream repo commit, including model-card edits with unchanged
|
||||
# weights. The fingerprint is pinned to declared inputs only; the
|
||||
# resolved values stay in the stored recipe for auditability.
|
||||
pruned = {
|
||||
key: (dict(value) if isinstance(value, Mapping) else value)
|
||||
for key, value in recipe.items()
|
||||
}
|
||||
model = pruned.get("model")
|
||||
if isinstance(model, dict):
|
||||
model.pop("resolved_revision", None)
|
||||
attention = pruned.get("attention")
|
||||
if isinstance(attention, dict):
|
||||
# Same reasoning as resolved_revision: with requested_backend="auto",
|
||||
# a silent runtime fallback (e.g. a broken flash-attn wheel) must NOT
|
||||
# open a fresh ungated cohort and seed degraded numbers as the new
|
||||
# baseline — it should fail the regression gate in the old cohort.
|
||||
# Intentional backend changes are declared via requested_backend,
|
||||
# which stays in the hash.
|
||||
attention.pop("resolved_backend", None)
|
||||
return sha256_hexdigest(canonical_json(pruned))
|
||||
|
||||
|
||||
def profile_id(prefix: str, profile: Mapping[str, Any]) -> str:
|
||||
return f"{prefix}-{sha256_hexdigest(canonical_json(profile))[:PROFILE_ID_LENGTH]}"
|
||||
|
||||
|
||||
def hardware_profile_id(profile: Mapping[str, Any]) -> str:
|
||||
return profile_id("hw", profile)
|
||||
|
||||
|
||||
def software_profile_id(profile: Mapping[str, Any]) -> str:
|
||||
return profile_id("sw", profile)
|
||||
|
||||
|
||||
def environment_fingerprint(metadata: Mapping[str, Any]) -> str:
|
||||
return profile_id("env", metadata)
|
||||
|
||||
|
||||
def benchmark_identity_from_config(cfg: Mapping[str, Any]) -> dict[str, Any]:
|
||||
missing = [
|
||||
key for key in REQUIRED_BENCHMARK_IDENTITY_KEYS
|
||||
if _none_if_empty(cfg.get(key)) is None
|
||||
]
|
||||
if missing:
|
||||
raise ValueError("Benchmark config missing required identity fields: " + ", ".join(missing))
|
||||
return {
|
||||
key: cfg[key]
|
||||
for key in REQUIRED_BENCHMARK_IDENTITY_KEYS
|
||||
}
|
||||
|
||||
|
||||
def build_recipe_from_benchmark_config(
|
||||
cfg: Mapping[str, Any],
|
||||
*,
|
||||
attention_backend: str | None = None,
|
||||
resolved_attention_backend: Any | None = None,
|
||||
resolved_model_revision: Any | None = None,
|
||||
measured_prompts: Sequence[Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a deterministic recipe document from a benchmark config.
|
||||
|
||||
The recipe captures fields that describe what benchmark workload was run. It
|
||||
intentionally excludes timings, timestamps, commit metadata, and output paths.
|
||||
"""
|
||||
|
||||
model = dict(cfg.get("model") or {})
|
||||
init_kwargs = dict(cfg.get("init_kwargs") or {})
|
||||
generation_kwargs = dict(cfg.get("generation_kwargs") or {})
|
||||
benchmark_identity = benchmark_identity_from_config(cfg)
|
||||
prompts = list(measured_prompts) if measured_prompts is not None else list(
|
||||
cfg.get("test_prompts") or ["A cinematic video."])
|
||||
if attention_backend is None:
|
||||
attention_backend = os.environ.get("FASTVIDEO_ATTENTION_BACKEND")
|
||||
|
||||
negative_prompt = generation_kwargs.pop("negative_prompt", generation_kwargs.pop("neg_prompt", None))
|
||||
generation_recipe = {
|
||||
key: value
|
||||
for key, value in generation_kwargs.items()
|
||||
if key not in _PATH_EXCLUDED_GENERATION_KEYS and key not in _PROMPT_KEYS
|
||||
}
|
||||
|
||||
return {
|
||||
"recipe_schema_version": RECIPE_SCHEMA_VERSION,
|
||||
"benchmark": benchmark_identity,
|
||||
"model": {
|
||||
"model_path": normalize_model_path(model.get("model_path")),
|
||||
"revision": _none_if_empty(model.get("revision") or cfg.get("revision")),
|
||||
"resolved_revision": _none_if_empty(resolved_model_revision),
|
||||
},
|
||||
"init_kwargs": init_kwargs,
|
||||
"generation_kwargs": generation_recipe,
|
||||
"inputs": {
|
||||
"prompt_count": len(prompts),
|
||||
"prompt_sha256": [_optional_digest(prompt) for prompt in prompts],
|
||||
"negative_prompt_sha256": _optional_digest(negative_prompt),
|
||||
},
|
||||
"attention": {
|
||||
"requested_backend": _none_if_empty(attention_backend) or "auto",
|
||||
"resolved_backend": _none_if_empty(resolved_attention_backend),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def normalize_model_path(value: Any) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
text = os.fspath(value) if isinstance(value, os.PathLike) else str(value)
|
||||
text = text.strip()
|
||||
if not text:
|
||||
return None
|
||||
if _looks_like_local_path(text):
|
||||
return Path(text).expanduser().as_posix()
|
||||
return re.sub(r"/+", "/", text)
|
||||
|
||||
|
||||
def resolved_revision_from_model_path(value: Any) -> str | None:
|
||||
normalized = normalize_model_path(value)
|
||||
if normalized is None:
|
||||
return None
|
||||
parts = Path(normalized).parts
|
||||
for index, part in enumerate(parts[:-1]):
|
||||
if part == "snapshots":
|
||||
revision = parts[index + 1]
|
||||
if re.fullmatch(r"[0-9a-f]{40}", revision):
|
||||
return revision
|
||||
return None
|
||||
|
||||
|
||||
def hardware_profile(
|
||||
*,
|
||||
num_gpus: int | None = None,
|
||||
gpu_devices: Sequence[Mapping[str, Any]] | None = None,
|
||||
device_type: str | None = None,
|
||||
interconnect: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Return normalized hardware cohort metadata.
|
||||
|
||||
``gpu_devices`` is an injection point for tests and non-CUDA callers.
|
||||
"""
|
||||
|
||||
if gpu_devices is not None:
|
||||
devices = [_normalize_gpu_device(device) for device in gpu_devices]
|
||||
return {
|
||||
"device_type": device_type or "cuda",
|
||||
"gpu_count": num_gpus if num_gpus is not None else len(devices),
|
||||
"gpus": devices,
|
||||
"interconnect": interconnect or "unknown",
|
||||
}
|
||||
|
||||
if not torch.cuda.is_available():
|
||||
return {
|
||||
"device_type": device_type or "cpu",
|
||||
"gpu_count": 0,
|
||||
"gpus": [],
|
||||
"interconnect": interconnect or "none",
|
||||
}
|
||||
|
||||
gpu_count = int(num_gpus if num_gpus is not None else torch.cuda.device_count())
|
||||
devices = []
|
||||
cuda_device_count = torch.cuda.device_count()
|
||||
for device_id in range(gpu_count):
|
||||
if device_id >= cuda_device_count:
|
||||
devices.append(_normalize_gpu_device({}))
|
||||
continue
|
||||
props = torch.cuda.get_device_properties(device_id)
|
||||
capability = torch.cuda.get_device_capability(device_id)
|
||||
devices.append(
|
||||
_normalize_gpu_device({
|
||||
"name": props.name,
|
||||
"memory_bytes": props.total_memory,
|
||||
"compute_capability": f"{capability[0]}.{capability[1]}",
|
||||
}))
|
||||
|
||||
return {
|
||||
"device_type": device_type or "cuda",
|
||||
"gpu_count": gpu_count,
|
||||
"gpus": devices,
|
||||
"interconnect": interconnect or _detect_cuda_interconnect(gpu_count),
|
||||
}
|
||||
|
||||
|
||||
def software_profile(
|
||||
*,
|
||||
python_version: str | None = None,
|
||||
torch_version: str | None = None,
|
||||
cuda_version: str | None = None,
|
||||
package_versions: Mapping[str, str | None] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Return normalized software cohort metadata.
|
||||
|
||||
Python, PyTorch, and CUDA use major/minor cohorts. Attention and kernel
|
||||
packages keep exact versions because patch releases can change performance.
|
||||
"""
|
||||
|
||||
versions = dict(package_versions) if package_versions is not None else _installed_package_versions()
|
||||
return {
|
||||
"python": _major_minor(python_version or platform.python_version()),
|
||||
"pytorch": _major_minor(torch_version or torch.__version__),
|
||||
"cuda": _major_minor(cuda_version or torch.version.cuda),
|
||||
"packages": {
|
||||
name: str(version)
|
||||
for name, version in sorted(versions.items())
|
||||
if version
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def environment_metadata(
|
||||
*,
|
||||
env: Mapping[str, str] | None = None,
|
||||
package_versions: Mapping[str, str | None] | None = None,
|
||||
hardware: Mapping[str, Any] | None = None,
|
||||
software: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Return audit metadata kept separate from comparable identity keys."""
|
||||
|
||||
source_env = env if env is not None else os.environ
|
||||
full_package_versions = dict(package_versions) if package_versions is not None else _installed_package_versions()
|
||||
return {
|
||||
"python": {
|
||||
"version": platform.python_version(),
|
||||
"implementation": platform.python_implementation(),
|
||||
},
|
||||
"platform": {
|
||||
"system": platform.system(),
|
||||
"machine": platform.machine(),
|
||||
"release": platform.release(),
|
||||
},
|
||||
"torch": {
|
||||
"version": torch.__version__,
|
||||
"cuda": torch.version.cuda,
|
||||
"cudnn": torch.backends.cudnn.version(),
|
||||
},
|
||||
"env": {
|
||||
key: source_env.get(key)
|
||||
for key in _PROFILE_ENV_VARS
|
||||
if source_env.get(key) is not None
|
||||
},
|
||||
"packages": {
|
||||
key: value
|
||||
for key, value in sorted(full_package_versions.items())
|
||||
if value
|
||||
},
|
||||
"hardware_profile": dict(hardware) if hardware is not None else hardware_profile(),
|
||||
"software_profile": dict(software) if software is not None else software_profile(
|
||||
package_versions=full_package_versions),
|
||||
}
|
||||
|
||||
|
||||
def _canonicalize(value: Any) -> Any:
|
||||
if dataclasses.is_dataclass(value) and not isinstance(value, type):
|
||||
return _canonicalize(dataclasses.asdict(value))
|
||||
if isinstance(value, Mapping):
|
||||
return {
|
||||
str(key): _canonicalize(value[key])
|
||||
for key in sorted(value, key=lambda item: str(item))
|
||||
}
|
||||
if isinstance(value, (set, frozenset)):
|
||||
return [_canonicalize(item) for item in sorted(value, key=lambda item: str(item))]
|
||||
if isinstance(value, tuple):
|
||||
return [_canonicalize(item) for item in value]
|
||||
if isinstance(value, list):
|
||||
return [_canonicalize(item) for item in value]
|
||||
if isinstance(value, enum.Enum):
|
||||
return value.name
|
||||
if isinstance(value, Path):
|
||||
return value.expanduser().as_posix()
|
||||
if isinstance(value, os.PathLike):
|
||||
return os.fspath(value)
|
||||
if isinstance(value, type):
|
||||
return f"{value.__module__}.{value.__qualname__}"
|
||||
if isinstance(value, torch.dtype):
|
||||
return str(value).removeprefix("torch.")
|
||||
return value
|
||||
|
||||
|
||||
def _none_if_empty(value: Any) -> Any | None:
|
||||
if value is None:
|
||||
return None
|
||||
if isinstance(value, str) and not value.strip():
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def _optional_digest(value: Any) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
return sha256_hexdigest(str(value))
|
||||
|
||||
|
||||
def _looks_like_local_path(value: str) -> bool:
|
||||
if value.startswith(("/", "./", "../", "~")):
|
||||
return True
|
||||
looks_like_hf_repo = re.match(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$", value)
|
||||
has_separator = "/" in value or "\\" in value or os.sep in value
|
||||
return has_separator and looks_like_hf_repo is None
|
||||
|
||||
|
||||
def _normalize_gpu_device(device: Mapping[str, Any]) -> dict[str, Any]:
|
||||
memory_gb = device.get("memory_gb")
|
||||
if memory_gb is None and device.get("memory_bytes") is not None:
|
||||
memory_gb = round(float(device["memory_bytes"]) / (1024**3))
|
||||
return {
|
||||
"name": _none_if_empty(device.get("name")) or "unknown",
|
||||
"memory_gb": memory_gb,
|
||||
"compute_capability": _none_if_empty(device.get("compute_capability")),
|
||||
}
|
||||
|
||||
|
||||
def _detect_cuda_interconnect(gpu_count: int) -> str:
|
||||
if gpu_count <= 1:
|
||||
return "single_gpu"
|
||||
try:
|
||||
from fastvideo.platforms import current_platform
|
||||
|
||||
if current_platform.is_cuda() and current_platform.is_full_nvlink(list(range(gpu_count))):
|
||||
return "full_nvlink"
|
||||
return "none_or_partial"
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _installed_package_versions() -> dict[str, str | None]:
|
||||
return {
|
||||
name: _distribution_version(*distribution_names)
|
||||
for name, distribution_names in _PACKAGE_DISTRIBUTIONS.items()
|
||||
}
|
||||
|
||||
|
||||
def _distribution_version(*names: str) -> str | None:
|
||||
for name in names:
|
||||
try:
|
||||
return importlib.metadata.version(name)
|
||||
except importlib.metadata.PackageNotFoundError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _major_minor(version: str | None) -> str | None:
|
||||
if not version:
|
||||
return None
|
||||
match = re.search(r"(\d+)\.(\d+)", version)
|
||||
if match is None:
|
||||
return version
|
||||
return f"{match.group(1)}.{match.group(2)}"
|
||||
|
||||
|
||||
__all__ = [
|
||||
"RECIPE_SCHEMA_VERSION",
|
||||
"REQUIRED_BENCHMARK_IDENTITY_KEYS",
|
||||
"benchmark_identity_from_config",
|
||||
"canonical_json",
|
||||
"environment_fingerprint",
|
||||
"environment_metadata",
|
||||
"hardware_profile",
|
||||
"hardware_profile_id",
|
||||
"normalize_model_path",
|
||||
"profile_id",
|
||||
"recipe_fingerprint",
|
||||
"resolved_revision_from_model_path",
|
||||
"build_recipe_from_benchmark_config",
|
||||
"sha256_hexdigest",
|
||||
"software_profile",
|
||||
"software_profile_id",
|
||||
]
|
||||
@@ -14,9 +14,9 @@ def _v2_config():
|
||||
return {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v-1.3b",
|
||||
"variant_id": "canonical",
|
||||
"benchmark_version": 1,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
}
|
||||
|
||||
|
||||
@@ -41,9 +41,9 @@ def test_v2_benchmark_config_identity_validates_and_is_preserved():
|
||||
assert _is_v2_config(cfg) is True
|
||||
assert _config_identity_metadata(cfg) == {
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v-1.3b",
|
||||
"variant_id": "canonical",
|
||||
"benchmark_version": 1,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"quality_metadata": {"some": "data"},
|
||||
}
|
||||
|
||||
@@ -91,9 +91,9 @@ def test_v2_benchmark_config_rejects_invalid_benchmark_version_values(value):
|
||||
def test_partial_v2_identity_requires_schema_version():
|
||||
cfg = {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"workload_id": "wan-t2v-1.3b",
|
||||
"variant_id": "canonical",
|
||||
"benchmark_version": 1,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
}
|
||||
|
||||
expected = "wan.json: v2 benchmark identity fields require config_schema_version=2"
|
||||
@@ -105,9 +105,9 @@ def test_optional_v2_metadata_fields_must_be_objects():
|
||||
cfg = {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v-1.3b",
|
||||
"variant_id": "canonical",
|
||||
"benchmark_version": 1,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"quality_metadata": ["not", "an", "object"],
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fastvideo.tests.performance import compare_baseline
|
||||
from fastvideo.performance.metric_policy import resolve_metric_policies
|
||||
|
||||
@@ -119,6 +123,196 @@ def test_boolean_regression_threshold_values_are_ignored():
|
||||
assert latency.threshold_percent == 0.08
|
||||
assert latency.threshold_absolute == 0.5
|
||||
assert latency.gated is False
|
||||
def test_normalized_record_preserves_identity_metadata(monkeypatch):
|
||||
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
|
||||
raw = _raw_result()
|
||||
raw.update({
|
||||
"result_schema_version": 2,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"recipe": {
|
||||
"recipe_schema_version": 1,
|
||||
},
|
||||
"recipe_fingerprint": "recipe-1",
|
||||
"hardware_profile": {
|
||||
"gpu_count": 1,
|
||||
},
|
||||
"hardware_profile_id": "hw-1",
|
||||
"software_profile": {
|
||||
"python": "3.12",
|
||||
},
|
||||
"software_profile_id": "sw-1",
|
||||
"environment_metadata": {
|
||||
"env": {
|
||||
"IMAGE_VERSION": "latest",
|
||||
},
|
||||
},
|
||||
"environment_fingerprint": "env-1",
|
||||
"quality_metadata": {
|
||||
"quality_status": "canonical",
|
||||
},
|
||||
})
|
||||
|
||||
record = compare_baseline.normalize_performance_result(raw)
|
||||
|
||||
assert record["result_schema_version"] == 2
|
||||
assert record["workload_id"] == "wan-t2v"
|
||||
assert record["variant_id"] == "1.3b-sp2"
|
||||
assert record["benchmark_version"] == 2
|
||||
assert record["recipe"] == {"recipe_schema_version": 1}
|
||||
assert record["recipe_fingerprint"] == "recipe-1"
|
||||
assert record["hardware_profile"] == {"gpu_count": 1}
|
||||
assert record["hardware_profile_id"] == "hw-1"
|
||||
assert record["software_profile"] == {"python": "3.12"}
|
||||
assert record["software_profile_id"] == "sw-1"
|
||||
assert record["environment_metadata"] == {"env": {"IMAGE_VERSION": "latest"}}
|
||||
assert record["environment_fingerprint"] == "env-1"
|
||||
assert record["quality_metadata"] == {"quality_status": "canonical"}
|
||||
|
||||
|
||||
def test_normalized_record_prefers_raw_v2_provenance(monkeypatch):
|
||||
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
|
||||
monkeypatch.setenv("BUILDKITE_BRANCH", "feature/from-env")
|
||||
monkeypatch.setenv("TEST_SCOPE", "direct")
|
||||
monkeypatch.setenv("BUILDKITE_BUILD_URL", "https://buildkite.example/env")
|
||||
monkeypatch.setenv("BUILDKITE_BUILD_ID", "env-build")
|
||||
monkeypatch.setenv("BUILDKITE_JOB_ID", "env-job")
|
||||
raw = _raw_result()
|
||||
raw.update({
|
||||
"result_schema_version": 2,
|
||||
"run_source": "scheduled_main",
|
||||
"branch": "main",
|
||||
"test_scope": "full",
|
||||
"build_url": "https://buildkite.example/raw",
|
||||
"build_id": "raw-build",
|
||||
"job_id": "raw-job",
|
||||
"pr_number": "false",
|
||||
})
|
||||
|
||||
record = compare_baseline.normalize_performance_result(raw)
|
||||
|
||||
assert record["result_schema_version"] == 2
|
||||
assert record["run_source"] == "scheduled_main"
|
||||
assert record["branch"] == "main"
|
||||
assert record["test_scope"] == "full"
|
||||
assert record["build_url"] == "https://buildkite.example/raw"
|
||||
assert record["build_id"] == "raw-build"
|
||||
assert record["job_id"] == "raw-job"
|
||||
assert record["pr_number"] == ""
|
||||
|
||||
|
||||
def test_v1_normalized_record_has_no_result_schema_version(monkeypatch):
|
||||
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
|
||||
|
||||
record = compare_baseline.normalize_performance_result(_raw_result())
|
||||
|
||||
assert "result_schema_version" not in record
|
||||
|
||||
|
||||
def test_main_writes_v1_current_artifact_without_comparison_identity(monkeypatch, tmp_path, capsys):
|
||||
results_dir = tmp_path / "results"
|
||||
reports_dir = tmp_path / "reports"
|
||||
tracking_root = tmp_path / "tracking"
|
||||
results_dir.mkdir()
|
||||
(results_dir / "perf_legacy.json").write_text(json.dumps(_raw_result()), encoding="utf-8")
|
||||
uploaded_records = []
|
||||
|
||||
monkeypatch.setenv("PERF_RUN_SOURCE", "scheduled_main")
|
||||
monkeypatch.delenv("PERF_PYTEST_RC", raising=False)
|
||||
monkeypatch.setattr(compare_baseline, "RESULTS_DIR", str(results_dir))
|
||||
monkeypatch.setattr(compare_baseline, "PERF_REPORTS_DIR", str(reports_dir))
|
||||
monkeypatch.setattr(compare_baseline, "TRACKING_ROOT", str(tracking_root))
|
||||
monkeypatch.setattr(compare_baseline, "UPLOAD_POLICY", "always")
|
||||
monkeypatch.setattr(compare_baseline, "sync_from_hf", lambda local_dir, strict=False: local_dir)
|
||||
|
||||
def fail_load_records_for_model(*_args, **_kwargs):
|
||||
raise AssertionError("legacy current records should skip v2 baseline lookup")
|
||||
|
||||
def fake_upload_record(_path, record, *, strict=False):
|
||||
uploaded_records.append(record.copy())
|
||||
|
||||
monkeypatch.setattr(compare_baseline, "load_records_for_model", fail_load_records_for_model)
|
||||
monkeypatch.setattr(compare_baseline, "upload_record", fake_upload_record)
|
||||
|
||||
assert compare_baseline.main() == 0
|
||||
|
||||
output = capsys.readouterr().out
|
||||
normalized_files = list((reports_dir / "results").glob("normalized_perf_*.json"))
|
||||
assert "Skipping rolling baseline comparison" in output
|
||||
assert "workload_id" in output
|
||||
assert len(normalized_files) == 1
|
||||
|
||||
normalized = json.loads(normalized_files[0].read_text(encoding="utf-8"))
|
||||
assert "result_schema_version" not in normalized
|
||||
assert normalized["success"] is True
|
||||
assert normalized["baseline_eligible"] is False
|
||||
assert len(uploaded_records) == 1
|
||||
assert uploaded_records[0]["baseline_eligible"] is False
|
||||
|
||||
|
||||
def test_normalized_record_reads_identity_labels_from_recipe(monkeypatch):
|
||||
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
|
||||
raw = _raw_result()
|
||||
raw["recipe"] = {
|
||||
"benchmark": {
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
},
|
||||
}
|
||||
|
||||
record = compare_baseline.normalize_performance_result(raw)
|
||||
|
||||
assert record["workload_id"] == "wan-t2v"
|
||||
assert record["variant_id"] == "1.3b-sp2"
|
||||
assert record["benchmark_version"] == 2
|
||||
|
||||
|
||||
def test_comparison_identity_filters_use_full_issue_key():
|
||||
record = {
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"recipe_fingerprint": "recipe-1",
|
||||
"hardware_profile_id": "hw-1",
|
||||
"software_profile_id": "sw-1",
|
||||
}
|
||||
|
||||
assert compare_baseline._comparison_identity_filters(record) == {
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": "2",
|
||||
"recipe_fingerprint": "recipe-1",
|
||||
"hardware_profile_id": "hw-1",
|
||||
"software_profile_id": "sw-1",
|
||||
}
|
||||
|
||||
|
||||
def test_comparison_identity_filters_require_full_issue_key():
|
||||
record = {
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 0,
|
||||
"recipe_fingerprint": "recipe-1",
|
||||
"hardware_profile_id": "hw-1",
|
||||
}
|
||||
|
||||
with pytest.raises(ValueError, match="software_profile_id"):
|
||||
compare_baseline._comparison_identity_filters(record)
|
||||
|
||||
|
||||
def test_comparison_identity_filters_keep_zero_version():
|
||||
record = {
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 0,
|
||||
"recipe_fingerprint": "recipe-1",
|
||||
"hardware_profile_id": "hw-1",
|
||||
"software_profile_id": "sw-1",
|
||||
}
|
||||
|
||||
assert compare_baseline._comparison_identity_filters(record)["benchmark_version"] == "0"
|
||||
|
||||
|
||||
def test_baseline_eligibility_only_for_successful_scheduled_main():
|
||||
|
||||
@@ -56,8 +56,34 @@ def _record(model_id, gpu_type, ts, commit, latency, throughput, success=True, *
|
||||
|
||||
def test_summary_endpoint_returns_latest_group_status():
|
||||
app = create_app(FakeStore([
|
||||
_record("wan", "NVIDIA L40S", "2026-01-01T00:00:00+00:00", "a" * 40, 10.0, 10.0),
|
||||
_record("wan", "NVIDIA L40S", "2026-01-02T00:00:00+00:00", "b" * 40, 11.0, 9.0),
|
||||
_record(
|
||||
"wan",
|
||||
"NVIDIA L40S",
|
||||
"2026-01-01T00:00:00+00:00",
|
||||
"a" * 40,
|
||||
10.0,
|
||||
10.0,
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=2,
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
),
|
||||
_record(
|
||||
"wan",
|
||||
"NVIDIA L40S",
|
||||
"2026-01-02T00:00:00+00:00",
|
||||
"b" * 40,
|
||||
11.0,
|
||||
9.0,
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=2,
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
),
|
||||
]))
|
||||
client = TestClient(app)
|
||||
|
||||
@@ -71,6 +97,10 @@ def test_summary_endpoint_returns_latest_group_status():
|
||||
assert body["rows"][0]["metrics"]["latency"]["threshold_exceeded"] is True
|
||||
assert body["rows"][0]["threshold_exceeded_metrics"] == ["latency", "throughput"]
|
||||
assert body["rows"][0]["computed_regression_status"] == "fail"
|
||||
assert body["rows"][0]["workload_id"] == "wan-t2v"
|
||||
assert body["rows"][0]["variant_id"] == "1.3b-sp2"
|
||||
assert body["rows"][0]["benchmark_version"] == 2
|
||||
assert body["rows"][0]["recipe_fingerprint"] == "recipe-a"
|
||||
|
||||
|
||||
def test_summary_status_is_independent_of_days_window():
|
||||
@@ -124,8 +154,8 @@ def test_dashboard_endpoints_filter_and_return_run_source_metadata():
|
||||
assert summary["count"] == 1
|
||||
assert summary["rows"][0]["run_source"] == "pr"
|
||||
assert summary["rows"][0]["pr_number"] == "123"
|
||||
assert summary["rows"][0]["baseline_n"] == 1
|
||||
assert summary["rows"][0]["metrics"]["latency"]["baseline"] == 11.0
|
||||
assert summary["rows"][0]["baseline_n"] == 0
|
||||
assert summary["rows"][0]["metrics"]["latency"]["baseline"] is None
|
||||
assert summary["filters"]["run_source"] == "pr"
|
||||
assert trends["count"] == 1
|
||||
assert trends["groups"][0]["points"][0]["run_source"] == "pr"
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from fastvideo.tests.performance import dashboard
|
||||
|
||||
|
||||
def _record(recipe_fingerprint="recipe-a", software_profile_id="sw-a", **overrides):
|
||||
record = {
|
||||
"model_id": "wan-t2v-1.3b-2gpu",
|
||||
"gpu_type": "NVIDIA L40S",
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"recipe_fingerprint": recipe_fingerprint,
|
||||
"hardware_profile_id": "hw-l40s",
|
||||
"software_profile_id": software_profile_id,
|
||||
"timestamp": "2026-01-01T00:00:00+00:00",
|
||||
"commit_sha": "a" * 40,
|
||||
"config_id": "aaaaaaa",
|
||||
"latency": 10.0,
|
||||
"throughput": 4.5,
|
||||
"memory": 10000.0,
|
||||
"text_encoder_time_s": 2.0,
|
||||
"dit_time_s": 8.0,
|
||||
"vae_decode_time_s": 3.0,
|
||||
}
|
||||
record.update(overrides)
|
||||
return record
|
||||
|
||||
|
||||
def test_group_data_uses_full_comparison_cohort():
|
||||
df = pd.DataFrame([
|
||||
_record(recipe_fingerprint="recipe-a", software_profile_id="sw-a"),
|
||||
_record(recipe_fingerprint="recipe-b", software_profile_id="sw-b"),
|
||||
])
|
||||
|
||||
groups = list(dashboard.group_data(df))
|
||||
|
||||
assert len(groups) == 2
|
||||
assert {key[5] for key, _group in groups} == {"recipe-a", "recipe-b"}
|
||||
assert {key[7] for key, _group in groups} == {"sw-a", "sw-b"}
|
||||
|
||||
|
||||
def test_group_data_fills_legacy_cohort_columns():
|
||||
df = pd.DataFrame([{
|
||||
"model_id": "legacy-wan",
|
||||
"gpu_type": "NVIDIA L40S",
|
||||
"timestamp": "2026-01-01T00:00:00+00:00",
|
||||
"commit_sha": "a" * 40,
|
||||
"config_id": "aaaaaaa",
|
||||
"latency": 10.0,
|
||||
}])
|
||||
|
||||
groups = list(dashboard.group_data(df))
|
||||
|
||||
assert len(groups) == 1
|
||||
assert groups[0][0][2:] == ("", "", "", "", "", "")
|
||||
|
||||
|
||||
def test_build_plots_labels_distinct_cohorts():
|
||||
df = pd.DataFrame([
|
||||
_record(recipe_fingerprint="recipe-a", software_profile_id="sw-a"),
|
||||
_record(recipe_fingerprint="recipe-b", software_profile_id="sw-b"),
|
||||
])
|
||||
|
||||
figs, skipped_metrics = dashboard.build_plots(df)
|
||||
titles = [fig.layout.title.text for fig in figs]
|
||||
|
||||
assert len(figs) == len(dashboard.METRICS) * 2
|
||||
assert skipped_metrics == []
|
||||
assert any("recipe recipe-a | hw-l40s | sw-a" in title for title in titles)
|
||||
assert any("recipe recipe-b | hw-l40s | sw-b" in title for title in titles)
|
||||
|
||||
|
||||
def test_skipped_metric_table_includes_cohort_identity():
|
||||
df = pd.DataFrame([{
|
||||
"model_id": "wan-t2v-1.3b-2gpu",
|
||||
"gpu_type": "NVIDIA L40S",
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"recipe_fingerprint": "recipe-a",
|
||||
"hardware_profile_id": "hw-l40s",
|
||||
"software_profile_id": "sw-a",
|
||||
"timestamp": "2026-01-01T00:00:00+00:00",
|
||||
"commit_sha": "a" * 40,
|
||||
"config_id": "aaaaaaa",
|
||||
"latency": 10.0,
|
||||
}])
|
||||
|
||||
_figs, skipped_metrics = dashboard.build_plots(df)
|
||||
html = dashboard.render_skipped_metrics(skipped_metrics[:1])
|
||||
|
||||
assert skipped_metrics[0]["cohort"] == "wan-t2v / 1.3b-sp2 / v2"
|
||||
assert skipped_metrics[0]["cohort_detail"] == "recipe recipe-a | hw-l40s | sw-a"
|
||||
assert "wan-t2v / 1.3b-sp2 / v2" in html
|
||||
assert "recipe recipe-a | hw-l40s | sw-a" in html
|
||||
@@ -144,6 +144,204 @@ def test_build_latest_summary_separates_informational_threshold_crossing():
|
||||
assert rows[0]["computed_regression_status"] == "pass"
|
||||
|
||||
|
||||
def test_build_latest_summary_run_source_filter_excludes_future_baselines():
|
||||
records = [
|
||||
_record(
|
||||
"2026-01-01T00:00:00+00:00",
|
||||
"a" * 40,
|
||||
10.0,
|
||||
10.0,
|
||||
run_source="scheduled_main",
|
||||
baseline_eligible=True,
|
||||
),
|
||||
_record(
|
||||
"2026-01-02T00:00:00+00:00",
|
||||
"b" * 40,
|
||||
11.0,
|
||||
9.0,
|
||||
run_source="pr",
|
||||
baseline_eligible=False,
|
||||
pr_number="123",
|
||||
),
|
||||
_record(
|
||||
"2026-01-03T00:00:00+00:00",
|
||||
"c" * 40,
|
||||
30.0,
|
||||
3.0,
|
||||
run_source="scheduled_main",
|
||||
baseline_eligible=True,
|
||||
),
|
||||
]
|
||||
|
||||
rows = build_latest_summary(records, run_source="pr")
|
||||
|
||||
assert len(rows) == 1
|
||||
assert rows[0]["run_source"] == "pr"
|
||||
assert rows[0]["baseline_n"] == 1
|
||||
assert rows[0]["metrics"]["latency"]["baseline"] == 10.0
|
||||
assert rows[0]["metrics"]["throughput"]["baseline"] == 10.0
|
||||
|
||||
|
||||
|
||||
def test_build_latest_summary_keeps_identity_cohorts_separate():
|
||||
records = [
|
||||
_record(
|
||||
"2026-01-01T00:00:00+00:00",
|
||||
"a" * 40,
|
||||
10.0,
|
||||
10.0,
|
||||
recipe_fingerprint="recipe-a",
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=2,
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
run_source="scheduled_main",
|
||||
baseline_eligible=True,
|
||||
),
|
||||
_record(
|
||||
"2026-01-02T00:00:00+00:00",
|
||||
"b" * 40,
|
||||
20.0,
|
||||
5.0,
|
||||
recipe_fingerprint="recipe-b",
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=2,
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
run_source="scheduled_main",
|
||||
baseline_eligible=True,
|
||||
),
|
||||
_record(
|
||||
"2026-01-03T00:00:00+00:00",
|
||||
"c" * 40,
|
||||
22.0,
|
||||
4.5,
|
||||
recipe_fingerprint="recipe-b",
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=2,
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
),
|
||||
]
|
||||
|
||||
rows = build_latest_summary(records)
|
||||
recipe_b_row = next(row for row in rows if row["recipe_fingerprint"] == "recipe-b")
|
||||
|
||||
assert len(rows) == 2
|
||||
assert recipe_b_row["baseline_n"] == 1
|
||||
assert recipe_b_row["metrics"]["latency"]["baseline"] == 20.0
|
||||
assert recipe_b_row["workload_id"] == "wan-t2v"
|
||||
assert recipe_b_row["variant_id"] == "1.3b-sp2"
|
||||
assert recipe_b_row["benchmark_version"] == 2
|
||||
|
||||
|
||||
def test_build_latest_summary_keeps_variant_versions_separate():
|
||||
records = [
|
||||
_record(
|
||||
"2026-01-01T00:00:00+00:00",
|
||||
"a" * 40,
|
||||
10.0,
|
||||
10.0,
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=1,
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
run_source="scheduled_main",
|
||||
baseline_eligible=True,
|
||||
),
|
||||
_record(
|
||||
"2026-01-02T00:00:00+00:00",
|
||||
"b" * 40,
|
||||
20.0,
|
||||
5.0,
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=2,
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
run_source="scheduled_main",
|
||||
baseline_eligible=True,
|
||||
),
|
||||
_record(
|
||||
"2026-01-03T00:00:00+00:00",
|
||||
"c" * 40,
|
||||
22.0,
|
||||
4.5,
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=2,
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
),
|
||||
]
|
||||
|
||||
rows = build_latest_summary(records)
|
||||
version_2_row = next(row for row in rows if row["benchmark_version"] == 2)
|
||||
|
||||
assert len(rows) == 2
|
||||
assert version_2_row["baseline_n"] == 1
|
||||
assert version_2_row["metrics"]["latency"]["baseline"] == 20.0
|
||||
|
||||
|
||||
def test_dashboard_identity_preserves_zero_benchmark_version():
|
||||
records = [
|
||||
_record(
|
||||
"2026-01-01T00:00:00+00:00",
|
||||
"a" * 40,
|
||||
10.0,
|
||||
10.0,
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=0,
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
run_source="scheduled_main",
|
||||
baseline_eligible=True,
|
||||
),
|
||||
_record(
|
||||
"2026-01-02T00:00:00+00:00",
|
||||
"b" * 40,
|
||||
11.0,
|
||||
9.0,
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version=0,
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
),
|
||||
_record(
|
||||
"2026-01-03T00:00:00+00:00",
|
||||
"c" * 40,
|
||||
20.0,
|
||||
5.0,
|
||||
),
|
||||
]
|
||||
|
||||
rows = build_latest_summary(records)
|
||||
version_zero_row = next(row for row in rows if row["benchmark_version"] == 0)
|
||||
legacy_row = next(row for row in rows if row["benchmark_version"] == "")
|
||||
trends = build_trends(records)
|
||||
version_zero_trend = next(trend for trend in trends if trend["benchmark_version"] == 0)
|
||||
legacy_trend = next(trend for trend in trends if trend["benchmark_version"] == "")
|
||||
|
||||
assert len(rows) == 2
|
||||
assert version_zero_row["baseline_n"] == 1
|
||||
assert version_zero_row["metrics"]["latency"]["baseline"] == 10.0
|
||||
assert legacy_row["baseline_n"] == 0
|
||||
assert len(trends) == 2
|
||||
assert version_zero_trend["points"][0]["benchmark_version"] == 0
|
||||
assert legacy_trend["points"][0]["benchmark_version"] == ""
|
||||
|
||||
|
||||
def test_filter_records_and_trends_preserve_metric_points():
|
||||
records = [
|
||||
_record("2026-01-01T00:00:00+00:00", "a" * 40, 10.0, 10.0),
|
||||
@@ -218,3 +416,78 @@ def test_load_records_can_filter_baseline_eligible_records(tmp_path):
|
||||
"2026-01-02T00:00:00+00:00",
|
||||
"2026-01-03T00:00:00+00:00",
|
||||
}
|
||||
|
||||
|
||||
def test_load_records_for_model_filters_identity_cohort(tmp_path):
|
||||
model_dir = tmp_path / "wan"
|
||||
model_dir.mkdir()
|
||||
(model_dir / "matching.json").write_text(
|
||||
"""
|
||||
{
|
||||
"model_id": "wan",
|
||||
"gpu_type": "NVIDIA L40S",
|
||||
"timestamp": "2026-01-01T00:00:00+00:00",
|
||||
"success": true,
|
||||
"baseline_eligible": true,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"recipe_fingerprint": "recipe-a",
|
||||
"hardware_profile_id": "hw-l40s",
|
||||
"software_profile_id": "sw-cu130"
|
||||
}
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(model_dir / "other_recipe.json").write_text(
|
||||
"""
|
||||
{
|
||||
"model_id": "wan",
|
||||
"gpu_type": "NVIDIA L40S",
|
||||
"timestamp": "2026-01-02T00:00:00+00:00",
|
||||
"success": true,
|
||||
"baseline_eligible": true,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"recipe_fingerprint": "recipe-b",
|
||||
"hardware_profile_id": "hw-l40s",
|
||||
"software_profile_id": "sw-cu130"
|
||||
}
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(model_dir / "other_version.json").write_text(
|
||||
"""
|
||||
{
|
||||
"model_id": "wan",
|
||||
"gpu_type": "NVIDIA L40S",
|
||||
"timestamp": "2026-01-03T00:00:00+00:00",
|
||||
"success": true,
|
||||
"baseline_eligible": true,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 3,
|
||||
"recipe_fingerprint": "recipe-a",
|
||||
"hardware_profile_id": "hw-l40s",
|
||||
"software_profile_id": "sw-cu130"
|
||||
}
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
records = hf_store.load_records_for_model(
|
||||
str(tmp_path),
|
||||
"wan",
|
||||
"NVIDIA L40S",
|
||||
workload_id="wan-t2v",
|
||||
variant_id="1.3b-sp2",
|
||||
benchmark_version="2",
|
||||
recipe_fingerprint="recipe-a",
|
||||
hardware_profile_id="hw-l40s",
|
||||
software_profile_id="sw-cu130",
|
||||
baseline_eligible_only=True,
|
||||
)
|
||||
|
||||
assert len(records) == 1
|
||||
assert records[0]["recipe_fingerprint"] == "recipe-a"
|
||||
|
||||
@@ -0,0 +1,422 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
from copy import deepcopy
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
from fastvideo.tests.performance import identity as identity_module
|
||||
from fastvideo.tests.performance.identity import (
|
||||
build_recipe_from_benchmark_config,
|
||||
canonical_json,
|
||||
environment_fingerprint,
|
||||
environment_metadata,
|
||||
hardware_profile,
|
||||
hardware_profile_id,
|
||||
recipe_fingerprint,
|
||||
resolved_revision_from_model_path,
|
||||
software_profile,
|
||||
software_profile_id,
|
||||
)
|
||||
|
||||
|
||||
def _benchmark_config():
|
||||
return {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "Wan2.1-T2V-1.3B",
|
||||
},
|
||||
"init_kwargs": {
|
||||
"num_gpus": 2,
|
||||
"sp_size": 2,
|
||||
"tp_size": 1,
|
||||
"vae_sp": True,
|
||||
"vae_tiling": True,
|
||||
"text_encoder_precisions": ["fp32"],
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 45,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 3,
|
||||
"embedded_cfg_scale": 6,
|
||||
"seed": 1024,
|
||||
"fps": 24,
|
||||
"neg_prompt": "low quality",
|
||||
},
|
||||
"test_prompts": ["A cinematic video."],
|
||||
}
|
||||
|
||||
|
||||
def _fingerprint(cfg, attention_backend="FLASH_ATTN"):
|
||||
recipe = build_recipe_from_benchmark_config(cfg, attention_backend=attention_backend)
|
||||
return recipe_fingerprint(recipe)
|
||||
|
||||
|
||||
def test_canonical_json_is_stable_for_mapping_order():
|
||||
left = {"b": [2, 1], "a": {"z": "last", "m": "middle"}}
|
||||
right = {"a": {"m": "middle", "z": "last"}, "b": [2, 1]}
|
||||
|
||||
assert canonical_json(left) == canonical_json(right)
|
||||
|
||||
|
||||
def test_canonical_json_handles_sets_deterministically():
|
||||
assert canonical_json({"values": {"b", "a"}}) == '{"values":["a","b"]}'
|
||||
assert canonical_json({"values": frozenset(("b", "a"))}) == '{"values":["a","b"]}'
|
||||
|
||||
|
||||
def test_same_config_produces_same_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
|
||||
assert _fingerprint(cfg) == _fingerprint(deepcopy(cfg))
|
||||
|
||||
|
||||
def test_recipe_includes_first_class_benchmark_identity():
|
||||
recipe = build_recipe_from_benchmark_config(_benchmark_config())
|
||||
|
||||
assert recipe["benchmark"] == {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
}
|
||||
|
||||
|
||||
def test_recipe_requires_first_class_benchmark_identity():
|
||||
cfg = _benchmark_config()
|
||||
del cfg["workload_id"]
|
||||
|
||||
with pytest.raises(ValueError, match="workload_id"):
|
||||
build_recipe_from_benchmark_config(cfg)
|
||||
|
||||
|
||||
def test_semantically_equivalent_sequence_forms_match():
|
||||
cfg = _benchmark_config()
|
||||
equivalent = deepcopy(cfg)
|
||||
equivalent["init_kwargs"]["text_encoder_precisions"] = tuple(
|
||||
equivalent["init_kwargs"]["text_encoder_precisions"])
|
||||
|
||||
assert _fingerprint(cfg) == _fingerprint(equivalent)
|
||||
|
||||
|
||||
def test_output_path_does_not_change_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
with_output_path = deepcopy(cfg)
|
||||
with_output_path["generation_kwargs"]["output_path"] = "/tmp/generated"
|
||||
|
||||
assert _fingerprint(cfg) == _fingerprint(with_output_path)
|
||||
|
||||
|
||||
def test_runtime_resolved_values_do_not_change_recipe_fingerprint():
|
||||
# Runtime-resolved values are audit metadata: an upstream repo commit
|
||||
# (even a README-only edit) or a silent attention-backend fallback must
|
||||
# not churn the cohort — that would open a fresh ungated cohort and let
|
||||
# degraded numbers seed a new baseline. Declared inputs (including
|
||||
# requested_backend) stay in the hash.
|
||||
cfg = _benchmark_config()
|
||||
recipe = build_recipe_from_benchmark_config(
|
||||
cfg, attention_backend="FLASH_ATTN",
|
||||
resolved_attention_backend="FLASH_ATTN",
|
||||
resolved_model_revision="aaaa1111")
|
||||
changed = build_recipe_from_benchmark_config(
|
||||
cfg, attention_backend="FLASH_ATTN",
|
||||
resolved_attention_backend="TORCH_SDPA",
|
||||
resolved_model_revision="bbbb2222")
|
||||
|
||||
assert recipe["model"]["resolved_revision"] == "aaaa1111" # kept for audit
|
||||
assert recipe["attention"]["resolved_backend"] == "FLASH_ATTN" # kept for audit
|
||||
assert recipe_fingerprint(recipe) == recipe_fingerprint(changed)
|
||||
|
||||
|
||||
def test_measured_prompt_override_ignores_unused_configured_prompts():
|
||||
cfg = _benchmark_config()
|
||||
changed = deepcopy(cfg)
|
||||
changed["test_prompts"] = [
|
||||
"Unused prompt that should not affect this measured workload.",
|
||||
cfg["test_prompts"][0],
|
||||
]
|
||||
|
||||
recipe = build_recipe_from_benchmark_config(cfg, measured_prompts=[cfg["test_prompts"][0]])
|
||||
changed_recipe = build_recipe_from_benchmark_config(changed, measured_prompts=[cfg["test_prompts"][0]])
|
||||
|
||||
assert recipe_fingerprint(recipe) == recipe_fingerprint(changed_recipe)
|
||||
|
||||
|
||||
def test_num_inference_steps_changes_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
changed = deepcopy(cfg)
|
||||
changed["generation_kwargs"]["num_inference_steps"] = 8
|
||||
|
||||
assert _fingerprint(cfg) != _fingerprint(changed)
|
||||
|
||||
|
||||
def test_benchmark_version_changes_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
changed = deepcopy(cfg)
|
||||
changed["benchmark_version"] = 3
|
||||
|
||||
assert _fingerprint(cfg) != _fingerprint(changed)
|
||||
|
||||
|
||||
def test_attention_backend_changes_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
|
||||
assert _fingerprint(cfg, attention_backend="FLASH_ATTN") != _fingerprint(
|
||||
cfg, attention_backend="TORCH_SDPA")
|
||||
|
||||
|
||||
def test_precision_changes_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
changed = deepcopy(cfg)
|
||||
changed["init_kwargs"]["text_encoder_precisions"] = ["bf16"]
|
||||
|
||||
assert _fingerprint(cfg) != _fingerprint(changed)
|
||||
|
||||
|
||||
def test_dimensions_and_seed_change_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
different_dimensions = deepcopy(cfg)
|
||||
different_dimensions["generation_kwargs"]["width"] = 1024
|
||||
different_seed = deepcopy(cfg)
|
||||
different_seed["generation_kwargs"]["seed"] = 2048
|
||||
|
||||
assert _fingerprint(cfg) != _fingerprint(different_dimensions)
|
||||
assert _fingerprint(cfg) != _fingerprint(different_seed)
|
||||
|
||||
|
||||
def test_distributed_layout_changes_recipe_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
changed = deepcopy(cfg)
|
||||
changed["init_kwargs"]["sp_size"] = 1
|
||||
changed["init_kwargs"]["tp_size"] = 2
|
||||
|
||||
assert _fingerprint(cfg) != _fingerprint(changed)
|
||||
|
||||
|
||||
def test_software_profile_uses_exact_attention_kernel_package_versions_for_id():
|
||||
profile_a = software_profile(
|
||||
python_version="3.12.4",
|
||||
torch_version="2.12.0+cu130",
|
||||
cuda_version="13.0",
|
||||
package_versions={
|
||||
"triton": "3.4.1",
|
||||
"fastvideo_kernel": "0.3.2",
|
||||
"flash_attn_4": "4.0.0.dev0",
|
||||
"flashinfer": "0.2.11",
|
||||
"nvidia_cutlass_dsl": "4.5.0",
|
||||
},
|
||||
)
|
||||
profile_b = software_profile(
|
||||
python_version="3.12.5",
|
||||
torch_version="2.12.1+cu130",
|
||||
cuda_version="13.0",
|
||||
package_versions={
|
||||
"triton": "3.4.9",
|
||||
"fastvideo_kernel": "0.3.7",
|
||||
"flash_attn_4": "4.0.0.dev1",
|
||||
"flashinfer": "0.2.12",
|
||||
"nvidia_cutlass_dsl": "4.5.1",
|
||||
},
|
||||
)
|
||||
|
||||
assert profile_a == {
|
||||
"python": "3.12",
|
||||
"pytorch": "2.12",
|
||||
"cuda": "13.0",
|
||||
"packages": {
|
||||
"fastvideo_kernel": "0.3.2",
|
||||
"flash_attn_4": "4.0.0.dev0",
|
||||
"flashinfer": "0.2.11",
|
||||
"nvidia_cutlass_dsl": "4.5.0",
|
||||
"triton": "3.4.1",
|
||||
},
|
||||
}
|
||||
assert profile_b["python"] == "3.12"
|
||||
assert profile_b["pytorch"] == "2.12"
|
||||
assert profile_b["cuda"] == "13.0"
|
||||
assert software_profile_id(profile_a) != software_profile_id(profile_b)
|
||||
|
||||
|
||||
def test_installed_package_versions_checks_attention_kernel_distributions(monkeypatch):
|
||||
versions = {
|
||||
"fastvideo-kernel": "0.3.2",
|
||||
"flash-attn": "2.8.1",
|
||||
"flash-attn-4": "4.0.0.dev0",
|
||||
"flash-attention-fp4": "0.1.0",
|
||||
"flashinfer-python": "0.2.11",
|
||||
"nvidia-cutlass-dsl": "4.5.0",
|
||||
}
|
||||
|
||||
def fake_version(name):
|
||||
if name not in versions:
|
||||
raise identity_module.importlib.metadata.PackageNotFoundError(name)
|
||||
return versions[name]
|
||||
|
||||
monkeypatch.setattr(identity_module.importlib.metadata, "version", fake_version)
|
||||
|
||||
installed = identity_module._installed_package_versions()
|
||||
|
||||
assert installed["fastvideo_kernel"] == "0.3.2"
|
||||
assert installed["flash_attn"] == "2.8.1"
|
||||
assert installed["flash_attn_4"] == "4.0.0.dev0"
|
||||
assert installed["flash_attention_fp4"] == "0.1.0"
|
||||
assert installed["flashinfer"] == "0.2.11"
|
||||
assert installed["nvidia_cutlass_dsl"] == "4.5.0"
|
||||
|
||||
|
||||
def test_hardware_profile_id_uses_normalized_gpu_cohort():
|
||||
profile = hardware_profile(
|
||||
num_gpus=2,
|
||||
gpu_devices=[
|
||||
{
|
||||
"name": "NVIDIA L40S",
|
||||
"memory_bytes": 48 * 1024**3,
|
||||
"compute_capability": "8.9",
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA L40S",
|
||||
"memory_gb": 48,
|
||||
"compute_capability": "8.9",
|
||||
},
|
||||
],
|
||||
interconnect="none_or_partial",
|
||||
)
|
||||
|
||||
assert profile == {
|
||||
"device_type": "cuda",
|
||||
"gpu_count": 2,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA L40S",
|
||||
"memory_gb": 48,
|
||||
"compute_capability": "8.9",
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA L40S",
|
||||
"memory_gb": 48,
|
||||
"compute_capability": "8.9",
|
||||
},
|
||||
],
|
||||
"interconnect": "none_or_partial",
|
||||
}
|
||||
assert hardware_profile_id(profile).startswith("hw-")
|
||||
|
||||
|
||||
def test_hardware_profile_pads_missing_requested_cuda_devices(monkeypatch):
|
||||
class Props:
|
||||
name = "NVIDIA L40S"
|
||||
total_memory = 48 * 1024**3
|
||||
|
||||
monkeypatch.setattr(torch.cuda, "is_available", lambda: True)
|
||||
monkeypatch.setattr(torch.cuda, "device_count", lambda: 1)
|
||||
monkeypatch.setattr(torch.cuda, "get_device_properties", lambda device_id: Props())
|
||||
monkeypatch.setattr(torch.cuda, "get_device_capability", lambda device_id: (8, 9))
|
||||
|
||||
profile = hardware_profile(num_gpus=2, interconnect="unknown")
|
||||
|
||||
assert profile["gpu_count"] == 2
|
||||
assert profile["gpus"] == [
|
||||
{
|
||||
"name": "NVIDIA L40S",
|
||||
"memory_gb": 48,
|
||||
"compute_capability": "8.9",
|
||||
},
|
||||
{
|
||||
"name": "unknown",
|
||||
"memory_gb": None,
|
||||
"compute_capability": None,
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def test_local_path_detection_checks_both_path_separators(monkeypatch):
|
||||
monkeypatch.setattr(identity_module.os, "sep", "\\")
|
||||
|
||||
assert identity_module._looks_like_local_path("models/local/checkpoint") is True
|
||||
assert identity_module._looks_like_local_path(r"C:\models\checkpoint") is True
|
||||
assert identity_module._looks_like_local_path("Wan-AI/Wan2.1-T2V-1.3B-Diffusers") is False
|
||||
|
||||
|
||||
def test_environment_metadata_is_separate_audit_fingerprint():
|
||||
cfg = _benchmark_config()
|
||||
recipe_hash = _fingerprint(cfg)
|
||||
audit_a = environment_metadata(
|
||||
env={"IMAGE_VERSION": "py3.12-cuda13.0.0"},
|
||||
package_versions={"triton": "3.4.1"},
|
||||
hardware={"device_type": "cuda", "gpu_count": 2},
|
||||
software={"python": "3.12", "pytorch": "2.12", "cuda": "13.0"},
|
||||
)
|
||||
audit_b = environment_metadata(
|
||||
env={"IMAGE_VERSION": "py3.12-cuda13.0.1"},
|
||||
package_versions={"triton": "3.4.1"},
|
||||
hardware={"device_type": "cuda", "gpu_count": 2},
|
||||
software={"python": "3.12", "pytorch": "2.12", "cuda": "13.0"},
|
||||
)
|
||||
|
||||
assert recipe_hash == _fingerprint(cfg)
|
||||
assert environment_fingerprint(audit_a) != environment_fingerprint(audit_b)
|
||||
|
||||
|
||||
def test_resolved_revision_from_hf_snapshot_path():
|
||||
revision = "a" * 40
|
||||
|
||||
assert resolved_revision_from_model_path(
|
||||
f"/root/.cache/huggingface/hub/models--org--repo/snapshots/{revision}") == revision
|
||||
assert resolved_revision_from_model_path("/models/local-repo") is None
|
||||
|
||||
|
||||
def test_runtime_identity_from_generator_summarizes_worker_records():
|
||||
from fastvideo.tests.performance.test_inference_performance import _runtime_identity_from_generator
|
||||
|
||||
revision = "b" * 40
|
||||
|
||||
class FakeExecutor:
|
||||
def collective_rpc(self, _method):
|
||||
return [
|
||||
{
|
||||
"resolved_attention_backends": ["FLASH_ATTN"],
|
||||
"resolved_model_path": f"/root/.cache/huggingface/hub/models--org--repo/snapshots/{revision}",
|
||||
},
|
||||
{
|
||||
"resolved_attention_backends": ["FLASH_ATTN"],
|
||||
"resolved_model_path": f"/root/.cache/huggingface/hub/models--org--repo/snapshots/{revision}",
|
||||
},
|
||||
]
|
||||
|
||||
class FakeGenerator:
|
||||
executor = FakeExecutor()
|
||||
|
||||
assert _runtime_identity_from_generator(FakeGenerator()) == {
|
||||
"resolved_attention_backend": "FLASH_ATTN",
|
||||
"resolved_model_revision": revision,
|
||||
}
|
||||
|
||||
|
||||
def test_collect_worker_identity_handles_torch_module_pipeline():
|
||||
from fastvideo.tests.performance.test_inference_performance import _collect_worker_identity
|
||||
|
||||
class Backend:
|
||||
name = "FLASH_ATTN"
|
||||
|
||||
class Leaf(torch.nn.Module):
|
||||
backend = Backend()
|
||||
|
||||
class Pipeline(torch.nn.Module):
|
||||
model_path = "/models/local"
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.leaf = Leaf()
|
||||
|
||||
class Worker:
|
||||
pipeline = Pipeline()
|
||||
|
||||
assert _collect_worker_identity(Worker()) == {
|
||||
"resolved_attention_backends": ["FLASH_ATTN"],
|
||||
"resolved_model_path": "/models/local",
|
||||
}
|
||||
@@ -12,12 +12,25 @@ import os
|
||||
import time
|
||||
from collections.abc import Mapping
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
import pytest
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.logger import init_logger
|
||||
from fastvideo.tests.performance.identity import (
|
||||
benchmark_identity_from_config,
|
||||
build_recipe_from_benchmark_config,
|
||||
environment_fingerprint,
|
||||
environment_metadata,
|
||||
hardware_profile,
|
||||
hardware_profile_id,
|
||||
recipe_fingerprint,
|
||||
resolved_revision_from_model_path,
|
||||
software_profile,
|
||||
software_profile_id,
|
||||
)
|
||||
from fastvideo.worker.multiproc_executor import MultiprocExecutor
|
||||
|
||||
logger = init_logger(__name__)
|
||||
@@ -34,11 +47,17 @@ V2_REQUIRED_IDENTITY_FIELDS = (
|
||||
"variant_id",
|
||||
"benchmark_version",
|
||||
)
|
||||
# "recipe" is no longer config-declarable: the fingerprint follow-up landed,
|
||||
# and the generated recipe document owns that key in emitted records. A config
|
||||
# declaring it now fails validation loudly instead of being silently
|
||||
# overwritten by the generated one.
|
||||
V2_OPTIONAL_METADATA_FIELDS = (
|
||||
"recipe",
|
||||
"metric_threshold_policy",
|
||||
"quality_metadata",
|
||||
)
|
||||
RESULT_SCHEMA_VERSION = 2
|
||||
VALID_RUN_SOURCES = {"pr", "local", "scheduled_main", "unknown"}
|
||||
OPTIONAL_RESULT_METADATA_FIELDS = ("quality_metadata", "variant_metadata")
|
||||
|
||||
# -- Config discovery -------------------------------------------------------
|
||||
|
||||
@@ -215,6 +234,17 @@ def _run_generation(generator, prompt, generation_kwargs):
|
||||
component_times = _extract_component_times(result)
|
||||
return elapsed, peak_memory_mb, component_times
|
||||
|
||||
|
||||
def _validate_run_counts(run_config: Mapping[str, Any], benchmark_id: str) -> tuple[int, int]:
|
||||
num_warmup = run_config.get("num_warmup_runs", 1)
|
||||
num_measure = run_config.get("num_measurement_runs", 3)
|
||||
if isinstance(num_warmup, bool) or not isinstance(num_warmup, int) or num_warmup < 0:
|
||||
raise ValueError(f"{benchmark_id}: run_config.num_warmup_runs must be a non-negative integer")
|
||||
if isinstance(num_measure, bool) or not isinstance(num_measure, int) or num_measure < 1:
|
||||
raise ValueError(f"{benchmark_id}: run_config.num_measurement_runs must be a positive integer")
|
||||
return num_warmup, num_measure
|
||||
|
||||
|
||||
def _write_results(results):
|
||||
"""Write JSON results to the results directory."""
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
@@ -231,6 +261,205 @@ def _write_results(results):
|
||||
logger.info("Performance results written to %s", filepath)
|
||||
|
||||
|
||||
def _backend_name(value) -> str:
|
||||
if hasattr(value, "name"):
|
||||
return str(value.name)
|
||||
return str(value)
|
||||
|
||||
|
||||
def _collect_worker_identity(worker) -> dict[str, Any]:
|
||||
pipeline = getattr(worker, "pipeline", None)
|
||||
model_path = getattr(pipeline, "model_path", None)
|
||||
backends: set[str] = set()
|
||||
|
||||
modules = getattr(pipeline, "modules", None)
|
||||
if isinstance(modules, Mapping):
|
||||
module_iter = modules.values()
|
||||
elif isinstance(pipeline, torch.nn.Module):
|
||||
module_iter = (pipeline,)
|
||||
else:
|
||||
module_iter = ()
|
||||
|
||||
for module in module_iter:
|
||||
if isinstance(module, torch.nn.Module):
|
||||
for submodule in module.modules():
|
||||
backend = getattr(submodule, "backend", None)
|
||||
if backend is not None:
|
||||
backends.add(_backend_name(backend))
|
||||
else:
|
||||
backend = getattr(module, "backend", None)
|
||||
if backend is not None:
|
||||
backends.add(_backend_name(backend))
|
||||
|
||||
return {
|
||||
"resolved_attention_backends": sorted(backends),
|
||||
"resolved_model_path": model_path,
|
||||
}
|
||||
|
||||
|
||||
def _single_or_list(values: set[str]) -> str | list[str] | None:
|
||||
ordered = sorted(values)
|
||||
if not ordered:
|
||||
return None
|
||||
if len(ordered) == 1:
|
||||
return ordered[0]
|
||||
return ordered
|
||||
|
||||
|
||||
def _runtime_identity_from_generator(generator) -> dict[str, Any]:
|
||||
worker_records = generator.executor.collective_rpc(_collect_worker_identity)
|
||||
resolved_backends: set[str] = set()
|
||||
resolved_revisions: set[str] = set()
|
||||
|
||||
for record in worker_records:
|
||||
resolved_backends.update(record.get("resolved_attention_backends") or [])
|
||||
revision = resolved_revision_from_model_path(record.get("resolved_model_path"))
|
||||
if revision is not None:
|
||||
resolved_revisions.add(revision)
|
||||
|
||||
return {
|
||||
"resolved_attention_backend": _single_or_list(resolved_backends),
|
||||
"resolved_model_revision": _single_or_list(resolved_revisions),
|
||||
}
|
||||
|
||||
|
||||
def _benchmark_identity_fields(cfg):
|
||||
identity = benchmark_identity_from_config(cfg)
|
||||
return {
|
||||
key: identity[key]
|
||||
for key in ("workload_id", "variant_id", "benchmark_version")
|
||||
}
|
||||
|
||||
|
||||
def _build_identity_fields(cfg, init_kwargs, prompt, runtime_identity):
|
||||
# Legacy v1 configs (no config_schema_version) stay on the legacy
|
||||
# (model_id, gpu_type) cohort: they lack the required identity fields,
|
||||
# and building a recipe for them would raise AFTER the GPU measurement
|
||||
# completed. The validator already enforces that v2 configs carry the
|
||||
# identity fields, so v2 records always get the full identity block.
|
||||
if cfg.get("config_schema_version") is None:
|
||||
return {}
|
||||
recipe = build_recipe_from_benchmark_config(
|
||||
cfg,
|
||||
resolved_attention_backend=runtime_identity.get("resolved_attention_backend"),
|
||||
resolved_model_revision=runtime_identity.get("resolved_model_revision"),
|
||||
measured_prompts=[prompt],
|
||||
)
|
||||
num_gpus = init_kwargs.get("num_gpus", cfg.get("run_config", {}).get("required_gpus", 1))
|
||||
hw_profile = hardware_profile(num_gpus=num_gpus)
|
||||
sw_profile = software_profile()
|
||||
env_metadata = environment_metadata(
|
||||
hardware=hw_profile,
|
||||
software=sw_profile,
|
||||
)
|
||||
return {
|
||||
"recipe": recipe,
|
||||
"recipe_fingerprint": recipe_fingerprint(recipe),
|
||||
"hardware_profile": hw_profile,
|
||||
"hardware_profile_id": hardware_profile_id(hw_profile),
|
||||
"software_profile": sw_profile,
|
||||
"software_profile_id": software_profile_id(sw_profile),
|
||||
"environment_metadata": env_metadata,
|
||||
"environment_fingerprint": environment_fingerprint(env_metadata),
|
||||
**_benchmark_identity_fields(cfg),
|
||||
}
|
||||
|
||||
|
||||
def _truthy_pr_number(value: str | None) -> bool:
|
||||
return bool(value and value not in {"false", "0", "None", "none"})
|
||||
|
||||
|
||||
def _detect_run_source() -> str:
|
||||
explicit = os.environ.get("PERF_RUN_SOURCE", "").strip().lower()
|
||||
if explicit in VALID_RUN_SOURCES:
|
||||
return explicit
|
||||
if _truthy_pr_number(os.environ.get("BUILDKITE_PULL_REQUEST")):
|
||||
return "pr"
|
||||
if os.environ.get("BUILDKITE_BRANCH") == "main" and os.environ.get("TEST_SCOPE") == "full":
|
||||
return "scheduled_main"
|
||||
if not os.environ.get("BUILDKITE_COMMIT"):
|
||||
return "local"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _ci_provenance_fields() -> dict[str, str]:
|
||||
pr_number = os.environ.get("BUILDKITE_PULL_REQUEST", "")
|
||||
if not _truthy_pr_number(pr_number):
|
||||
pr_number = ""
|
||||
return {
|
||||
"run_source": _detect_run_source(),
|
||||
"branch": os.environ.get("BUILDKITE_BRANCH", ""),
|
||||
"pr_number": pr_number,
|
||||
"test_scope": os.environ.get("TEST_SCOPE", ""),
|
||||
"build_url": os.environ.get("BUILDKITE_BUILD_URL", ""),
|
||||
"build_id": os.environ.get("BUILDKITE_BUILD_ID", ""),
|
||||
"job_id": os.environ.get("BUILDKITE_JOB_ID", ""),
|
||||
}
|
||||
|
||||
|
||||
def _configured_result_metadata(cfg: Mapping[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
field: cfg[field]
|
||||
for field in OPTIONAL_RESULT_METADATA_FIELDS
|
||||
if field in cfg and cfg[field] is not None
|
||||
}
|
||||
|
||||
|
||||
def _build_result_record(
|
||||
*,
|
||||
cfg: Mapping[str, Any],
|
||||
model_info: Mapping[str, Any],
|
||||
init_kwargs: Mapping[str, Any],
|
||||
gen_kwargs: Mapping[str, Any],
|
||||
num_warmup: int,
|
||||
num_measure: int,
|
||||
thresholds: Mapping[str, Any],
|
||||
times: list[float],
|
||||
peak_memories: list[float],
|
||||
all_component_times: list[dict],
|
||||
prompt: str,
|
||||
runtime_identity: Mapping[str, Any],
|
||||
device_name: str,
|
||||
timestamp: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
if not times or not peak_memories:
|
||||
raise ValueError("Cannot build a performance result record without measurement runs")
|
||||
avg_time = sum(times) / len(times)
|
||||
max_peak_memory = max(peak_memories)
|
||||
num_frames = gen_kwargs.get("num_frames")
|
||||
throughput_fps = (1.0 / avg_time) if avg_time > 0 else None
|
||||
if isinstance(num_frames, (int, float)) and avg_time > 0:
|
||||
throughput_fps = num_frames / avg_time
|
||||
|
||||
return {
|
||||
"benchmark_id": cfg["benchmark_id"],
|
||||
"result_schema_version": RESULT_SCHEMA_VERSION,
|
||||
"model_short_name": model_info.get("model_short_name", ""),
|
||||
"device": device_name,
|
||||
"num_gpus": init_kwargs.get("num_gpus", 1),
|
||||
"num_warmup_runs": num_warmup,
|
||||
"num_measurement_runs": num_measure,
|
||||
"avg_generation_time_s": round(avg_time, 3),
|
||||
"individual_times_s": [round(t, 3) for t in times],
|
||||
"throughput_fps": round(throughput_fps, 3)
|
||||
if throughput_fps is not None else None,
|
||||
"max_peak_memory_mb": round(max_peak_memory, 1),
|
||||
"individual_peak_memories_mb": [round(m, 1) for m in peak_memories],
|
||||
"thresholds": dict(thresholds),
|
||||
"regression_thresholds": cfg.get("regression_thresholds", {}),
|
||||
"commit": os.environ.get("BUILDKITE_COMMIT", ""),
|
||||
**_ci_provenance_fields(),
|
||||
"timestamp": timestamp or datetime.now(timezone.utc).isoformat(),
|
||||
**_configured_result_metadata(cfg),
|
||||
"text_encoder_time_s": _avg_component(all_component_times,
|
||||
"text_encoder_time_s"),
|
||||
"dit_time_s": _avg_component(all_component_times, "dit_time_s"),
|
||||
"vae_decode_time_s": _avg_component(all_component_times,
|
||||
"vae_decode_time_s"),
|
||||
**_build_identity_fields(cfg, init_kwargs, prompt, runtime_identity),
|
||||
}
|
||||
|
||||
|
||||
# -- Test -------------------------------------------------------------------
|
||||
|
||||
def _run_benchmark(cfg):
|
||||
@@ -246,8 +475,7 @@ def _run_benchmark(cfg):
|
||||
prompts = cfg.get("test_prompts", ["A cinematic video."])
|
||||
prompt = prompts[0]
|
||||
|
||||
num_warmup = run_config.get("num_warmup_runs", 1)
|
||||
num_measure = run_config.get("num_measurement_runs", 3)
|
||||
num_warmup, num_measure = _validate_run_counts(run_config, cfg["benchmark_id"])
|
||||
thresholds = _get_thresholds(cfg)
|
||||
|
||||
# Remap JSON keys to VideoGenerator kwargs
|
||||
@@ -268,6 +496,7 @@ def _run_benchmark(cfg):
|
||||
model_path=model_info["model_path"],
|
||||
**init_kwargs,
|
||||
)
|
||||
runtime_identity = _runtime_identity_from_generator(generator)
|
||||
|
||||
for i in range(num_warmup):
|
||||
logger.info("Warmup run %d/%d", i + 1, num_warmup)
|
||||
@@ -293,36 +522,21 @@ def _run_benchmark(cfg):
|
||||
avg_time = sum(times) / len(times)
|
||||
max_peak_memory = max(peak_memories)
|
||||
device_name = torch.cuda.get_device_name()
|
||||
num_frames = gen_kwargs.get("num_frames")
|
||||
throughput_fps = (1.0 / avg_time) if avg_time > 0 else None
|
||||
if isinstance(num_frames, (int, float)) and avg_time > 0:
|
||||
throughput_fps = num_frames / avg_time
|
||||
|
||||
results = {
|
||||
"benchmark_id": cfg["benchmark_id"],
|
||||
**_config_identity_metadata(cfg),
|
||||
"model_short_name": model_info.get("model_short_name", ""),
|
||||
"device": device_name,
|
||||
"num_gpus": init_kwargs.get("num_gpus", 1),
|
||||
"num_warmup_runs": num_warmup,
|
||||
"num_measurement_runs": num_measure,
|
||||
"avg_generation_time_s": round(avg_time, 3),
|
||||
"individual_times_s": [round(t, 3) for t in times],
|
||||
"throughput_fps": round(throughput_fps, 3)
|
||||
if throughput_fps is not None else None,
|
||||
"max_peak_memory_mb": round(max_peak_memory, 1),
|
||||
"individual_peak_memories_mb": [round(m, 1) for m in peak_memories],
|
||||
"thresholds": thresholds,
|
||||
"regression_thresholds": cfg.get("regression_thresholds", {}),
|
||||
"commit": os.environ.get("BUILDKITE_COMMIT", ""),
|
||||
"pr_number": os.environ.get("BUILDKITE_PULL_REQUEST", ""),
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"text_encoder_time_s": _avg_component(all_component_times,
|
||||
"text_encoder_time_s"),
|
||||
"dit_time_s": _avg_component(all_component_times, "dit_time_s"),
|
||||
"vae_decode_time_s": _avg_component(all_component_times,
|
||||
"vae_decode_time_s"),
|
||||
}
|
||||
results = _build_result_record(
|
||||
cfg=cfg,
|
||||
model_info=model_info,
|
||||
init_kwargs=init_kwargs,
|
||||
gen_kwargs=gen_kwargs,
|
||||
num_warmup=num_warmup,
|
||||
num_measure=num_measure,
|
||||
thresholds=thresholds,
|
||||
times=times,
|
||||
peak_memories=peak_memories,
|
||||
all_component_times=all_component_times,
|
||||
prompt=prompt,
|
||||
runtime_identity=runtime_identity,
|
||||
device_name=device_name,
|
||||
)
|
||||
|
||||
logger.info(
|
||||
"Performance results: avg_time=%.2fs, "
|
||||
|
||||
@@ -0,0 +1,176 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import pytest
|
||||
|
||||
from fastvideo.tests.performance import test_inference_performance as perf_test
|
||||
|
||||
|
||||
def _benchmark_config():
|
||||
return {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"quality_metadata": {
|
||||
"quality_status": "canonical",
|
||||
},
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "Wan2.1-T2V-1.3B",
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"num_frames": 45,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _identity_fields():
|
||||
return {
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
"recipe": {
|
||||
"recipe_schema_version": 1,
|
||||
"benchmark": {
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 2,
|
||||
},
|
||||
},
|
||||
"recipe_fingerprint": "recipe-1",
|
||||
"hardware_profile": {
|
||||
"device_type": "cuda",
|
||||
"gpu_count": 2,
|
||||
},
|
||||
"hardware_profile_id": "hw-l40s",
|
||||
"software_profile": {
|
||||
"python": "3.12",
|
||||
"pytorch": "2.12",
|
||||
"cuda": "13.0",
|
||||
},
|
||||
"software_profile_id": "sw-cu130",
|
||||
"environment_metadata": {
|
||||
"env": {
|
||||
"IMAGE_VERSION": "py3.12-cuda13.0.0",
|
||||
},
|
||||
},
|
||||
"environment_fingerprint": "env-ci",
|
||||
}
|
||||
|
||||
|
||||
def test_build_result_record_emits_v2_wan_shape(monkeypatch):
|
||||
monkeypatch.setenv("PERF_RUN_SOURCE", "scheduled_main")
|
||||
monkeypatch.setenv("BUILDKITE_COMMIT", "a" * 40)
|
||||
monkeypatch.setenv("BUILDKITE_PULL_REQUEST", "false")
|
||||
monkeypatch.setenv("BUILDKITE_BRANCH", "main")
|
||||
monkeypatch.setenv("TEST_SCOPE", "full")
|
||||
monkeypatch.setenv("BUILDKITE_BUILD_URL", "https://buildkite.example/build")
|
||||
monkeypatch.setenv("BUILDKITE_BUILD_ID", "build-1")
|
||||
monkeypatch.setenv("BUILDKITE_JOB_ID", "job-1")
|
||||
monkeypatch.setattr(perf_test, "_build_identity_fields", lambda *_args: _identity_fields())
|
||||
|
||||
record = perf_test._build_result_record(
|
||||
cfg=_benchmark_config(),
|
||||
model_info={"model_short_name": "Wan2.1-T2V-1.3B"},
|
||||
init_kwargs={"num_gpus": 2},
|
||||
gen_kwargs={"num_frames": 45},
|
||||
num_warmup=2,
|
||||
num_measure=3,
|
||||
thresholds={
|
||||
"max_generation_time_s": 34.0,
|
||||
"max_peak_memory_mb": 11000.0,
|
||||
},
|
||||
times=[30.0, 31.0, 32.0],
|
||||
peak_memories=[10000.0, 10100.0, 10050.0],
|
||||
all_component_times=[
|
||||
{
|
||||
"text_encoder_time_s": 1.0,
|
||||
"dit_time_s": 8.0,
|
||||
"vae_decode_time_s": 3.0,
|
||||
},
|
||||
{
|
||||
"text_encoder_time_s": 1.2,
|
||||
"dit_time_s": 8.2,
|
||||
"vae_decode_time_s": 3.2,
|
||||
},
|
||||
{
|
||||
"text_encoder_time_s": None,
|
||||
"dit_time_s": 8.4,
|
||||
"vae_decode_time_s": 3.4,
|
||||
},
|
||||
],
|
||||
prompt="A cinematic video.",
|
||||
runtime_identity={
|
||||
"resolved_attention_backend": "FLASH_ATTN",
|
||||
},
|
||||
device_name="NVIDIA L40S",
|
||||
timestamp="2026-07-05T00:00:00+00:00",
|
||||
)
|
||||
|
||||
assert record["result_schema_version"] == perf_test.RESULT_SCHEMA_VERSION
|
||||
assert record["benchmark_id"] == "wan-t2v-1.3b-2gpu"
|
||||
assert record["workload_id"] == "wan-t2v"
|
||||
assert record["variant_id"] == "1.3b-sp2"
|
||||
assert record["benchmark_version"] == 2
|
||||
assert record["recipe_fingerprint"] == "recipe-1"
|
||||
assert record["hardware_profile_id"] == "hw-l40s"
|
||||
assert record["software_profile_id"] == "sw-cu130"
|
||||
assert record["environment_fingerprint"] == "env-ci"
|
||||
assert record["environment_metadata"]["env"]["IMAGE_VERSION"] == "py3.12-cuda13.0.0"
|
||||
assert record["quality_metadata"] == {"quality_status": "canonical"}
|
||||
assert record["run_source"] == "scheduled_main"
|
||||
assert record["branch"] == "main"
|
||||
assert record["test_scope"] == "full"
|
||||
assert record["build_url"] == "https://buildkite.example/build"
|
||||
assert record["build_id"] == "build-1"
|
||||
assert record["job_id"] == "job-1"
|
||||
assert record["pr_number"] == ""
|
||||
assert record["commit"] == "a" * 40
|
||||
assert record["avg_generation_time_s"] == 31.0
|
||||
assert record["throughput_fps"] == 1.452
|
||||
assert record["max_peak_memory_mb"] == 10100.0
|
||||
assert record["text_encoder_time_s"] == 1.1
|
||||
assert record["dit_time_s"] == 8.2
|
||||
assert record["vae_decode_time_s"] == 3.2
|
||||
|
||||
|
||||
def test_validate_run_counts_rejects_zero_measurement_runs():
|
||||
with pytest.raises(ValueError, match="num_measurement_runs"):
|
||||
perf_test._validate_run_counts({
|
||||
"num_warmup_runs": 1,
|
||||
"num_measurement_runs": 0,
|
||||
}, "wan-t2v-1.3b-2gpu")
|
||||
|
||||
|
||||
def test_validate_run_counts_rejects_negative_warmup_runs():
|
||||
with pytest.raises(ValueError, match="num_warmup_runs"):
|
||||
perf_test._validate_run_counts({
|
||||
"num_warmup_runs": -1,
|
||||
"num_measurement_runs": 3,
|
||||
}, "wan-t2v-1.3b-2gpu")
|
||||
|
||||
|
||||
def test_validate_run_counts_returns_defaults():
|
||||
assert perf_test._validate_run_counts({}, "wan-t2v-1.3b-2gpu") == (1, 3)
|
||||
|
||||
|
||||
def test_build_result_record_rejects_empty_measurements(monkeypatch):
|
||||
monkeypatch.setattr(perf_test, "_build_identity_fields", lambda *_args: _identity_fields())
|
||||
|
||||
with pytest.raises(ValueError, match="measurement runs"):
|
||||
perf_test._build_result_record(
|
||||
cfg=_benchmark_config(),
|
||||
model_info={},
|
||||
init_kwargs={},
|
||||
gen_kwargs={},
|
||||
num_warmup=0,
|
||||
num_measure=0,
|
||||
thresholds={},
|
||||
times=[],
|
||||
peak_memories=[],
|
||||
all_component_times=[],
|
||||
prompt="A cinematic video.",
|
||||
runtime_identity={},
|
||||
device_name="NVIDIA L40S",
|
||||
)
|
||||
Reference in New Issue
Block a user