Compare commits

...
14 Commits
Author SHA1 Message Date
SolitaryThinker 19d9ee06fe [test]: align benchmark-config fixtures with the merged identity taxonomy 2026-07-05 16:08:27 -07:00
Mac Lee eb75e08f6e [ci]: validate performance measurement counts (#1531) 2026-07-05 16:06:49 -07:00
Mac Lee 9bba72edab [ci]: tolerate legacy performance result artifacts (#1531) 2026-07-05 16:06:41 -07:00
Mac Lee 95a775b14a [docs]: clarify scheduled-main performance uploads (#1531) 2026-07-05 16:03:38 -07:00
Mac Lee bc51cd6d37 [ci]: fix v2 dashboard cohort handling (#1531) 2026-07-05 16:02:56 -07:00
Mac Lee f43bcdefae [ci]: emit v2 performance result schema (#1531) 2026-07-05 15:59:43 -07:00
SolitaryThinker 2e0dba1a4f [ci]: adopt per-parallelism-config baseline cohorts for wan-t2v (author decision)
Flip the wan benchmark identity from workload wan-t2v-1.3b / variant
canonical / benchmark_version 1 to workload wan-t2v / variant 1.3b-sp2 /
benchmark_version 2, adopting the contributor's per-parallelism-config
cohort labels and matching the #1551 fixtures. This deliberately opens a
new comparison cohort; the baseline reseed follows post-merge.
config_schema_version stays 2 as required by the #1544 validator.
2026-07-05 14:39:57 -07:00
SolitaryThinker dc8a661696 [bugfix]: address verify-pass findings on the 1546 fix branch
- recipe_fingerprint also excludes attention.resolved_backend (a silent
  runtime fallback must fail the gate in its cohort, not open a fresh
  ungated one); requested_backend stays in the hash; regression test
  extended to both runtime-resolved values.
- Legacy v1 configs no longer crash after the GPU measurement: identity
  fields are emitted only for config_schema_version=2 configs, and the
  comparator falls back to the legacy (model_id, gpu_type) cohort when a
  record carries no identity fields (partial identity still raises).
- 'recipe' removed from V2_OPTIONAL_METADATA_FIELDS: the generated recipe
  owns the key now; declaring it in a config fails validation loudly
  instead of being silently overwritten.
- Docs: stale 'until those land' paragraph reconciled with what this
  branch actually lands; baseline_status documented.
2026-07-05 14:39:57 -07:00
SolitaryThinker e03864a852 [docs]: align example records with the merged identity taxonomy 2026-07-05 14:39:57 -07:00
SolitaryThinker 5902ab2669 [bugfix]: pin recipe_fingerprint to declared inputs and make cohort initialization loud
Rebase of #1546 onto main resolved the identity taxonomy in favor of the
merged #1544 values (workload_id=wan-t2v-1.3b, variant_id=canonical,
benchmark_version=1, config_schema_version=2). On top:
- recipe_fingerprint() no longer hashes the runtime-resolved HF snapshot
  revision (kept in the recipe as audit metadata) — upstream repo commits
  with unchanged weights no longer churn the cohort and reset baselines.
- The no-baseline branch now emits a loud warning block and a
  machine-readable baseline_status=initialized_new_cohort field instead of
  an indistinguishable success, so cohort shifts never silently disable
  regression gating.
- Regression test: resolved revision must not change the fingerprint.
2026-07-05 14:39:57 -07:00
Mac Lee 53ce043b32 [ci]: align remaining performance cohort consumers 2026-07-05 14:39:57 -07:00
Mac Lee 7b74b94ad4 [ci]: complete performance cohort identity 2026-07-05 14:39:57 -07:00
Mac Lee 752248e466 [ci]: tighten performance cohort identity 2026-07-05 14:39:57 -07:00
Mac Lee cc7848ab46 [ci]: add performance fingerprint cohorts 2026-07-05 14:39:57 -07:00
18 changed files with 2436 additions and 147 deletions
@@ -1,9 +1,9 @@
{
"benchmark_id": "wan-t2v-1.3b-2gpu",
"config_schema_version": 2,
"workload_id": "wan-t2v-1.3b",
"variant_id": "canonical",
"benchmark_version": 1,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"description": "Wan2.1 T2V 1.3B inference performance",
"model": {
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
@@ -1,6 +1,7 @@
import { useEffect, useMemo, useState } from "react";
import { fetchSummary, fetchTrends, refreshData, RunSource, SummaryResponse, TrendGroup, TrendPoint } from "./api";
import { fetchSummary, fetchTrends, refreshData } from "./api";
import type { CohortValue, RunSource, SummaryResponse, TrendGroup, TrendPoint } from "./api";
const METRIC_KEYS = ["latency", "throughput", "memory", "text_encoder_time_s", "dit_time_s", "vae_decode_time_s"];
const RUN_SOURCES: Array<{ value: "" | RunSource; label: string }> = [
@@ -109,6 +110,61 @@ function metricLabel(metricKey: string) {
return METRIC_DEFINITIONS[metricKey]?.label ?? metricKey;
}
type CohortFields = {
model_id: string;
gpu_type: string;
workload_id: CohortValue;
variant_id: CohortValue;
benchmark_version: CohortValue;
recipe_fingerprint: CohortValue;
hardware_profile_id: CohortValue;
software_profile_id: CohortValue;
};
function cohortValue(value: CohortValue) {
if (value === null || value === undefined || value === "") {
return "legacy";
}
return String(value);
}
function shortCohortValue(value: CohortValue) {
const text = cohortValue(value);
if (text === "legacy" || text.length <= 14) {
return text;
}
return text.slice(0, 12);
}
function cohortKey(cohort: CohortFields) {
return [
cohort.model_id,
cohort.gpu_type,
cohortValue(cohort.workload_id),
cohortValue(cohort.variant_id),
cohortValue(cohort.benchmark_version),
cohortValue(cohort.recipe_fingerprint),
cohortValue(cohort.hardware_profile_id),
cohortValue(cohort.software_profile_id)
].join("|");
}
function cohortTitle(cohort: CohortFields) {
const workload = cohortValue(cohort.workload_id);
const variant = cohortValue(cohort.variant_id);
const version = cohortValue(cohort.benchmark_version);
const versionLabel = version === "legacy" ? version : `v${version}`;
return `${workload} / ${variant} / ${versionLabel}`;
}
function cohortDetail(cohort: CohortFields) {
return [
`recipe ${shortCohortValue(cohort.recipe_fingerprint)}`,
shortCohortValue(cohort.hardware_profile_id),
shortCohortValue(cohort.software_profile_id)
].join(" | ");
}
function formatMetricValue(metricKey: string, value: number | null | undefined, tooltip = false) {
const definition = METRIC_DEFINITIONS[metricKey];
if (!definition) {
@@ -171,7 +227,9 @@ function TrendChart({ group, metricKey }: { group: TrendGroup; metricKey: string
top: `${(activePoint.y / height) * 100}%`
}
: undefined;
const ariaLabel = `${metricLabel(metricKey)} trend for ${group.model_id} on ${group.gpu_type}`;
const ariaLabel = `${metricLabel(metricKey)} trend for ${group.model_id} on ${group.gpu_type}, ${cohortTitle(
group
)}`;
return (
<div className="chart-shell">
@@ -419,7 +477,7 @@ export default function App() {
<section className="panel">
<div className="panel-header">
<h2>Latest Status</h2>
<span>{latestRows.length} model/GPU groups</span>
<span>{latestRows.length} comparison cohorts</span>
</div>
{latestRows.length === 0 ? (
<div className="empty">No records match the selected filters.</div>
@@ -432,6 +490,7 @@ export default function App() {
<th>Recomputed</th>
<th>Model</th>
<th>GPU</th>
<th>Cohort</th>
<th>Commit</th>
<th>Source</th>
<th>Baseline</th>
@@ -446,7 +505,7 @@ export default function App() {
</thead>
<tbody>
{latestRows.map((row) => (
<tr key={`${row.model_id}-${row.gpu_type}`}>
<tr key={cohortKey(row)}>
<td>
<span className={`badge ${row.status}`}>{row.status}</span>
</td>
@@ -457,6 +516,12 @@ export default function App() {
</td>
<td>{row.model_id}</td>
<td>{row.gpu_type}</td>
<td>
<div className="cohort-cell">
<strong>{cohortTitle(row)}</strong>
<span>{cohortDetail(row)}</span>
</div>
</td>
<td>{shortSha(row.commit_sha)}</td>
<td>
<span className={`source-badge source-${row.run_source}`}>{runSourceLabel(row.run_source)}</span>
@@ -495,11 +560,13 @@ export default function App() {
) : (
trends.map((group) =>
METRIC_KEYS.map((metricKey) => (
<article className="trend-card" key={`${group.model_id}-${group.gpu_type}-${metricKey}`}>
<article className="trend-card" key={`${cohortKey(group)}-${metricKey}`}>
<div>
<h3>{metricLabel(metricKey)}</h3>
<p>
{group.model_id} | {group.gpu_type}
<span>{cohortTitle(group)}</span>
<span>{cohortDetail(group)}</span>
</p>
</div>
<TrendChart group={group} metricKey={metricKey} />
+14 -3
View File
@@ -13,6 +13,17 @@ export type MetricValue = {
precision: number;
};
export type CohortValue = string | number | null;
export type ComparisonCohort = {
workload_id: CohortValue;
variant_id: CohortValue;
benchmark_version: CohortValue;
recipe_fingerprint: CohortValue;
hardware_profile_id: CohortValue;
software_profile_id: CohortValue;
};
export type SummaryRow = {
model_id: string;
gpu_type: string;
@@ -34,7 +45,7 @@ export type SummaryRow = {
build_id: string;
job_id: string;
metrics: Record<string, MetricValue>;
};
} & ComparisonCohort;
export type RunSource = "pr" | "local" | "scheduled_main" | "unknown";
@@ -68,13 +79,13 @@ export type TrendPoint = {
build_id: string;
job_id: string;
metrics: Record<string, number | null>;
};
} & ComparisonCohort;
export type TrendGroup = {
model_id: string;
gpu_type: string;
points: TrendPoint[];
};
} & ComparisonCohort;
export type TrendsResponse = {
groups: TrendGroup[];
@@ -149,6 +149,11 @@ h3 {
font-size: 0.82rem;
}
.trend-card p {
display: grid;
gap: 2px;
}
.stat strong {
display: block;
margin-top: 8px;
@@ -186,7 +191,7 @@ h3 {
table {
width: 100%;
min-width: 1120px;
min-width: 1260px;
border-collapse: collapse;
}
@@ -209,6 +214,25 @@ td {
font-size: 0.9rem;
}
.cohort-cell {
display: grid;
gap: 2px;
}
.cohort-cell strong,
.trend-card p span {
color: #1b2836;
font-size: 0.78rem;
font-weight: 700;
}
.cohort-cell span,
.trend-card p span + span {
color: #607080;
font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, "Liberation Mono", monospace;
font-size: 0.72rem;
}
.badge {
display: inline-flex;
align-items: center;
+131 -40
View File
@@ -79,10 +79,11 @@ fastvideo/performance/
```
The HF dataset (`FastVideo/performance-tracking` by default) holds one
normalized JSON per `(model_id, gpu_type, run)` tuple. The rolling baseline is
the median of the last 5 successful, baseline-eligible records for that
model+GPU. PR and local records are visible in the dashboard but are not
baseline eligible.
normalized JSON per run. For v2 records, the rolling baseline is the median of
the last 5 successful, baseline-eligible records in the same comparison cohort:
`model_id`, `gpu_type`, `workload_id`, `variant_id`, `benchmark_version`,
`recipe_fingerprint`, `hardware_profile_id`, and `software_profile_id`. PR and
local records are visible in the dashboard but are not baseline eligible.
## Planned Coverage
@@ -158,13 +159,16 @@ unrealistic memory growth, and optionally large component-specific slowdowns
even when the rolling baseline is empty. They are hand-set with generous
headroom and almost never need touching.
### Rolling baseline (per `(model_id, gpu_type)`)
### Rolling baseline (per comparison cohort)
`compare_baseline.py` loads the last 5 successful, baseline-eligible records
for the same `(model_id, gpu_type)` from the HF dataset, computes the median
for each available metric, and evaluates the current run with the metric's
rolling regression policy. For latency, memory, and component times, higher
values are regressions. For throughput, lower values are regressions.
for the same comparison cohort from the HF dataset, computes the median for
each available metric, and evaluates the current run with the metric's
rolling regression policy. For v2 records, that cohort is `model_id`,
`gpu_type`, `workload_id`, `variant_id`, `benchmark_version`,
`recipe_fingerprint`, `hardware_profile_id`, and `software_profile_id`. For
latency, memory, and component times, higher values are regressions. For
throughput, lower values are regressions.
A metric exceeds its rolling threshold when both of these are true:
@@ -201,9 +205,9 @@ configs and remain loadable. New or migrated configs should use
{
"benchmark_id": "wan-t2v-1.3b-2gpu",
"config_schema_version": 2,
"workload_id": "wan-t2v-1.3b",
"variant_id": "canonical",
"benchmark_version": 1
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2
}
```
@@ -214,21 +218,27 @@ metadata that make the measured workload explicit:
| Field | Purpose |
|---|---|
| `workload_id` | Stable benchmark family, such as `wan-t2v-1.3b`. |
| `variant_id` | Intentional recipe family, such as `canonical`. |
| `workload_id` | Stable benchmark family, such as `wan-t2v`. |
| `variant_id` | Intentional recipe family, including model size and parallelism config, such as `1.3b-sp2`. |
| `benchmark_version` | Version of the measurement protocol and comparison policy. |
If a config declares `config_schema_version: 2`, loading fails clearly when any
required v2 identity field is missing. If v2 identity or metadata fields are
added without `config_schema_version: 2`, loading also fails so partial
migrations do not silently run as v1 configs. Optional v2 metadata fields
reserved for follow-up work, such as `recipe`, `metric_threshold_policy`, and
`quality_metadata`, must be JSON objects when present.
reserved for follow-up work, such as `metric_threshold_policy` and
`quality_metadata`, must be JSON objects when present. (`recipe` is emitted
by the harness and is not config-declarable.)
Recipe fingerprinting, hardware/software profile IDs, exact-identity
comparison, metric-specific threshold policy behavior, promoted baselines, and
dashboard regrouping are separate follow-up changes. Until those land, rolling
baseline comparison remains keyed by `(model_id, gpu_type)`.
comparison, and dashboard cohort grouping land with this change: v2 records
compare only within their identity cohort, and a record that opens a NEW
cohort is marked `baseline_status: "initialized_new_cohort"` (regression
gating starts once that cohort accumulates history). Legacy v1 configs still
run and are normalized for reporting, but their records skip rolling-baseline
comparison entirely (`baseline_status: "skipped_missing_identity"`, never
baseline eligible); only static thresholds gate them. Metric-specific
threshold policies and promoted baselines remain separate follow-ups.
### Raw record (`results/perf_*.json`)
@@ -237,10 +247,10 @@ Written by `test_inference_performance.py`. One file per benchmark run.
```jsonc
{
"benchmark_id": "wan-t2v-1.3b-2gpu",
"config_schema_version": 2,
"workload_id": "wan-t2v-1.3b",
"variant_id": "canonical",
"benchmark_version": 1,
"result_schema_version": 2,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"model_short_name": "Wan2.1-T2V-1.3B-Diffusers",
"device": "NVIDIA L40S",
"num_gpus": 2,
@@ -266,11 +276,54 @@ Written by `test_inference_performance.py`. One file per benchmark run.
}
},
"commit": "<full sha>",
"run_source": "pr",
"branch": "feature/perf-change",
"pr_number": "1234",
"test_scope": "direct",
"build_url": "https://buildkite.example/build",
"build_id": "<buildkite-build-id>",
"job_id": "<buildkite-job-id>",
"timestamp": "2026-05-08T22:00:00+00:00",
"quality_metadata": { "quality_status": "canonical" },
"text_encoder_time_s": 2.141,
"dit_time_s": 8.437,
"vae_decode_time_s": 3.208
"vae_decode_time_s": 3.208,
"recipe": {
"recipe_schema_version": 1,
"benchmark": {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2
},
"model": { "model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" },
"init_kwargs": { "num_gpus": 2, "sp_size": 2, "tp_size": 1 },
"generation_kwargs": { "height": 480, "width": 832, "num_frames": 45 },
"inputs": { "prompt_count": 1, "prompt_sha256": ["<measured-prompt-sha256>"] },
"attention": { "requested_backend": "FLASH_ATTN", "resolved_backend": "FLASH_ATTN" }
},
"recipe_fingerprint": "<sha256>",
"hardware_profile": {
"device_type": "cuda",
"gpu_count": 2,
"gpus": [{ "name": "NVIDIA L40S", "memory_gb": 48, "compute_capability": "8.9" }],
"interconnect": "none_or_partial"
},
"hardware_profile_id": "hw-<sha256-prefix>",
"software_profile": {
"python": "3.12",
"pytorch": "2.12",
"cuda": "13.0",
"packages": {
"fastvideo_kernel": "0.3.2",
"flashinfer": "0.2.11",
"nvidia_cutlass_dsl": "4.5.0",
"triton": "3.4.1"
}
},
"software_profile_id": "sw-<sha256-prefix>",
"environment_metadata": { "env": { "IMAGE_VERSION": "py3.12-cuda13.0.0" } },
"environment_fingerprint": "env-<sha256-prefix>"
}
```
@@ -282,6 +335,10 @@ result, used as the rolling-baseline source of truth.
```jsonc
{
"model_id": "wan-t2v-1.3b-2gpu",
"result_schema_version": 2,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"timestamp": "2026-05-08T22:00:00+00:00",
"commit_sha": "<full sha>",
"gpu_type": "NVIDIA L40S",
@@ -298,17 +355,46 @@ result, used as the rolling-baseline source of truth.
"gated": true
}
},
"recipe_fingerprint": "<sha256>",
"hardware_profile_id": "hw-<sha256-prefix>",
"software_profile_id": "sw-<sha256-prefix>",
"environment_fingerprint": "env-<sha256-prefix>",
"run_source": "pr",
"branch": "feature/perf-change",
"pr_number": "1234",
"test_scope": "direct",
"build_url": "https://buildkite.example/build",
"build_id": "<buildkite-build-id>",
"job_id": "<buildkite-job-id>",
"quality_metadata": { "quality_status": "canonical" },
"success": true
}
```
### Compatibility with legacy records
Older records in the HF dataset may not have component timing fields. The
comparator ignores missing or `null` metrics when computing a median, and the
dashboard lists skipped plots for metric series that have no non-null values.
Records missing both `run_source` and `baseline_eligible` are treated as legacy
successful main/full-suite uploads and remain eligible for rolling baselines.
Older records in the HF dataset may not have `result_schema_version`,
component timing fields, or v2 identity/profile fields. Records without
`result_schema_version` are treated as v1. The comparator ignores missing or
`null` metrics when computing a median, and the dashboard lists skipped plots
for metric series that have no non-null values. Records missing both
`run_source` and `baseline_eligible` are treated as legacy successful
main/full-suite uploads and remain eligible for rolling baselines.
Current `perf_*.json` artifacts that lack the v2 comparison identity are
normalized for reporting but skip rolling-baseline comparison and are not marked
baseline eligible.
New records compare only against the same `model_id`, `gpu_type`,
`workload_id`, `variant_id`, `benchmark_version`, `recipe_fingerprint`,
`hardware_profile_id`, and `software_profile_id` cohort.
`environment_metadata` and `environment_fingerprint` are audit data and are not
part of the comparison key.
The recipe prompt digests describe the prompts actually measured by the
benchmark run; extra configured prompts are ignored unless the benchmark runner
executes them.
Software profile package cohorts keep exact versions for relevant
attention/kernel packages, including FastVideo kernels, FlashAttention,
FlashInfer, Cutlass DSL, SageAttention, Triton, and xFormers when installed.
## Environment variable reference
@@ -318,7 +404,7 @@ successful main/full-suite uploads and remain eligible for rolling baselines.
| `PERF_REPORTS_DIR` | `/root/data/perf_reports` | `compare_baseline.py`, `dashboard.py` | Where the Markdown summary and Plotly HTML get written for Buildkite to pick up. |
| `HF_REPO_ID` | `FastVideo/performance-tracking` | `fastvideo/performance/hf_store.py` | HF dataset repo holding rolling-baseline records. |
| `HF_API_KEY`, `HUGGINGFACE_HUB_TOKEN`, `HF_TOKEN` | unset | `fastvideo/performance/hf_store.py` | Required for upload or private dataset reads. |
| `PERF_RUN_SOURCE` | inferred | `compare_baseline.py` | Source metadata for uploaded records: `pr`, `local`, `scheduled_main`, or `unknown`. |
| `PERF_RUN_SOURCE` | inferred | `compare_baseline.py`, `test_inference_performance.py` | Source metadata for uploaded records: `pr`, `local`, `scheduled_main`, or `unknown`. |
| `PERF_UPLOAD_POLICY` | `never` | `compare_baseline.py` | Upload policy: `never`, `pass`, or `always`. |
| `PERF_PYTEST_RC` | unset | `compare_baseline.py` | Static-threshold pytest exit code, used so scheduled-main failures can be uploaded with `success=false`. |
| `TEST_SCOPE` | unset | `compare_baseline.py` | CI context used to infer scheduled-main runs together with `BUILDKITE_BRANCH=main`. |
@@ -335,11 +421,14 @@ point is `fastvideo/tests/modal/pr_test.py:run_performance_tests` and the
Buildkite artifact upload is in
`.buildkite/scripts/pr_test.sh:upload_performance_artifacts`.
Each performance build runs pytest first. If that fixed-threshold phase fails,
PR/direct runs skip `compare_baseline.py` because they only upload passing
records. Scheduled-main runs still execute `compare_baseline.py` with
`PERF_PYTEST_RC` set so the failed canonical attempt is visible in normalized
JSON and dashboard history. The dashboard runs best-effort for observability.
Each performance build runs pytest first. PR and direct runs only continue to
`compare_baseline.py` when that fixed-threshold phase passes; if pytest fails,
Markdown summaries and normalized JSON artifacts are not emitted. Scheduled
main runs set `PERF_UPLOAD_POLICY=always`, so they still run
`compare_baseline.py` (with `PERF_PYTEST_RC` set) after a fixed-threshold
failure. Those failed scheduled main runs emit summaries and normalized
records, upload records with `success=false`, and are excluded from future
rolling baselines. The dashboard still runs best-effort for observability.
When the rolling-baseline phase runs, it emits:
* **Markdown summary** — appended to `$GITHUB_STEP_SUMMARY` when that variable
@@ -347,7 +436,7 @@ When the rolling-baseline phase runs, it emits:
per-benchmark row with current vs. baseline values for latency, throughput,
memory, text encoder time, DiT time, and VAE decode time.
* **Plotly dashboard** — `dashboard_<sha>_<ts>.html` showing time-series for
each metric grouped by `(model_id, gpu_type)`.
each metric grouped by comparison cohort.
* **Normalized records** — `normalized_perf_*.json`, one per benchmark.
Useful as input to the
[`reseed-performance-baseline`](https://github.com/hao-ai-lab/FastVideo/blob/main/.agents/skills/reseed-performance-baseline/SKILL.md)
@@ -364,7 +453,7 @@ When the rolling-baseline phase runs, it emits:
"benchmark_id": "<unique-id>",
"config_schema_version": 2,
"workload_id": "<stable-workload-id>",
"variant_id": "canonical",
"variant_id": "<variant, e.g. 1.3b-sp2>",
"benchmark_version": 1,
"model": { "model_path": "...", "model_short_name": "..." },
"init_kwargs": { "num_gpus": 1, ... },
@@ -391,7 +480,9 @@ When the rolling-baseline phase runs, it emits:
Legacy v1 configs without `config_schema_version` still load, but should not
gain v2 identity or metadata fields until they are migrated to
`config_schema_version: 2`.
`config_schema_version: 2`. For v2 configs, `workload_id`, `variant_id`,
and `benchmark_version` are part of the comparison key; benchmark runs
fail if any of these identity fields are missing.
2. The pytest test auto-discovers all configs — no test code needed. CI
picks it up on the next `/test performance` run.
@@ -419,8 +510,8 @@ When the rolling-baseline phase runs, it emits:
## Troubleshooting
**"No baseline for ... Initializing"** — first run for this `(model_id,
gpu_type)`. Run will pass and (if persisting) seed the first record.
**"No baseline for ... Initializing"** — first run for this comparison cohort.
Run will pass and (if persisting) seed the first record.
**Persistent failure right after a torch / kernel / image upgrade** —
genuine regression *or* baseline drift. Compare the failing normalized record
+28 -1
View File
@@ -268,16 +268,31 @@ def load_records_for_model(
model_id: str,
gpu_type: str | None = None,
*,
workload_id: str | None = None,
variant_id: str | None = None,
benchmark_version: str | None = None,
recipe_fingerprint: str | None = None,
hardware_profile_id: str | None = None,
software_profile_id: str | None = None,
last_n: int | None = None,
successful_only: bool = True,
baseline_eligible_only: bool = False,
) -> list[dict[str, Any]]:
"""Return records for a specific *model_id*, optionally filtered by GPU.
"""Return records for a specific *model_id*, optionally filtered by cohort.
Args:
local_dir: Root directory previously populated by :func:`sync_from_hf`.
model_id: Matches the ``model_id`` field inside each JSON record.
gpu_type: When set, only records whose ``gpu_type`` matches are returned.
workload_id: When set, only records from the same workload are returned.
variant_id: When set, only records from the same workload variant are returned.
benchmark_version: When set, only records from the same benchmark version are returned.
recipe_fingerprint: When set, only records from the same benchmark
recipe are returned.
hardware_profile_id: When set, only records from the same hardware
cohort are returned.
software_profile_id: When set, only records from the same software
cohort are returned.
last_n: When set, return only the most recent *n* records (after all
other filters). Useful for sliding-window baseline calculations.
successful_only: Passed through to :func:`load_records`.
@@ -299,6 +314,18 @@ def load_records_for_model(
if gpu_type is not None:
records = [r for r in records if r.get("gpu_type") == gpu_type]
identity_filters = {
"workload_id": workload_id,
"variant_id": variant_id,
"benchmark_version": benchmark_version,
"recipe_fingerprint": recipe_fingerprint,
"hardware_profile_id": hardware_profile_id,
"software_profile_id": software_profile_id,
}
for key, expected in identity_filters.items():
if expected is not None:
records = [r for r in records if str(r.get(key)) == str(expected)]
if last_n is not None:
records = records[-last_n:]
+67 -11
View File
@@ -17,6 +17,15 @@ from fastvideo.performance.hf_store import is_baseline_eligible_record, safe_flo
from fastvideo.performance.metric_policy import regression_delta, resolve_metric_policies
Record = dict[str, Any]
CohortKey = tuple[str, str, str, str, str, str, str, str]
COMPARISON_COHORT_KEYS = (
"workload_id",
"variant_id",
"benchmark_version",
"recipe_fingerprint",
"hardware_profile_id",
"software_profile_id",
)
def parse_timestamp(value: Any) -> datetime | None:
@@ -77,15 +86,56 @@ def record_metadata(record: Record) -> Record:
}
def group_by_model_gpu(records: list[Record]) -> dict[tuple[str, str], list[Record]]:
groups: dict[tuple[str, str], list[Record]] = defaultdict(list)
def _cohort_metadata_value(value: Any) -> Any:
if value is None:
return ""
if isinstance(value, str) and not value.strip():
return ""
return value
def _cohort_key_value(value: Any) -> str:
return str(_cohort_metadata_value(value))
def record_comparison_metadata(record: Record) -> Record:
return {key: _cohort_metadata_value(record.get(key)) for key in COMPARISON_COHORT_KEYS}
def comparison_cohort_key(record: Record) -> CohortKey:
return (
str(record.get("model_id") or "unknown"),
str(record.get("gpu_type") or "unknown"),
_cohort_key_value(record.get("workload_id")),
_cohort_key_value(record.get("variant_id")),
_cohort_key_value(record.get("benchmark_version")),
_cohort_key_value(record.get("recipe_fingerprint")),
_cohort_key_value(record.get("hardware_profile_id")),
_cohort_key_value(record.get("software_profile_id")),
)
def group_by_comparison_cohort(records: list[Record]) -> dict[CohortKey, list[Record]]:
groups: dict[CohortKey, list[Record]] = defaultdict(list)
for record in records:
model_id = str(record.get("model_id") or "unknown")
gpu_type = str(record.get("gpu_type") or "unknown")
groups[(model_id, gpu_type)].append(record)
groups[comparison_cohort_key(record)].append(record)
return {key: sorted(value, key=record_sort_key) for key, value in groups.items()}
def comparison_sort_key(record: Record) -> CohortKey:
return comparison_cohort_key(record)
def latest_row_sort_key(row: Record) -> tuple[Any, ...]:
return (row["status"] != "fail", *comparison_sort_key(row))
def group_identity(key: CohortKey) -> tuple[str, str]:
model_id = key[0]
gpu_type = key[1]
return model_id, gpu_type
def baseline_value(records: list[Record], metric_key: str) -> float | None:
values = [safe_float(record.get(metric_key)) for record in records]
values = [value for value in values if value is not None]
@@ -99,7 +149,8 @@ def build_latest_summary(records: list[Record],
baseline_window: int = 5,
run_source: str | None = None) -> list[Record]:
rows: list[Record] = []
for (model_id, gpu_type), group in group_by_model_gpu(records).items():
for key, group in group_by_comparison_cohort(records).items():
model_id, gpu_type = group_identity(key)
latest_candidates = group
if run_source:
latest_candidates = [record for record in group if record_run_source(record) == run_source]
@@ -107,9 +158,10 @@ def build_latest_summary(records: list[Record],
continue
latest = latest_candidates[-1]
latest_index = next(index for index, record in enumerate(group) if record is latest)
baseline_pool = [
record for record in group
if record is not latest and record.get("success", True) and is_baseline_eligible_record(record)
record for record in group[:latest_index]
if record.get("success", True) and is_baseline_eligible_record(record)
]
baseline_records = baseline_pool[-baseline_window:]
metric_policies = resolve_metric_policies(latest.get("regression_thresholds"))
@@ -156,6 +208,7 @@ def build_latest_summary(records: list[Record],
"timestamp": latest.get("timestamp"),
"commit_sha": latest.get("commit_sha"),
**record_metadata(latest),
**record_comparison_metadata(latest),
"success": success,
"baseline_n": len(baseline_records),
"worst_regression_pct": worst_regression,
@@ -166,12 +219,13 @@ def build_latest_summary(records: list[Record],
"metrics": metrics,
})
return sorted(rows, key=lambda row: (row["status"] != "fail", row["model_id"], row["gpu_type"]))
return sorted(rows, key=latest_row_sort_key)
def build_trends(records: list[Record]) -> list[Record]:
trends: list[Record] = []
for (model_id, gpu_type), group in group_by_model_gpu(records).items():
for key, group in group_by_comparison_cohort(records).items():
model_id, gpu_type = group_identity(key)
points = []
for record in group:
metric_policies = resolve_metric_policies(record.get("regression_thresholds"))
@@ -179,6 +233,7 @@ def build_trends(records: list[Record]) -> list[Record]:
"timestamp": record.get("timestamp"),
"commit_sha": record.get("commit_sha"),
**record_metadata(record),
**record_comparison_metadata(record),
"success": bool(record.get("success", True)),
"metrics": {
policy.key: safe_float(record.get(policy.key))
@@ -189,6 +244,7 @@ def build_trends(records: list[Record]) -> list[Record]:
trends.append({
"model_id": model_id,
"gpu_type": gpu_type,
**record_comparison_metadata(group[-1]),
"points": points,
})
return sorted(trends, key=lambda trend: (trend["model_id"], trend["gpu_type"]))
return sorted(trends, key=comparison_sort_key)
+124 -22
View File
@@ -5,7 +5,8 @@ This script:
1) reads current benchmark results from fastvideo/tests/performance/results,
2) syncs the canonical baseline from the configured HF dataset repo,
3) compares each current record against the median of up to 5 prior
baseline-eligible successful records (filtered by gpu_type),
baseline-eligible successful records in the same workload/variant/version,
GPU, recipe, hardware, and software cohort,
4) writes normalized records back to the HF dataset repo according to
PERF_UPLOAD_POLICY,
5) exits non-zero if any gated metric exceeds both its percent and absolute
@@ -64,6 +65,30 @@ PERF_REPORTS_DIR = os.environ.get("PERF_REPORTS_DIR", "/root/data/perf_reports")
UPLOAD_POLICY = os.environ.get("PERF_UPLOAD_POLICY", "never").strip().lower()
VALID_UPLOAD_POLICIES = {"never", "pass", "always"}
VALID_RUN_SOURCES = {"pr", "local", "scheduled_main", "unknown"}
IDENTITY_KEYS = (
"result_schema_version",
"workload_id",
"variant_id",
"benchmark_version",
"recipe",
"recipe_fingerprint",
"hardware_profile",
"hardware_profile_id",
"software_profile",
"software_profile_id",
"environment_metadata",
"environment_fingerprint",
"quality_metadata",
"variant_metadata",
)
COMPARISON_IDENTITY_KEYS = (
"workload_id",
"variant_id",
"benchmark_version",
"recipe_fingerprint",
"hardware_profile_id",
"software_profile_id",
)
def _should_persist_tracking() -> bool:
@@ -121,21 +146,72 @@ def _result_failed_static_thresholds() -> bool:
def _record_metadata(run_source: str, result: dict[str, Any]) -> dict[str, Any]:
raw_run_source = str(result.get("run_source") or "").strip().lower()
if raw_run_source in VALID_RUN_SOURCES:
run_source = raw_run_source
pr_number = result.get("pr_number") or os.environ.get("BUILDKITE_PULL_REQUEST", "")
if not _truthy_pr_number(str(pr_number)):
pr_number = ""
return {
"run_source": run_source,
"baseline_eligible": False,
"branch": os.environ.get("BUILDKITE_BRANCH", ""),
"branch": result.get("branch") or os.environ.get("BUILDKITE_BRANCH", ""),
"pr_number": pr_number,
"test_scope": os.environ.get("TEST_SCOPE", ""),
"build_url": os.environ.get("BUILDKITE_BUILD_URL", ""),
"build_id": os.environ.get("BUILDKITE_BUILD_ID", ""),
"job_id": os.environ.get("BUILDKITE_JOB_ID", ""),
"test_scope": result.get("test_scope") or os.environ.get("TEST_SCOPE", ""),
"build_url": result.get("build_url") or os.environ.get("BUILDKITE_BUILD_URL", ""),
"build_id": result.get("build_id") or os.environ.get("BUILDKITE_BUILD_ID", ""),
"job_id": result.get("job_id") or os.environ.get("BUILDKITE_JOB_ID", ""),
}
def _identity_metadata(result: dict[str, Any]) -> dict[str, Any]:
metadata = {
key: result[key]
for key in IDENTITY_KEYS
if key in result and result[key] is not None
}
recipe = result.get("recipe")
benchmark = recipe.get("benchmark") if isinstance(recipe, dict) else None
if isinstance(benchmark, dict):
for key in ("workload_id", "variant_id", "benchmark_version"):
if key not in metadata and benchmark.get(key) is not None:
metadata[key] = benchmark[key]
return metadata
def _comparison_identity_filters(record: dict[str, Any]) -> dict[str, str]:
missing = [
key for key in COMPARISON_IDENTITY_KEYS
if key not in record or record[key] is None or (isinstance(record[key], str) and not record[key].strip())
]
if missing:
raise ValueError("Performance record missing required comparison identity fields: " + ", ".join(missing))
return {
key: str(record[key])
for key in COMPARISON_IDENTITY_KEYS
}
def _comparison_identity_filters_or_none(record: dict[str, Any]) -> dict[str, str] | None:
try:
return _comparison_identity_filters(record)
except ValueError as exc:
print("Skipping rolling baseline comparison for "
f"{record.get('model_id', 'unknown')} on "
f"{record.get('gpu_type', 'unknown')}: {exc}. "
"The normalized artifact will still be written, but this record "
"will not be baseline eligible.")
return None
def _format_identity_filters(filters: dict[str, str]) -> str:
if not filters:
return ""
compact = ", ".join(f"{key}={value}" for key, value in filters.items())
return f" ({compact})"
def _load_current_results() -> list[dict[str, Any]]:
pattern = os.path.join(RESULTS_DIR, "perf_*.json")
records: list[dict[str, Any]] = []
@@ -169,7 +245,7 @@ def normalize_performance_result(result: dict[str, Any]) -> dict[str, Any]:
vae_decode_time = safe_float(result.get("vae_decode_time_s"))
metric_policies = resolve_metric_policies(result.get("regression_thresholds"))
return {
record = {
"model_id": model_id,
"timestamp": timestamp,
"commit_sha": commit_sha,
@@ -184,6 +260,8 @@ def normalize_performance_result(result: dict[str, Any]) -> dict[str, Any]:
"success": True,
**_record_metadata(_detect_run_source(), result),
}
record.update(_identity_metadata(result))
return record
def _normalize_record(result: dict[str, Any]) -> dict[str, Any]:
@@ -417,29 +495,53 @@ def main() -> int:
for raw in current_results:
record = _normalize_record(raw)
metric_policies = resolve_metric_policies(record.get("regression_thresholds"))
identity_filters = _comparison_identity_filters_or_none(record)
baseline_records = load_records_for_model(
TRACKING_ROOT,
record["model_id"],
record["gpu_type"],
last_n=5,
successful_only=True,
baseline_eligible_only=True,
)
if not baseline_records:
print(f"No baseline for {record['model_id']} on "
f"{record['gpu_type']}. Initializing...")
baseline_records: list[dict[str, Any]] = []
if identity_filters is None:
# Records without the full v2 identity block skip rolling-baseline
# comparison entirely: only the static thresholds gate them and
# they never become baseline eligible.
record["baseline_status"] = "skipped_missing_identity"
failures: list[str] = []
record["success"] = True
else:
failures = _check_regressions(record, baseline_records, metric_policies)
baseline_records = load_records_for_model(
TRACKING_ROOT,
record["model_id"],
record["gpu_type"],
**identity_filters,
last_n=5,
successful_only=True,
baseline_eligible_only=True,
)
if not baseline_records:
# A brand-new cohort has NO regression gating until history
# accumulates — make that loud and machine-readable instead of
# an indistinguishable pass, so a cohort shift (intended or
# accidental, e.g. an identity-field change) never silently
# blinds the comparison.
record["baseline_status"] = "initialized_new_cohort"
print("=" * 72)
print(f"WARNING: NO BASELINE — initializing a NEW cohort for "
f"{record['model_id']} on {record['gpu_type']}"
f"{_format_identity_filters(identity_filters)}")
print("Regression gating is INACTIVE for this cohort until "
"baseline history accumulates. If this cohort shift is "
"unexpected, check the identity fields above.")
print("=" * 72)
failures = []
else:
record["baseline_status"] = "compared"
failures = _check_regressions(record, baseline_records, metric_policies)
if static_threshold_failed:
failures.append(f"{record['model_id']} fixed-threshold phase failed "
f"(PERF_PYTEST_RC={os.environ.get('PERF_PYTEST_RC')})")
record["success"] = not failures
record["baseline_eligible"] = _is_baseline_eligible(record["run_source"], record["success"])
record["baseline_eligible"] = (
identity_filters is not None and _is_baseline_eligible(record["run_source"], record["success"])
)
all_failures.extend(failures)
_write_normalized_artifact(record)
+81 -12
View File
@@ -29,15 +29,76 @@ METRICS = (
"dit_time_s",
"vae_decode_time_s",
)
COMPARISON_COHORT_KEYS = (
"workload_id",
"variant_id",
"benchmark_version",
"recipe_fingerprint",
"hardware_profile_id",
"software_profile_id",
)
GROUP_KEYS = ("model_id", "gpu_type", *COMPARISON_COHORT_KEYS)
def _cohort_value(value: object) -> str:
if value is None:
return ""
try:
if pd.isna(value):
return ""
except (TypeError, ValueError):
pass
return str(value)
def _display_value(value: object) -> str:
return _cohort_value(value) or "legacy"
def _short_value(value: object) -> str:
text = _display_value(value)
if text == "legacy" or len(text) <= 14:
return text
return text[:12]
def _cohort_title(group_key: tuple[object, ...]) -> str:
workload = _display_value(group_key[2])
variant = _display_value(group_key[3])
version = _display_value(group_key[4])
version_label = version if version == "legacy" else f"v{version}"
return f"{workload} / {variant} / {version_label}"
def _cohort_detail(group_key: tuple[object, ...]) -> str:
return " | ".join((
f"recipe {_short_value(group_key[5])}",
_short_value(group_key[6]),
_short_value(group_key[7]),
))
def _group_record(group_key: tuple[object, ...]) -> dict[str, str]:
return {
key: _cohort_value(value)
for key, value in zip(GROUP_KEYS, group_key)
}
def _dashboard_frame(df: pd.DataFrame) -> pd.DataFrame:
dashboard_df = df.copy()
for key in GROUP_KEYS:
if key not in dashboard_df.columns:
dashboard_df[key] = ""
dashboard_df[key] = dashboard_df[key].map(_cohort_value)
return dashboard_df
# -----------------------------
# 1. Grouping
# -----------------------------
def group_data(df: pd.DataFrame):
# Group only by model+GPU so each group produces a time-series line.
# config_id (commit SHA) is carried as a column for hover/color use.
keys = ["model_id", "gpu_type"]
return df.groupby(keys, dropna=False)
# Group by the same comparison cohort used by baseline gating.
return _dashboard_frame(df).groupby(list(GROUP_KEYS), dropna=False)
# -----------------------------
# 2. Plot builder
@@ -46,15 +107,20 @@ def build_plots(df: pd.DataFrame) -> tuple[list, list[dict[str, object]]]:
figs = []
skipped_metrics: list[dict[str, object]] = []
for (model_id, gpu_type), g in group_data(df):
for group_key, g in group_data(df):
model_id, gpu_type = group_key[:2]
group_record = _group_record(group_key)
cohort_title = _cohort_title(group_key)
cohort_detail = _cohort_detail(group_key)
g = g.sort_values("timestamp")
# One chart per metric so the y-axes aren't on wildly different scales
for metric in METRICS:
if metric not in g.columns:
skipped_metrics.append({
"model_id": model_id,
"gpu_type": gpu_type,
**group_record,
"cohort": cohort_title,
"cohort_detail": cohort_detail,
"metric": metric,
"reason": "column missing from loaded records",
"records": len(g),
@@ -65,8 +131,9 @@ def build_plots(df: pd.DataFrame) -> tuple[list, list[dict[str, object]]]:
non_null = int(g[metric].notna().sum())
if non_null == 0:
skipped_metrics.append({
"model_id": model_id,
"gpu_type": gpu_type,
**group_record,
"cohort": cohort_title,
"cohort_detail": cohort_detail,
"metric": metric,
"reason": "no non-null values in loaded records",
"records": len(g),
@@ -79,8 +146,8 @@ def build_plots(df: pd.DataFrame) -> tuple[list, list[dict[str, object]]]:
x="timestamp",
y=metric,
markers=True,
hover_data=["config_id", "commit_sha"],
title=f"{model_id} | {gpu_type} | {metric}",
hover_data=["config_id", "commit_sha", *COMPARISON_COHORT_KEYS],
title=f"{model_id} | {gpu_type} | {cohort_title} | {cohort_detail} | {metric}",
labels={"timestamp": "Time", metric: metric},
)
figs.append(fig)
@@ -95,7 +162,7 @@ def render_skipped_metrics(skipped_metrics: list[dict[str, object]]) -> str:
rows = [
"<h3>Skipped Metric Plots</h3>",
"<table>",
("<thead><tr><th>Model</th><th>GPU</th><th>Metric</th>"
("<thead><tr><th>Model</th><th>GPU</th><th>Cohort</th><th>Metric</th>"
"<th>Records</th><th>Non-null</th><th>Reason</th></tr></thead>"),
"<tbody>",
]
@@ -104,6 +171,7 @@ def render_skipped_metrics(skipped_metrics: list[dict[str, object]]) -> str:
"<tr>"
f"<td>{escape(str(item['model_id']))}</td>"
f"<td>{escape(str(item['gpu_type']))}</td>"
f"<td>{escape(str(item['cohort']))}<br><code>{escape(str(item['cohort_detail']))}</code></td>"
f"<td>{escape(str(item['metric']))}</td>"
f"<td>{item['records']}</td>"
f"<td>{item['non_null']}</td>"
@@ -165,6 +233,7 @@ def main() -> None:
print("Skipped metric plots:")
for item in skipped_metrics:
print(f" - {item['model_id']} | {item['gpu_type']} | "
f"{item['cohort']} | {item['cohort_detail']} | "
f"{item['metric']}: {item['reason']} "
f"({item['non_null']}/{item['records']} non-null)")
print(f"Generated {len(figs)} metric plot(s)")
+435
View File
@@ -0,0 +1,435 @@
# SPDX-License-Identifier: Apache-2.0
"""Comparable identity helpers for performance benchmark records."""
from __future__ import annotations
import dataclasses
import enum
import hashlib
import importlib.metadata
import json
import os
import platform
import re
from collections.abc import Mapping, Sequence
from pathlib import Path
from typing import Any
import torch
RECIPE_SCHEMA_VERSION = 1
PROFILE_ID_LENGTH = 16
_PATH_EXCLUDED_GENERATION_KEYS = {"output_path", "output_video_name"}
_PROMPT_KEYS = {"prompt", "negative_prompt", "neg_prompt"}
REQUIRED_BENCHMARK_IDENTITY_KEYS = (
"benchmark_id",
"workload_id",
"variant_id",
"benchmark_version",
)
_PROFILE_ENV_VARS = (
"CUDA_VISIBLE_DEVICES",
"FASTVIDEO_ATTENTION_BACKEND",
"IMAGE_VERSION",
"UV_TORCH_BACKEND",
)
_PACKAGE_DISTRIBUTIONS = {
"fastvideo_kernel": ("fastvideo-kernel", "fastvideo_kernel"),
"flash_attn": ("flash-attn", "flash_attn"),
"flash_attn_4": ("flash-attn-4",),
"flash_attention_fp4": ("flash-attention-fp4",),
"flashinfer": ("flashinfer-python", "flashinfer"),
"nvidia_cutlass_dsl": ("nvidia-cutlass-dsl", "cutlass-dsl"),
"sageattention": ("sageattention",),
"sageattn3": ("sageattn3",),
"triton": ("triton",),
"xformers": ("xformers",),
}
def canonical_json(payload: Any) -> str:
"""Return deterministic, compact JSON for a recipe/profile payload."""
return json.dumps(
_canonicalize(payload),
sort_keys=True,
separators=(",", ":"),
ensure_ascii=True,
)
def sha256_hexdigest(payload: str | bytes) -> str:
if isinstance(payload, str):
payload = payload.encode("utf-8")
return hashlib.sha256(payload).hexdigest()
def recipe_fingerprint(recipe: Mapping[str, Any]) -> str:
# Runtime-resolved values are audit metadata, not identity: hashing the
# resolved HF snapshot revision would churn the cohort (and via the
# no-baseline init path, silently reset the rolling baseline) on ANY
# upstream repo commit, including model-card edits with unchanged
# weights. The fingerprint is pinned to declared inputs only; the
# resolved values stay in the stored recipe for auditability.
pruned = {
key: (dict(value) if isinstance(value, Mapping) else value)
for key, value in recipe.items()
}
model = pruned.get("model")
if isinstance(model, dict):
model.pop("resolved_revision", None)
attention = pruned.get("attention")
if isinstance(attention, dict):
# Same reasoning as resolved_revision: with requested_backend="auto",
# a silent runtime fallback (e.g. a broken flash-attn wheel) must NOT
# open a fresh ungated cohort and seed degraded numbers as the new
# baseline — it should fail the regression gate in the old cohort.
# Intentional backend changes are declared via requested_backend,
# which stays in the hash.
attention.pop("resolved_backend", None)
return sha256_hexdigest(canonical_json(pruned))
def profile_id(prefix: str, profile: Mapping[str, Any]) -> str:
return f"{prefix}-{sha256_hexdigest(canonical_json(profile))[:PROFILE_ID_LENGTH]}"
def hardware_profile_id(profile: Mapping[str, Any]) -> str:
return profile_id("hw", profile)
def software_profile_id(profile: Mapping[str, Any]) -> str:
return profile_id("sw", profile)
def environment_fingerprint(metadata: Mapping[str, Any]) -> str:
return profile_id("env", metadata)
def benchmark_identity_from_config(cfg: Mapping[str, Any]) -> dict[str, Any]:
missing = [
key for key in REQUIRED_BENCHMARK_IDENTITY_KEYS
if _none_if_empty(cfg.get(key)) is None
]
if missing:
raise ValueError("Benchmark config missing required identity fields: " + ", ".join(missing))
return {
key: cfg[key]
for key in REQUIRED_BENCHMARK_IDENTITY_KEYS
}
def build_recipe_from_benchmark_config(
cfg: Mapping[str, Any],
*,
attention_backend: str | None = None,
resolved_attention_backend: Any | None = None,
resolved_model_revision: Any | None = None,
measured_prompts: Sequence[Any] | None = None,
) -> dict[str, Any]:
"""Build a deterministic recipe document from a benchmark config.
The recipe captures fields that describe what benchmark workload was run. It
intentionally excludes timings, timestamps, commit metadata, and output paths.
"""
model = dict(cfg.get("model") or {})
init_kwargs = dict(cfg.get("init_kwargs") or {})
generation_kwargs = dict(cfg.get("generation_kwargs") or {})
benchmark_identity = benchmark_identity_from_config(cfg)
prompts = list(measured_prompts) if measured_prompts is not None else list(
cfg.get("test_prompts") or ["A cinematic video."])
if attention_backend is None:
attention_backend = os.environ.get("FASTVIDEO_ATTENTION_BACKEND")
negative_prompt = generation_kwargs.pop("negative_prompt", generation_kwargs.pop("neg_prompt", None))
generation_recipe = {
key: value
for key, value in generation_kwargs.items()
if key not in _PATH_EXCLUDED_GENERATION_KEYS and key not in _PROMPT_KEYS
}
return {
"recipe_schema_version": RECIPE_SCHEMA_VERSION,
"benchmark": benchmark_identity,
"model": {
"model_path": normalize_model_path(model.get("model_path")),
"revision": _none_if_empty(model.get("revision") or cfg.get("revision")),
"resolved_revision": _none_if_empty(resolved_model_revision),
},
"init_kwargs": init_kwargs,
"generation_kwargs": generation_recipe,
"inputs": {
"prompt_count": len(prompts),
"prompt_sha256": [_optional_digest(prompt) for prompt in prompts],
"negative_prompt_sha256": _optional_digest(negative_prompt),
},
"attention": {
"requested_backend": _none_if_empty(attention_backend) or "auto",
"resolved_backend": _none_if_empty(resolved_attention_backend),
},
}
def normalize_model_path(value: Any) -> str | None:
if value is None:
return None
text = os.fspath(value) if isinstance(value, os.PathLike) else str(value)
text = text.strip()
if not text:
return None
if _looks_like_local_path(text):
return Path(text).expanduser().as_posix()
return re.sub(r"/+", "/", text)
def resolved_revision_from_model_path(value: Any) -> str | None:
normalized = normalize_model_path(value)
if normalized is None:
return None
parts = Path(normalized).parts
for index, part in enumerate(parts[:-1]):
if part == "snapshots":
revision = parts[index + 1]
if re.fullmatch(r"[0-9a-f]{40}", revision):
return revision
return None
def hardware_profile(
*,
num_gpus: int | None = None,
gpu_devices: Sequence[Mapping[str, Any]] | None = None,
device_type: str | None = None,
interconnect: str | None = None,
) -> dict[str, Any]:
"""Return normalized hardware cohort metadata.
``gpu_devices`` is an injection point for tests and non-CUDA callers.
"""
if gpu_devices is not None:
devices = [_normalize_gpu_device(device) for device in gpu_devices]
return {
"device_type": device_type or "cuda",
"gpu_count": num_gpus if num_gpus is not None else len(devices),
"gpus": devices,
"interconnect": interconnect or "unknown",
}
if not torch.cuda.is_available():
return {
"device_type": device_type or "cpu",
"gpu_count": 0,
"gpus": [],
"interconnect": interconnect or "none",
}
gpu_count = int(num_gpus if num_gpus is not None else torch.cuda.device_count())
devices = []
cuda_device_count = torch.cuda.device_count()
for device_id in range(gpu_count):
if device_id >= cuda_device_count:
devices.append(_normalize_gpu_device({}))
continue
props = torch.cuda.get_device_properties(device_id)
capability = torch.cuda.get_device_capability(device_id)
devices.append(
_normalize_gpu_device({
"name": props.name,
"memory_bytes": props.total_memory,
"compute_capability": f"{capability[0]}.{capability[1]}",
}))
return {
"device_type": device_type or "cuda",
"gpu_count": gpu_count,
"gpus": devices,
"interconnect": interconnect or _detect_cuda_interconnect(gpu_count),
}
def software_profile(
*,
python_version: str | None = None,
torch_version: str | None = None,
cuda_version: str | None = None,
package_versions: Mapping[str, str | None] | None = None,
) -> dict[str, Any]:
"""Return normalized software cohort metadata.
Python, PyTorch, and CUDA use major/minor cohorts. Attention and kernel
packages keep exact versions because patch releases can change performance.
"""
versions = dict(package_versions) if package_versions is not None else _installed_package_versions()
return {
"python": _major_minor(python_version or platform.python_version()),
"pytorch": _major_minor(torch_version or torch.__version__),
"cuda": _major_minor(cuda_version or torch.version.cuda),
"packages": {
name: str(version)
for name, version in sorted(versions.items())
if version
},
}
def environment_metadata(
*,
env: Mapping[str, str] | None = None,
package_versions: Mapping[str, str | None] | None = None,
hardware: Mapping[str, Any] | None = None,
software: Mapping[str, Any] | None = None,
) -> dict[str, Any]:
"""Return audit metadata kept separate from comparable identity keys."""
source_env = env if env is not None else os.environ
full_package_versions = dict(package_versions) if package_versions is not None else _installed_package_versions()
return {
"python": {
"version": platform.python_version(),
"implementation": platform.python_implementation(),
},
"platform": {
"system": platform.system(),
"machine": platform.machine(),
"release": platform.release(),
},
"torch": {
"version": torch.__version__,
"cuda": torch.version.cuda,
"cudnn": torch.backends.cudnn.version(),
},
"env": {
key: source_env.get(key)
for key in _PROFILE_ENV_VARS
if source_env.get(key) is not None
},
"packages": {
key: value
for key, value in sorted(full_package_versions.items())
if value
},
"hardware_profile": dict(hardware) if hardware is not None else hardware_profile(),
"software_profile": dict(software) if software is not None else software_profile(
package_versions=full_package_versions),
}
def _canonicalize(value: Any) -> Any:
if dataclasses.is_dataclass(value) and not isinstance(value, type):
return _canonicalize(dataclasses.asdict(value))
if isinstance(value, Mapping):
return {
str(key): _canonicalize(value[key])
for key in sorted(value, key=lambda item: str(item))
}
if isinstance(value, (set, frozenset)):
return [_canonicalize(item) for item in sorted(value, key=lambda item: str(item))]
if isinstance(value, tuple):
return [_canonicalize(item) for item in value]
if isinstance(value, list):
return [_canonicalize(item) for item in value]
if isinstance(value, enum.Enum):
return value.name
if isinstance(value, Path):
return value.expanduser().as_posix()
if isinstance(value, os.PathLike):
return os.fspath(value)
if isinstance(value, type):
return f"{value.__module__}.{value.__qualname__}"
if isinstance(value, torch.dtype):
return str(value).removeprefix("torch.")
return value
def _none_if_empty(value: Any) -> Any | None:
if value is None:
return None
if isinstance(value, str) and not value.strip():
return None
return value
def _optional_digest(value: Any) -> str | None:
if value is None:
return None
return sha256_hexdigest(str(value))
def _looks_like_local_path(value: str) -> bool:
if value.startswith(("/", "./", "../", "~")):
return True
looks_like_hf_repo = re.match(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$", value)
has_separator = "/" in value or "\\" in value or os.sep in value
return has_separator and looks_like_hf_repo is None
def _normalize_gpu_device(device: Mapping[str, Any]) -> dict[str, Any]:
memory_gb = device.get("memory_gb")
if memory_gb is None and device.get("memory_bytes") is not None:
memory_gb = round(float(device["memory_bytes"]) / (1024**3))
return {
"name": _none_if_empty(device.get("name")) or "unknown",
"memory_gb": memory_gb,
"compute_capability": _none_if_empty(device.get("compute_capability")),
}
def _detect_cuda_interconnect(gpu_count: int) -> str:
if gpu_count <= 1:
return "single_gpu"
try:
from fastvideo.platforms import current_platform
if current_platform.is_cuda() and current_platform.is_full_nvlink(list(range(gpu_count))):
return "full_nvlink"
return "none_or_partial"
except Exception:
return "unknown"
def _installed_package_versions() -> dict[str, str | None]:
return {
name: _distribution_version(*distribution_names)
for name, distribution_names in _PACKAGE_DISTRIBUTIONS.items()
}
def _distribution_version(*names: str) -> str | None:
for name in names:
try:
return importlib.metadata.version(name)
except importlib.metadata.PackageNotFoundError:
continue
return None
def _major_minor(version: str | None) -> str | None:
if not version:
return None
match = re.search(r"(\d+)\.(\d+)", version)
if match is None:
return version
return f"{match.group(1)}.{match.group(2)}"
__all__ = [
"RECIPE_SCHEMA_VERSION",
"REQUIRED_BENCHMARK_IDENTITY_KEYS",
"benchmark_identity_from_config",
"canonical_json",
"environment_fingerprint",
"environment_metadata",
"hardware_profile",
"hardware_profile_id",
"normalize_model_path",
"profile_id",
"recipe_fingerprint",
"resolved_revision_from_model_path",
"build_recipe_from_benchmark_config",
"sha256_hexdigest",
"software_profile",
"software_profile_id",
]
@@ -14,9 +14,9 @@ def _v2_config():
return {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"config_schema_version": 2,
"workload_id": "wan-t2v-1.3b",
"variant_id": "canonical",
"benchmark_version": 1,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
}
@@ -41,9 +41,9 @@ def test_v2_benchmark_config_identity_validates_and_is_preserved():
assert _is_v2_config(cfg) is True
assert _config_identity_metadata(cfg) == {
"config_schema_version": 2,
"workload_id": "wan-t2v-1.3b",
"variant_id": "canonical",
"benchmark_version": 1,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"quality_metadata": {"some": "data"},
}
@@ -91,9 +91,9 @@ def test_v2_benchmark_config_rejects_invalid_benchmark_version_values(value):
def test_partial_v2_identity_requires_schema_version():
cfg = {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"workload_id": "wan-t2v-1.3b",
"variant_id": "canonical",
"benchmark_version": 1,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
}
expected = "wan.json: v2 benchmark identity fields require config_schema_version=2"
@@ -105,9 +105,9 @@ def test_optional_v2_metadata_fields_must_be_objects():
cfg = {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"config_schema_version": 2,
"workload_id": "wan-t2v-1.3b",
"variant_id": "canonical",
"benchmark_version": 1,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"quality_metadata": ["not", "an", "object"],
}
@@ -1,5 +1,9 @@
# SPDX-License-Identifier: Apache-2.0
import json
import pytest
from fastvideo.tests.performance import compare_baseline
from fastvideo.performance.metric_policy import resolve_metric_policies
@@ -119,6 +123,196 @@ def test_boolean_regression_threshold_values_are_ignored():
assert latency.threshold_percent == 0.08
assert latency.threshold_absolute == 0.5
assert latency.gated is False
def test_normalized_record_preserves_identity_metadata(monkeypatch):
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
raw = _raw_result()
raw.update({
"result_schema_version": 2,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"recipe": {
"recipe_schema_version": 1,
},
"recipe_fingerprint": "recipe-1",
"hardware_profile": {
"gpu_count": 1,
},
"hardware_profile_id": "hw-1",
"software_profile": {
"python": "3.12",
},
"software_profile_id": "sw-1",
"environment_metadata": {
"env": {
"IMAGE_VERSION": "latest",
},
},
"environment_fingerprint": "env-1",
"quality_metadata": {
"quality_status": "canonical",
},
})
record = compare_baseline.normalize_performance_result(raw)
assert record["result_schema_version"] == 2
assert record["workload_id"] == "wan-t2v"
assert record["variant_id"] == "1.3b-sp2"
assert record["benchmark_version"] == 2
assert record["recipe"] == {"recipe_schema_version": 1}
assert record["recipe_fingerprint"] == "recipe-1"
assert record["hardware_profile"] == {"gpu_count": 1}
assert record["hardware_profile_id"] == "hw-1"
assert record["software_profile"] == {"python": "3.12"}
assert record["software_profile_id"] == "sw-1"
assert record["environment_metadata"] == {"env": {"IMAGE_VERSION": "latest"}}
assert record["environment_fingerprint"] == "env-1"
assert record["quality_metadata"] == {"quality_status": "canonical"}
def test_normalized_record_prefers_raw_v2_provenance(monkeypatch):
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
monkeypatch.setenv("BUILDKITE_BRANCH", "feature/from-env")
monkeypatch.setenv("TEST_SCOPE", "direct")
monkeypatch.setenv("BUILDKITE_BUILD_URL", "https://buildkite.example/env")
monkeypatch.setenv("BUILDKITE_BUILD_ID", "env-build")
monkeypatch.setenv("BUILDKITE_JOB_ID", "env-job")
raw = _raw_result()
raw.update({
"result_schema_version": 2,
"run_source": "scheduled_main",
"branch": "main",
"test_scope": "full",
"build_url": "https://buildkite.example/raw",
"build_id": "raw-build",
"job_id": "raw-job",
"pr_number": "false",
})
record = compare_baseline.normalize_performance_result(raw)
assert record["result_schema_version"] == 2
assert record["run_source"] == "scheduled_main"
assert record["branch"] == "main"
assert record["test_scope"] == "full"
assert record["build_url"] == "https://buildkite.example/raw"
assert record["build_id"] == "raw-build"
assert record["job_id"] == "raw-job"
assert record["pr_number"] == ""
def test_v1_normalized_record_has_no_result_schema_version(monkeypatch):
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
record = compare_baseline.normalize_performance_result(_raw_result())
assert "result_schema_version" not in record
def test_main_writes_v1_current_artifact_without_comparison_identity(monkeypatch, tmp_path, capsys):
results_dir = tmp_path / "results"
reports_dir = tmp_path / "reports"
tracking_root = tmp_path / "tracking"
results_dir.mkdir()
(results_dir / "perf_legacy.json").write_text(json.dumps(_raw_result()), encoding="utf-8")
uploaded_records = []
monkeypatch.setenv("PERF_RUN_SOURCE", "scheduled_main")
monkeypatch.delenv("PERF_PYTEST_RC", raising=False)
monkeypatch.setattr(compare_baseline, "RESULTS_DIR", str(results_dir))
monkeypatch.setattr(compare_baseline, "PERF_REPORTS_DIR", str(reports_dir))
monkeypatch.setattr(compare_baseline, "TRACKING_ROOT", str(tracking_root))
monkeypatch.setattr(compare_baseline, "UPLOAD_POLICY", "always")
monkeypatch.setattr(compare_baseline, "sync_from_hf", lambda local_dir, strict=False: local_dir)
def fail_load_records_for_model(*_args, **_kwargs):
raise AssertionError("legacy current records should skip v2 baseline lookup")
def fake_upload_record(_path, record, *, strict=False):
uploaded_records.append(record.copy())
monkeypatch.setattr(compare_baseline, "load_records_for_model", fail_load_records_for_model)
monkeypatch.setattr(compare_baseline, "upload_record", fake_upload_record)
assert compare_baseline.main() == 0
output = capsys.readouterr().out
normalized_files = list((reports_dir / "results").glob("normalized_perf_*.json"))
assert "Skipping rolling baseline comparison" in output
assert "workload_id" in output
assert len(normalized_files) == 1
normalized = json.loads(normalized_files[0].read_text(encoding="utf-8"))
assert "result_schema_version" not in normalized
assert normalized["success"] is True
assert normalized["baseline_eligible"] is False
assert len(uploaded_records) == 1
assert uploaded_records[0]["baseline_eligible"] is False
def test_normalized_record_reads_identity_labels_from_recipe(monkeypatch):
monkeypatch.setenv("PERF_RUN_SOURCE", "pr")
raw = _raw_result()
raw["recipe"] = {
"benchmark": {
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
},
}
record = compare_baseline.normalize_performance_result(raw)
assert record["workload_id"] == "wan-t2v"
assert record["variant_id"] == "1.3b-sp2"
assert record["benchmark_version"] == 2
def test_comparison_identity_filters_use_full_issue_key():
record = {
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"recipe_fingerprint": "recipe-1",
"hardware_profile_id": "hw-1",
"software_profile_id": "sw-1",
}
assert compare_baseline._comparison_identity_filters(record) == {
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": "2",
"recipe_fingerprint": "recipe-1",
"hardware_profile_id": "hw-1",
"software_profile_id": "sw-1",
}
def test_comparison_identity_filters_require_full_issue_key():
record = {
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 0,
"recipe_fingerprint": "recipe-1",
"hardware_profile_id": "hw-1",
}
with pytest.raises(ValueError, match="software_profile_id"):
compare_baseline._comparison_identity_filters(record)
def test_comparison_identity_filters_keep_zero_version():
record = {
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 0,
"recipe_fingerprint": "recipe-1",
"hardware_profile_id": "hw-1",
"software_profile_id": "sw-1",
}
assert compare_baseline._comparison_identity_filters(record)["benchmark_version"] == "0"
def test_baseline_eligibility_only_for_successful_scheduled_main():
@@ -56,8 +56,34 @@ def _record(model_id, gpu_type, ts, commit, latency, throughput, success=True, *
def test_summary_endpoint_returns_latest_group_status():
app = create_app(FakeStore([
_record("wan", "NVIDIA L40S", "2026-01-01T00:00:00+00:00", "a" * 40, 10.0, 10.0),
_record("wan", "NVIDIA L40S", "2026-01-02T00:00:00+00:00", "b" * 40, 11.0, 9.0),
_record(
"wan",
"NVIDIA L40S",
"2026-01-01T00:00:00+00:00",
"a" * 40,
10.0,
10.0,
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=2,
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
),
_record(
"wan",
"NVIDIA L40S",
"2026-01-02T00:00:00+00:00",
"b" * 40,
11.0,
9.0,
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=2,
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
),
]))
client = TestClient(app)
@@ -71,6 +97,10 @@ def test_summary_endpoint_returns_latest_group_status():
assert body["rows"][0]["metrics"]["latency"]["threshold_exceeded"] is True
assert body["rows"][0]["threshold_exceeded_metrics"] == ["latency", "throughput"]
assert body["rows"][0]["computed_regression_status"] == "fail"
assert body["rows"][0]["workload_id"] == "wan-t2v"
assert body["rows"][0]["variant_id"] == "1.3b-sp2"
assert body["rows"][0]["benchmark_version"] == 2
assert body["rows"][0]["recipe_fingerprint"] == "recipe-a"
def test_summary_status_is_independent_of_days_window():
@@ -124,8 +154,8 @@ def test_dashboard_endpoints_filter_and_return_run_source_metadata():
assert summary["count"] == 1
assert summary["rows"][0]["run_source"] == "pr"
assert summary["rows"][0]["pr_number"] == "123"
assert summary["rows"][0]["baseline_n"] == 1
assert summary["rows"][0]["metrics"]["latency"]["baseline"] == 11.0
assert summary["rows"][0]["baseline_n"] == 0
assert summary["rows"][0]["metrics"]["latency"]["baseline"] is None
assert summary["filters"]["run_source"] == "pr"
assert trends["count"] == 1
assert trends["groups"][0]["points"][0]["run_source"] == "pr"
@@ -0,0 +1,98 @@
# SPDX-License-Identifier: Apache-2.0
import pandas as pd
from fastvideo.tests.performance import dashboard
def _record(recipe_fingerprint="recipe-a", software_profile_id="sw-a", **overrides):
record = {
"model_id": "wan-t2v-1.3b-2gpu",
"gpu_type": "NVIDIA L40S",
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"recipe_fingerprint": recipe_fingerprint,
"hardware_profile_id": "hw-l40s",
"software_profile_id": software_profile_id,
"timestamp": "2026-01-01T00:00:00+00:00",
"commit_sha": "a" * 40,
"config_id": "aaaaaaa",
"latency": 10.0,
"throughput": 4.5,
"memory": 10000.0,
"text_encoder_time_s": 2.0,
"dit_time_s": 8.0,
"vae_decode_time_s": 3.0,
}
record.update(overrides)
return record
def test_group_data_uses_full_comparison_cohort():
df = pd.DataFrame([
_record(recipe_fingerprint="recipe-a", software_profile_id="sw-a"),
_record(recipe_fingerprint="recipe-b", software_profile_id="sw-b"),
])
groups = list(dashboard.group_data(df))
assert len(groups) == 2
assert {key[5] for key, _group in groups} == {"recipe-a", "recipe-b"}
assert {key[7] for key, _group in groups} == {"sw-a", "sw-b"}
def test_group_data_fills_legacy_cohort_columns():
df = pd.DataFrame([{
"model_id": "legacy-wan",
"gpu_type": "NVIDIA L40S",
"timestamp": "2026-01-01T00:00:00+00:00",
"commit_sha": "a" * 40,
"config_id": "aaaaaaa",
"latency": 10.0,
}])
groups = list(dashboard.group_data(df))
assert len(groups) == 1
assert groups[0][0][2:] == ("", "", "", "", "", "")
def test_build_plots_labels_distinct_cohorts():
df = pd.DataFrame([
_record(recipe_fingerprint="recipe-a", software_profile_id="sw-a"),
_record(recipe_fingerprint="recipe-b", software_profile_id="sw-b"),
])
figs, skipped_metrics = dashboard.build_plots(df)
titles = [fig.layout.title.text for fig in figs]
assert len(figs) == len(dashboard.METRICS) * 2
assert skipped_metrics == []
assert any("recipe recipe-a | hw-l40s | sw-a" in title for title in titles)
assert any("recipe recipe-b | hw-l40s | sw-b" in title for title in titles)
def test_skipped_metric_table_includes_cohort_identity():
df = pd.DataFrame([{
"model_id": "wan-t2v-1.3b-2gpu",
"gpu_type": "NVIDIA L40S",
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"recipe_fingerprint": "recipe-a",
"hardware_profile_id": "hw-l40s",
"software_profile_id": "sw-a",
"timestamp": "2026-01-01T00:00:00+00:00",
"commit_sha": "a" * 40,
"config_id": "aaaaaaa",
"latency": 10.0,
}])
_figs, skipped_metrics = dashboard.build_plots(df)
html = dashboard.render_skipped_metrics(skipped_metrics[:1])
assert skipped_metrics[0]["cohort"] == "wan-t2v / 1.3b-sp2 / v2"
assert skipped_metrics[0]["cohort_detail"] == "recipe recipe-a | hw-l40s | sw-a"
assert "wan-t2v / 1.3b-sp2 / v2" in html
assert "recipe recipe-a | hw-l40s | sw-a" in html
@@ -144,6 +144,204 @@ def test_build_latest_summary_separates_informational_threshold_crossing():
assert rows[0]["computed_regression_status"] == "pass"
def test_build_latest_summary_run_source_filter_excludes_future_baselines():
records = [
_record(
"2026-01-01T00:00:00+00:00",
"a" * 40,
10.0,
10.0,
run_source="scheduled_main",
baseline_eligible=True,
),
_record(
"2026-01-02T00:00:00+00:00",
"b" * 40,
11.0,
9.0,
run_source="pr",
baseline_eligible=False,
pr_number="123",
),
_record(
"2026-01-03T00:00:00+00:00",
"c" * 40,
30.0,
3.0,
run_source="scheduled_main",
baseline_eligible=True,
),
]
rows = build_latest_summary(records, run_source="pr")
assert len(rows) == 1
assert rows[0]["run_source"] == "pr"
assert rows[0]["baseline_n"] == 1
assert rows[0]["metrics"]["latency"]["baseline"] == 10.0
assert rows[0]["metrics"]["throughput"]["baseline"] == 10.0
def test_build_latest_summary_keeps_identity_cohorts_separate():
records = [
_record(
"2026-01-01T00:00:00+00:00",
"a" * 40,
10.0,
10.0,
recipe_fingerprint="recipe-a",
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=2,
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
run_source="scheduled_main",
baseline_eligible=True,
),
_record(
"2026-01-02T00:00:00+00:00",
"b" * 40,
20.0,
5.0,
recipe_fingerprint="recipe-b",
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=2,
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
run_source="scheduled_main",
baseline_eligible=True,
),
_record(
"2026-01-03T00:00:00+00:00",
"c" * 40,
22.0,
4.5,
recipe_fingerprint="recipe-b",
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=2,
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
),
]
rows = build_latest_summary(records)
recipe_b_row = next(row for row in rows if row["recipe_fingerprint"] == "recipe-b")
assert len(rows) == 2
assert recipe_b_row["baseline_n"] == 1
assert recipe_b_row["metrics"]["latency"]["baseline"] == 20.0
assert recipe_b_row["workload_id"] == "wan-t2v"
assert recipe_b_row["variant_id"] == "1.3b-sp2"
assert recipe_b_row["benchmark_version"] == 2
def test_build_latest_summary_keeps_variant_versions_separate():
records = [
_record(
"2026-01-01T00:00:00+00:00",
"a" * 40,
10.0,
10.0,
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=1,
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
run_source="scheduled_main",
baseline_eligible=True,
),
_record(
"2026-01-02T00:00:00+00:00",
"b" * 40,
20.0,
5.0,
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=2,
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
run_source="scheduled_main",
baseline_eligible=True,
),
_record(
"2026-01-03T00:00:00+00:00",
"c" * 40,
22.0,
4.5,
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=2,
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
),
]
rows = build_latest_summary(records)
version_2_row = next(row for row in rows if row["benchmark_version"] == 2)
assert len(rows) == 2
assert version_2_row["baseline_n"] == 1
assert version_2_row["metrics"]["latency"]["baseline"] == 20.0
def test_dashboard_identity_preserves_zero_benchmark_version():
records = [
_record(
"2026-01-01T00:00:00+00:00",
"a" * 40,
10.0,
10.0,
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=0,
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
run_source="scheduled_main",
baseline_eligible=True,
),
_record(
"2026-01-02T00:00:00+00:00",
"b" * 40,
11.0,
9.0,
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version=0,
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
),
_record(
"2026-01-03T00:00:00+00:00",
"c" * 40,
20.0,
5.0,
),
]
rows = build_latest_summary(records)
version_zero_row = next(row for row in rows if row["benchmark_version"] == 0)
legacy_row = next(row for row in rows if row["benchmark_version"] == "")
trends = build_trends(records)
version_zero_trend = next(trend for trend in trends if trend["benchmark_version"] == 0)
legacy_trend = next(trend for trend in trends if trend["benchmark_version"] == "")
assert len(rows) == 2
assert version_zero_row["baseline_n"] == 1
assert version_zero_row["metrics"]["latency"]["baseline"] == 10.0
assert legacy_row["baseline_n"] == 0
assert len(trends) == 2
assert version_zero_trend["points"][0]["benchmark_version"] == 0
assert legacy_trend["points"][0]["benchmark_version"] == ""
def test_filter_records_and_trends_preserve_metric_points():
records = [
_record("2026-01-01T00:00:00+00:00", "a" * 40, 10.0, 10.0),
@@ -218,3 +416,78 @@ def test_load_records_can_filter_baseline_eligible_records(tmp_path):
"2026-01-02T00:00:00+00:00",
"2026-01-03T00:00:00+00:00",
}
def test_load_records_for_model_filters_identity_cohort(tmp_path):
model_dir = tmp_path / "wan"
model_dir.mkdir()
(model_dir / "matching.json").write_text(
"""
{
"model_id": "wan",
"gpu_type": "NVIDIA L40S",
"timestamp": "2026-01-01T00:00:00+00:00",
"success": true,
"baseline_eligible": true,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"recipe_fingerprint": "recipe-a",
"hardware_profile_id": "hw-l40s",
"software_profile_id": "sw-cu130"
}
""",
encoding="utf-8",
)
(model_dir / "other_recipe.json").write_text(
"""
{
"model_id": "wan",
"gpu_type": "NVIDIA L40S",
"timestamp": "2026-01-02T00:00:00+00:00",
"success": true,
"baseline_eligible": true,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"recipe_fingerprint": "recipe-b",
"hardware_profile_id": "hw-l40s",
"software_profile_id": "sw-cu130"
}
""",
encoding="utf-8",
)
(model_dir / "other_version.json").write_text(
"""
{
"model_id": "wan",
"gpu_type": "NVIDIA L40S",
"timestamp": "2026-01-03T00:00:00+00:00",
"success": true,
"baseline_eligible": true,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 3,
"recipe_fingerprint": "recipe-a",
"hardware_profile_id": "hw-l40s",
"software_profile_id": "sw-cu130"
}
""",
encoding="utf-8",
)
records = hf_store.load_records_for_model(
str(tmp_path),
"wan",
"NVIDIA L40S",
workload_id="wan-t2v",
variant_id="1.3b-sp2",
benchmark_version="2",
recipe_fingerprint="recipe-a",
hardware_profile_id="hw-l40s",
software_profile_id="sw-cu130",
baseline_eligible_only=True,
)
assert len(records) == 1
assert records[0]["recipe_fingerprint"] == "recipe-a"
@@ -0,0 +1,422 @@
# SPDX-License-Identifier: Apache-2.0
from copy import deepcopy
import pytest
import torch
from fastvideo.tests.performance import identity as identity_module
from fastvideo.tests.performance.identity import (
build_recipe_from_benchmark_config,
canonical_json,
environment_fingerprint,
environment_metadata,
hardware_profile,
hardware_profile_id,
recipe_fingerprint,
resolved_revision_from_model_path,
software_profile,
software_profile_id,
)
def _benchmark_config():
return {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"model": {
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
"model_short_name": "Wan2.1-T2V-1.3B",
},
"init_kwargs": {
"num_gpus": 2,
"sp_size": 2,
"tp_size": 1,
"vae_sp": True,
"vae_tiling": True,
"text_encoder_precisions": ["fp32"],
},
"generation_kwargs": {
"height": 480,
"width": 832,
"num_frames": 45,
"num_inference_steps": 4,
"guidance_scale": 3,
"embedded_cfg_scale": 6,
"seed": 1024,
"fps": 24,
"neg_prompt": "low quality",
},
"test_prompts": ["A cinematic video."],
}
def _fingerprint(cfg, attention_backend="FLASH_ATTN"):
recipe = build_recipe_from_benchmark_config(cfg, attention_backend=attention_backend)
return recipe_fingerprint(recipe)
def test_canonical_json_is_stable_for_mapping_order():
left = {"b": [2, 1], "a": {"z": "last", "m": "middle"}}
right = {"a": {"m": "middle", "z": "last"}, "b": [2, 1]}
assert canonical_json(left) == canonical_json(right)
def test_canonical_json_handles_sets_deterministically():
assert canonical_json({"values": {"b", "a"}}) == '{"values":["a","b"]}'
assert canonical_json({"values": frozenset(("b", "a"))}) == '{"values":["a","b"]}'
def test_same_config_produces_same_recipe_fingerprint():
cfg = _benchmark_config()
assert _fingerprint(cfg) == _fingerprint(deepcopy(cfg))
def test_recipe_includes_first_class_benchmark_identity():
recipe = build_recipe_from_benchmark_config(_benchmark_config())
assert recipe["benchmark"] == {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
}
def test_recipe_requires_first_class_benchmark_identity():
cfg = _benchmark_config()
del cfg["workload_id"]
with pytest.raises(ValueError, match="workload_id"):
build_recipe_from_benchmark_config(cfg)
def test_semantically_equivalent_sequence_forms_match():
cfg = _benchmark_config()
equivalent = deepcopy(cfg)
equivalent["init_kwargs"]["text_encoder_precisions"] = tuple(
equivalent["init_kwargs"]["text_encoder_precisions"])
assert _fingerprint(cfg) == _fingerprint(equivalent)
def test_output_path_does_not_change_recipe_fingerprint():
cfg = _benchmark_config()
with_output_path = deepcopy(cfg)
with_output_path["generation_kwargs"]["output_path"] = "/tmp/generated"
assert _fingerprint(cfg) == _fingerprint(with_output_path)
def test_runtime_resolved_values_do_not_change_recipe_fingerprint():
# Runtime-resolved values are audit metadata: an upstream repo commit
# (even a README-only edit) or a silent attention-backend fallback must
# not churn the cohort — that would open a fresh ungated cohort and let
# degraded numbers seed a new baseline. Declared inputs (including
# requested_backend) stay in the hash.
cfg = _benchmark_config()
recipe = build_recipe_from_benchmark_config(
cfg, attention_backend="FLASH_ATTN",
resolved_attention_backend="FLASH_ATTN",
resolved_model_revision="aaaa1111")
changed = build_recipe_from_benchmark_config(
cfg, attention_backend="FLASH_ATTN",
resolved_attention_backend="TORCH_SDPA",
resolved_model_revision="bbbb2222")
assert recipe["model"]["resolved_revision"] == "aaaa1111" # kept for audit
assert recipe["attention"]["resolved_backend"] == "FLASH_ATTN" # kept for audit
assert recipe_fingerprint(recipe) == recipe_fingerprint(changed)
def test_measured_prompt_override_ignores_unused_configured_prompts():
cfg = _benchmark_config()
changed = deepcopy(cfg)
changed["test_prompts"] = [
"Unused prompt that should not affect this measured workload.",
cfg["test_prompts"][0],
]
recipe = build_recipe_from_benchmark_config(cfg, measured_prompts=[cfg["test_prompts"][0]])
changed_recipe = build_recipe_from_benchmark_config(changed, measured_prompts=[cfg["test_prompts"][0]])
assert recipe_fingerprint(recipe) == recipe_fingerprint(changed_recipe)
def test_num_inference_steps_changes_recipe_fingerprint():
cfg = _benchmark_config()
changed = deepcopy(cfg)
changed["generation_kwargs"]["num_inference_steps"] = 8
assert _fingerprint(cfg) != _fingerprint(changed)
def test_benchmark_version_changes_recipe_fingerprint():
cfg = _benchmark_config()
changed = deepcopy(cfg)
changed["benchmark_version"] = 3
assert _fingerprint(cfg) != _fingerprint(changed)
def test_attention_backend_changes_recipe_fingerprint():
cfg = _benchmark_config()
assert _fingerprint(cfg, attention_backend="FLASH_ATTN") != _fingerprint(
cfg, attention_backend="TORCH_SDPA")
def test_precision_changes_recipe_fingerprint():
cfg = _benchmark_config()
changed = deepcopy(cfg)
changed["init_kwargs"]["text_encoder_precisions"] = ["bf16"]
assert _fingerprint(cfg) != _fingerprint(changed)
def test_dimensions_and_seed_change_recipe_fingerprint():
cfg = _benchmark_config()
different_dimensions = deepcopy(cfg)
different_dimensions["generation_kwargs"]["width"] = 1024
different_seed = deepcopy(cfg)
different_seed["generation_kwargs"]["seed"] = 2048
assert _fingerprint(cfg) != _fingerprint(different_dimensions)
assert _fingerprint(cfg) != _fingerprint(different_seed)
def test_distributed_layout_changes_recipe_fingerprint():
cfg = _benchmark_config()
changed = deepcopy(cfg)
changed["init_kwargs"]["sp_size"] = 1
changed["init_kwargs"]["tp_size"] = 2
assert _fingerprint(cfg) != _fingerprint(changed)
def test_software_profile_uses_exact_attention_kernel_package_versions_for_id():
profile_a = software_profile(
python_version="3.12.4",
torch_version="2.12.0+cu130",
cuda_version="13.0",
package_versions={
"triton": "3.4.1",
"fastvideo_kernel": "0.3.2",
"flash_attn_4": "4.0.0.dev0",
"flashinfer": "0.2.11",
"nvidia_cutlass_dsl": "4.5.0",
},
)
profile_b = software_profile(
python_version="3.12.5",
torch_version="2.12.1+cu130",
cuda_version="13.0",
package_versions={
"triton": "3.4.9",
"fastvideo_kernel": "0.3.7",
"flash_attn_4": "4.0.0.dev1",
"flashinfer": "0.2.12",
"nvidia_cutlass_dsl": "4.5.1",
},
)
assert profile_a == {
"python": "3.12",
"pytorch": "2.12",
"cuda": "13.0",
"packages": {
"fastvideo_kernel": "0.3.2",
"flash_attn_4": "4.0.0.dev0",
"flashinfer": "0.2.11",
"nvidia_cutlass_dsl": "4.5.0",
"triton": "3.4.1",
},
}
assert profile_b["python"] == "3.12"
assert profile_b["pytorch"] == "2.12"
assert profile_b["cuda"] == "13.0"
assert software_profile_id(profile_a) != software_profile_id(profile_b)
def test_installed_package_versions_checks_attention_kernel_distributions(monkeypatch):
versions = {
"fastvideo-kernel": "0.3.2",
"flash-attn": "2.8.1",
"flash-attn-4": "4.0.0.dev0",
"flash-attention-fp4": "0.1.0",
"flashinfer-python": "0.2.11",
"nvidia-cutlass-dsl": "4.5.0",
}
def fake_version(name):
if name not in versions:
raise identity_module.importlib.metadata.PackageNotFoundError(name)
return versions[name]
monkeypatch.setattr(identity_module.importlib.metadata, "version", fake_version)
installed = identity_module._installed_package_versions()
assert installed["fastvideo_kernel"] == "0.3.2"
assert installed["flash_attn"] == "2.8.1"
assert installed["flash_attn_4"] == "4.0.0.dev0"
assert installed["flash_attention_fp4"] == "0.1.0"
assert installed["flashinfer"] == "0.2.11"
assert installed["nvidia_cutlass_dsl"] == "4.5.0"
def test_hardware_profile_id_uses_normalized_gpu_cohort():
profile = hardware_profile(
num_gpus=2,
gpu_devices=[
{
"name": "NVIDIA L40S",
"memory_bytes": 48 * 1024**3,
"compute_capability": "8.9",
},
{
"name": "NVIDIA L40S",
"memory_gb": 48,
"compute_capability": "8.9",
},
],
interconnect="none_or_partial",
)
assert profile == {
"device_type": "cuda",
"gpu_count": 2,
"gpus": [
{
"name": "NVIDIA L40S",
"memory_gb": 48,
"compute_capability": "8.9",
},
{
"name": "NVIDIA L40S",
"memory_gb": 48,
"compute_capability": "8.9",
},
],
"interconnect": "none_or_partial",
}
assert hardware_profile_id(profile).startswith("hw-")
def test_hardware_profile_pads_missing_requested_cuda_devices(monkeypatch):
class Props:
name = "NVIDIA L40S"
total_memory = 48 * 1024**3
monkeypatch.setattr(torch.cuda, "is_available", lambda: True)
monkeypatch.setattr(torch.cuda, "device_count", lambda: 1)
monkeypatch.setattr(torch.cuda, "get_device_properties", lambda device_id: Props())
monkeypatch.setattr(torch.cuda, "get_device_capability", lambda device_id: (8, 9))
profile = hardware_profile(num_gpus=2, interconnect="unknown")
assert profile["gpu_count"] == 2
assert profile["gpus"] == [
{
"name": "NVIDIA L40S",
"memory_gb": 48,
"compute_capability": "8.9",
},
{
"name": "unknown",
"memory_gb": None,
"compute_capability": None,
},
]
def test_local_path_detection_checks_both_path_separators(monkeypatch):
monkeypatch.setattr(identity_module.os, "sep", "\\")
assert identity_module._looks_like_local_path("models/local/checkpoint") is True
assert identity_module._looks_like_local_path(r"C:\models\checkpoint") is True
assert identity_module._looks_like_local_path("Wan-AI/Wan2.1-T2V-1.3B-Diffusers") is False
def test_environment_metadata_is_separate_audit_fingerprint():
cfg = _benchmark_config()
recipe_hash = _fingerprint(cfg)
audit_a = environment_metadata(
env={"IMAGE_VERSION": "py3.12-cuda13.0.0"},
package_versions={"triton": "3.4.1"},
hardware={"device_type": "cuda", "gpu_count": 2},
software={"python": "3.12", "pytorch": "2.12", "cuda": "13.0"},
)
audit_b = environment_metadata(
env={"IMAGE_VERSION": "py3.12-cuda13.0.1"},
package_versions={"triton": "3.4.1"},
hardware={"device_type": "cuda", "gpu_count": 2},
software={"python": "3.12", "pytorch": "2.12", "cuda": "13.0"},
)
assert recipe_hash == _fingerprint(cfg)
assert environment_fingerprint(audit_a) != environment_fingerprint(audit_b)
def test_resolved_revision_from_hf_snapshot_path():
revision = "a" * 40
assert resolved_revision_from_model_path(
f"/root/.cache/huggingface/hub/models--org--repo/snapshots/{revision}") == revision
assert resolved_revision_from_model_path("/models/local-repo") is None
def test_runtime_identity_from_generator_summarizes_worker_records():
from fastvideo.tests.performance.test_inference_performance import _runtime_identity_from_generator
revision = "b" * 40
class FakeExecutor:
def collective_rpc(self, _method):
return [
{
"resolved_attention_backends": ["FLASH_ATTN"],
"resolved_model_path": f"/root/.cache/huggingface/hub/models--org--repo/snapshots/{revision}",
},
{
"resolved_attention_backends": ["FLASH_ATTN"],
"resolved_model_path": f"/root/.cache/huggingface/hub/models--org--repo/snapshots/{revision}",
},
]
class FakeGenerator:
executor = FakeExecutor()
assert _runtime_identity_from_generator(FakeGenerator()) == {
"resolved_attention_backend": "FLASH_ATTN",
"resolved_model_revision": revision,
}
def test_collect_worker_identity_handles_torch_module_pipeline():
from fastvideo.tests.performance.test_inference_performance import _collect_worker_identity
class Backend:
name = "FLASH_ATTN"
class Leaf(torch.nn.Module):
backend = Backend()
class Pipeline(torch.nn.Module):
model_path = "/models/local"
def __init__(self):
super().__init__()
self.leaf = Leaf()
class Worker:
pipeline = Pipeline()
assert _collect_worker_identity(Worker()) == {
"resolved_attention_backends": ["FLASH_ATTN"],
"resolved_model_path": "/models/local",
}
@@ -12,12 +12,25 @@ import os
import time
from collections.abc import Mapping
from datetime import datetime, timezone
from typing import Any
import torch
import pytest
from fastvideo import VideoGenerator
from fastvideo.logger import init_logger
from fastvideo.tests.performance.identity import (
benchmark_identity_from_config,
build_recipe_from_benchmark_config,
environment_fingerprint,
environment_metadata,
hardware_profile,
hardware_profile_id,
recipe_fingerprint,
resolved_revision_from_model_path,
software_profile,
software_profile_id,
)
from fastvideo.worker.multiproc_executor import MultiprocExecutor
logger = init_logger(__name__)
@@ -34,11 +47,17 @@ V2_REQUIRED_IDENTITY_FIELDS = (
"variant_id",
"benchmark_version",
)
# "recipe" is no longer config-declarable: the fingerprint follow-up landed,
# and the generated recipe document owns that key in emitted records. A config
# declaring it now fails validation loudly instead of being silently
# overwritten by the generated one.
V2_OPTIONAL_METADATA_FIELDS = (
"recipe",
"metric_threshold_policy",
"quality_metadata",
)
RESULT_SCHEMA_VERSION = 2
VALID_RUN_SOURCES = {"pr", "local", "scheduled_main", "unknown"}
OPTIONAL_RESULT_METADATA_FIELDS = ("quality_metadata", "variant_metadata")
# -- Config discovery -------------------------------------------------------
@@ -215,6 +234,17 @@ def _run_generation(generator, prompt, generation_kwargs):
component_times = _extract_component_times(result)
return elapsed, peak_memory_mb, component_times
def _validate_run_counts(run_config: Mapping[str, Any], benchmark_id: str) -> tuple[int, int]:
num_warmup = run_config.get("num_warmup_runs", 1)
num_measure = run_config.get("num_measurement_runs", 3)
if isinstance(num_warmup, bool) or not isinstance(num_warmup, int) or num_warmup < 0:
raise ValueError(f"{benchmark_id}: run_config.num_warmup_runs must be a non-negative integer")
if isinstance(num_measure, bool) or not isinstance(num_measure, int) or num_measure < 1:
raise ValueError(f"{benchmark_id}: run_config.num_measurement_runs must be a positive integer")
return num_warmup, num_measure
def _write_results(results):
"""Write JSON results to the results directory."""
script_dir = os.path.dirname(os.path.abspath(__file__))
@@ -231,6 +261,205 @@ def _write_results(results):
logger.info("Performance results written to %s", filepath)
def _backend_name(value) -> str:
if hasattr(value, "name"):
return str(value.name)
return str(value)
def _collect_worker_identity(worker) -> dict[str, Any]:
pipeline = getattr(worker, "pipeline", None)
model_path = getattr(pipeline, "model_path", None)
backends: set[str] = set()
modules = getattr(pipeline, "modules", None)
if isinstance(modules, Mapping):
module_iter = modules.values()
elif isinstance(pipeline, torch.nn.Module):
module_iter = (pipeline,)
else:
module_iter = ()
for module in module_iter:
if isinstance(module, torch.nn.Module):
for submodule in module.modules():
backend = getattr(submodule, "backend", None)
if backend is not None:
backends.add(_backend_name(backend))
else:
backend = getattr(module, "backend", None)
if backend is not None:
backends.add(_backend_name(backend))
return {
"resolved_attention_backends": sorted(backends),
"resolved_model_path": model_path,
}
def _single_or_list(values: set[str]) -> str | list[str] | None:
ordered = sorted(values)
if not ordered:
return None
if len(ordered) == 1:
return ordered[0]
return ordered
def _runtime_identity_from_generator(generator) -> dict[str, Any]:
worker_records = generator.executor.collective_rpc(_collect_worker_identity)
resolved_backends: set[str] = set()
resolved_revisions: set[str] = set()
for record in worker_records:
resolved_backends.update(record.get("resolved_attention_backends") or [])
revision = resolved_revision_from_model_path(record.get("resolved_model_path"))
if revision is not None:
resolved_revisions.add(revision)
return {
"resolved_attention_backend": _single_or_list(resolved_backends),
"resolved_model_revision": _single_or_list(resolved_revisions),
}
def _benchmark_identity_fields(cfg):
identity = benchmark_identity_from_config(cfg)
return {
key: identity[key]
for key in ("workload_id", "variant_id", "benchmark_version")
}
def _build_identity_fields(cfg, init_kwargs, prompt, runtime_identity):
# Legacy v1 configs (no config_schema_version) stay on the legacy
# (model_id, gpu_type) cohort: they lack the required identity fields,
# and building a recipe for them would raise AFTER the GPU measurement
# completed. The validator already enforces that v2 configs carry the
# identity fields, so v2 records always get the full identity block.
if cfg.get("config_schema_version") is None:
return {}
recipe = build_recipe_from_benchmark_config(
cfg,
resolved_attention_backend=runtime_identity.get("resolved_attention_backend"),
resolved_model_revision=runtime_identity.get("resolved_model_revision"),
measured_prompts=[prompt],
)
num_gpus = init_kwargs.get("num_gpus", cfg.get("run_config", {}).get("required_gpus", 1))
hw_profile = hardware_profile(num_gpus=num_gpus)
sw_profile = software_profile()
env_metadata = environment_metadata(
hardware=hw_profile,
software=sw_profile,
)
return {
"recipe": recipe,
"recipe_fingerprint": recipe_fingerprint(recipe),
"hardware_profile": hw_profile,
"hardware_profile_id": hardware_profile_id(hw_profile),
"software_profile": sw_profile,
"software_profile_id": software_profile_id(sw_profile),
"environment_metadata": env_metadata,
"environment_fingerprint": environment_fingerprint(env_metadata),
**_benchmark_identity_fields(cfg),
}
def _truthy_pr_number(value: str | None) -> bool:
return bool(value and value not in {"false", "0", "None", "none"})
def _detect_run_source() -> str:
explicit = os.environ.get("PERF_RUN_SOURCE", "").strip().lower()
if explicit in VALID_RUN_SOURCES:
return explicit
if _truthy_pr_number(os.environ.get("BUILDKITE_PULL_REQUEST")):
return "pr"
if os.environ.get("BUILDKITE_BRANCH") == "main" and os.environ.get("TEST_SCOPE") == "full":
return "scheduled_main"
if not os.environ.get("BUILDKITE_COMMIT"):
return "local"
return "unknown"
def _ci_provenance_fields() -> dict[str, str]:
pr_number = os.environ.get("BUILDKITE_PULL_REQUEST", "")
if not _truthy_pr_number(pr_number):
pr_number = ""
return {
"run_source": _detect_run_source(),
"branch": os.environ.get("BUILDKITE_BRANCH", ""),
"pr_number": pr_number,
"test_scope": os.environ.get("TEST_SCOPE", ""),
"build_url": os.environ.get("BUILDKITE_BUILD_URL", ""),
"build_id": os.environ.get("BUILDKITE_BUILD_ID", ""),
"job_id": os.environ.get("BUILDKITE_JOB_ID", ""),
}
def _configured_result_metadata(cfg: Mapping[str, Any]) -> dict[str, Any]:
return {
field: cfg[field]
for field in OPTIONAL_RESULT_METADATA_FIELDS
if field in cfg and cfg[field] is not None
}
def _build_result_record(
*,
cfg: Mapping[str, Any],
model_info: Mapping[str, Any],
init_kwargs: Mapping[str, Any],
gen_kwargs: Mapping[str, Any],
num_warmup: int,
num_measure: int,
thresholds: Mapping[str, Any],
times: list[float],
peak_memories: list[float],
all_component_times: list[dict],
prompt: str,
runtime_identity: Mapping[str, Any],
device_name: str,
timestamp: str | None = None,
) -> dict[str, Any]:
if not times or not peak_memories:
raise ValueError("Cannot build a performance result record without measurement runs")
avg_time = sum(times) / len(times)
max_peak_memory = max(peak_memories)
num_frames = gen_kwargs.get("num_frames")
throughput_fps = (1.0 / avg_time) if avg_time > 0 else None
if isinstance(num_frames, (int, float)) and avg_time > 0:
throughput_fps = num_frames / avg_time
return {
"benchmark_id": cfg["benchmark_id"],
"result_schema_version": RESULT_SCHEMA_VERSION,
"model_short_name": model_info.get("model_short_name", ""),
"device": device_name,
"num_gpus": init_kwargs.get("num_gpus", 1),
"num_warmup_runs": num_warmup,
"num_measurement_runs": num_measure,
"avg_generation_time_s": round(avg_time, 3),
"individual_times_s": [round(t, 3) for t in times],
"throughput_fps": round(throughput_fps, 3)
if throughput_fps is not None else None,
"max_peak_memory_mb": round(max_peak_memory, 1),
"individual_peak_memories_mb": [round(m, 1) for m in peak_memories],
"thresholds": dict(thresholds),
"regression_thresholds": cfg.get("regression_thresholds", {}),
"commit": os.environ.get("BUILDKITE_COMMIT", ""),
**_ci_provenance_fields(),
"timestamp": timestamp or datetime.now(timezone.utc).isoformat(),
**_configured_result_metadata(cfg),
"text_encoder_time_s": _avg_component(all_component_times,
"text_encoder_time_s"),
"dit_time_s": _avg_component(all_component_times, "dit_time_s"),
"vae_decode_time_s": _avg_component(all_component_times,
"vae_decode_time_s"),
**_build_identity_fields(cfg, init_kwargs, prompt, runtime_identity),
}
# -- Test -------------------------------------------------------------------
def _run_benchmark(cfg):
@@ -246,8 +475,7 @@ def _run_benchmark(cfg):
prompts = cfg.get("test_prompts", ["A cinematic video."])
prompt = prompts[0]
num_warmup = run_config.get("num_warmup_runs", 1)
num_measure = run_config.get("num_measurement_runs", 3)
num_warmup, num_measure = _validate_run_counts(run_config, cfg["benchmark_id"])
thresholds = _get_thresholds(cfg)
# Remap JSON keys to VideoGenerator kwargs
@@ -268,6 +496,7 @@ def _run_benchmark(cfg):
model_path=model_info["model_path"],
**init_kwargs,
)
runtime_identity = _runtime_identity_from_generator(generator)
for i in range(num_warmup):
logger.info("Warmup run %d/%d", i + 1, num_warmup)
@@ -293,36 +522,21 @@ def _run_benchmark(cfg):
avg_time = sum(times) / len(times)
max_peak_memory = max(peak_memories)
device_name = torch.cuda.get_device_name()
num_frames = gen_kwargs.get("num_frames")
throughput_fps = (1.0 / avg_time) if avg_time > 0 else None
if isinstance(num_frames, (int, float)) and avg_time > 0:
throughput_fps = num_frames / avg_time
results = {
"benchmark_id": cfg["benchmark_id"],
**_config_identity_metadata(cfg),
"model_short_name": model_info.get("model_short_name", ""),
"device": device_name,
"num_gpus": init_kwargs.get("num_gpus", 1),
"num_warmup_runs": num_warmup,
"num_measurement_runs": num_measure,
"avg_generation_time_s": round(avg_time, 3),
"individual_times_s": [round(t, 3) for t in times],
"throughput_fps": round(throughput_fps, 3)
if throughput_fps is not None else None,
"max_peak_memory_mb": round(max_peak_memory, 1),
"individual_peak_memories_mb": [round(m, 1) for m in peak_memories],
"thresholds": thresholds,
"regression_thresholds": cfg.get("regression_thresholds", {}),
"commit": os.environ.get("BUILDKITE_COMMIT", ""),
"pr_number": os.environ.get("BUILDKITE_PULL_REQUEST", ""),
"timestamp": datetime.now(timezone.utc).isoformat(),
"text_encoder_time_s": _avg_component(all_component_times,
"text_encoder_time_s"),
"dit_time_s": _avg_component(all_component_times, "dit_time_s"),
"vae_decode_time_s": _avg_component(all_component_times,
"vae_decode_time_s"),
}
results = _build_result_record(
cfg=cfg,
model_info=model_info,
init_kwargs=init_kwargs,
gen_kwargs=gen_kwargs,
num_warmup=num_warmup,
num_measure=num_measure,
thresholds=thresholds,
times=times,
peak_memories=peak_memories,
all_component_times=all_component_times,
prompt=prompt,
runtime_identity=runtime_identity,
device_name=device_name,
)
logger.info(
"Performance results: avg_time=%.2fs, "
@@ -0,0 +1,176 @@
# SPDX-License-Identifier: Apache-2.0
import pytest
from fastvideo.tests.performance import test_inference_performance as perf_test
def _benchmark_config():
return {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"quality_metadata": {
"quality_status": "canonical",
},
"model": {
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
"model_short_name": "Wan2.1-T2V-1.3B",
},
"generation_kwargs": {
"num_frames": 45,
},
}
def _identity_fields():
return {
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
"recipe": {
"recipe_schema_version": 1,
"benchmark": {
"benchmark_id": "wan-t2v-1.3b-2gpu",
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 2,
},
},
"recipe_fingerprint": "recipe-1",
"hardware_profile": {
"device_type": "cuda",
"gpu_count": 2,
},
"hardware_profile_id": "hw-l40s",
"software_profile": {
"python": "3.12",
"pytorch": "2.12",
"cuda": "13.0",
},
"software_profile_id": "sw-cu130",
"environment_metadata": {
"env": {
"IMAGE_VERSION": "py3.12-cuda13.0.0",
},
},
"environment_fingerprint": "env-ci",
}
def test_build_result_record_emits_v2_wan_shape(monkeypatch):
monkeypatch.setenv("PERF_RUN_SOURCE", "scheduled_main")
monkeypatch.setenv("BUILDKITE_COMMIT", "a" * 40)
monkeypatch.setenv("BUILDKITE_PULL_REQUEST", "false")
monkeypatch.setenv("BUILDKITE_BRANCH", "main")
monkeypatch.setenv("TEST_SCOPE", "full")
monkeypatch.setenv("BUILDKITE_BUILD_URL", "https://buildkite.example/build")
monkeypatch.setenv("BUILDKITE_BUILD_ID", "build-1")
monkeypatch.setenv("BUILDKITE_JOB_ID", "job-1")
monkeypatch.setattr(perf_test, "_build_identity_fields", lambda *_args: _identity_fields())
record = perf_test._build_result_record(
cfg=_benchmark_config(),
model_info={"model_short_name": "Wan2.1-T2V-1.3B"},
init_kwargs={"num_gpus": 2},
gen_kwargs={"num_frames": 45},
num_warmup=2,
num_measure=3,
thresholds={
"max_generation_time_s": 34.0,
"max_peak_memory_mb": 11000.0,
},
times=[30.0, 31.0, 32.0],
peak_memories=[10000.0, 10100.0, 10050.0],
all_component_times=[
{
"text_encoder_time_s": 1.0,
"dit_time_s": 8.0,
"vae_decode_time_s": 3.0,
},
{
"text_encoder_time_s": 1.2,
"dit_time_s": 8.2,
"vae_decode_time_s": 3.2,
},
{
"text_encoder_time_s": None,
"dit_time_s": 8.4,
"vae_decode_time_s": 3.4,
},
],
prompt="A cinematic video.",
runtime_identity={
"resolved_attention_backend": "FLASH_ATTN",
},
device_name="NVIDIA L40S",
timestamp="2026-07-05T00:00:00+00:00",
)
assert record["result_schema_version"] == perf_test.RESULT_SCHEMA_VERSION
assert record["benchmark_id"] == "wan-t2v-1.3b-2gpu"
assert record["workload_id"] == "wan-t2v"
assert record["variant_id"] == "1.3b-sp2"
assert record["benchmark_version"] == 2
assert record["recipe_fingerprint"] == "recipe-1"
assert record["hardware_profile_id"] == "hw-l40s"
assert record["software_profile_id"] == "sw-cu130"
assert record["environment_fingerprint"] == "env-ci"
assert record["environment_metadata"]["env"]["IMAGE_VERSION"] == "py3.12-cuda13.0.0"
assert record["quality_metadata"] == {"quality_status": "canonical"}
assert record["run_source"] == "scheduled_main"
assert record["branch"] == "main"
assert record["test_scope"] == "full"
assert record["build_url"] == "https://buildkite.example/build"
assert record["build_id"] == "build-1"
assert record["job_id"] == "job-1"
assert record["pr_number"] == ""
assert record["commit"] == "a" * 40
assert record["avg_generation_time_s"] == 31.0
assert record["throughput_fps"] == 1.452
assert record["max_peak_memory_mb"] == 10100.0
assert record["text_encoder_time_s"] == 1.1
assert record["dit_time_s"] == 8.2
assert record["vae_decode_time_s"] == 3.2
def test_validate_run_counts_rejects_zero_measurement_runs():
with pytest.raises(ValueError, match="num_measurement_runs"):
perf_test._validate_run_counts({
"num_warmup_runs": 1,
"num_measurement_runs": 0,
}, "wan-t2v-1.3b-2gpu")
def test_validate_run_counts_rejects_negative_warmup_runs():
with pytest.raises(ValueError, match="num_warmup_runs"):
perf_test._validate_run_counts({
"num_warmup_runs": -1,
"num_measurement_runs": 3,
}, "wan-t2v-1.3b-2gpu")
def test_validate_run_counts_returns_defaults():
assert perf_test._validate_run_counts({}, "wan-t2v-1.3b-2gpu") == (1, 3)
def test_build_result_record_rejects_empty_measurements(monkeypatch):
monkeypatch.setattr(perf_test, "_build_identity_fields", lambda *_args: _identity_fields())
with pytest.raises(ValueError, match="measurement runs"):
perf_test._build_result_record(
cfg=_benchmark_config(),
model_info={},
init_kwargs={},
gen_kwargs={},
num_warmup=0,
num_measure=0,
thresholds={},
times=[],
peak_memories=[],
all_component_times=[],
prompt="A cinematic video.",
runtime_identity={},
device_name="NVIDIA L40S",
)