Compare commits

...
Author SHA1 Message Date
Satyam Srivastava 8087677db6 Add performance redesign plan docs 2026-05-15 15:56:02 -07:00
Satyam Srivastava 4c6c4ac988 [ci] Addressed review comments on the last four commits 2026-05-08 23:43:16 -07:00
Satyam Srivastava 0a3b0c173c [ci] Cherrypick perf regression skills to component time commit 2026-05-08 23:43:16 -07:00
Satyam Srivastava 82b0942c5b [ci] Upload normalized perf artifacts for rolling comparison failures
[ci] Add performance baseline reseed skill and failed-run artifacts
2026-05-08 23:42:55 -07:00
Satyam Srivastava b65aef2bd6 [bugfix]: enable perf stage timing in spawned workers
[ci] Reuse performance tracking HF sync for dashboard

  Add a local sync marker after successfully downloading the performance
  tracking snapshot from Hugging Face. The dashboard now reuses that existing
  snapshot when it runs after compare_baseline.py in the same CI job, avoiding
  a second snapshot_download call.

  If no prior sync marker exists, dashboard still falls back to syncing from HF,
  so local and standalone dashboard runs keep working.
2026-05-08 23:41:26 -07:00
Satyam Srivastava 5af8e93b46 [ci] Add component-level performance timings
Capture text encoder, DiT, and VAE decode timings during inference performance benchmarks by enabling stage logging around the benchmark run and aggregating mapped pipeline stage execution times.

Persist the new timing fields into performance tracking records, compare them against rolling baselines, and include them in dashboard plots when present. Add per-component thresholds for the Wan T2V performance benchmark while keeping the dashboard compatible with older records that do not contain component timing columns.
2026-05-08 23:41:26 -07:00
12 changed files with 1930 additions and 63 deletions
+1
View File
@@ -8,3 +8,4 @@
{"name": "seed-ssim-references", "description": "Run a new or updated fastvideo/tests/ssim/ test on Modal, pull generated videos, and upload them to FastVideo/ssim-reference-videos so the test has a regression baseline", "path": "seed-ssim-references/SKILL.md", "status": "draft", "trust": "low"}
{"name": "reseed-ssim-references", "description": "Re-seed (overwrite) HF reference videos for an existing fastvideo/tests/ssim/ test and a single model id on Modal L40S. Always backs up current refs first, regenerates on Modal, pauses for the user to eyeball before-vs-after, then uploads with --force scoped to --model-id. Sister skill to seed-ssim-references; use when intentional code change has invalidated existing refs", "path": "reseed-ssim-references/SKILL.md", "status": "draft", "trust": "low"}
{"name": "decompose-pipeline-pr", "description": "Decompose an oversized FastVideo pipeline PR into a stack of independently-reviewable PRs. Tiers the diff by blast radius (invisible / dead code / cross-cutting infra / activation), produces a branch graph and worktree bootstrap, drafts the AGENTS.md manifest, flags missing tests on cross-cutting infra changes, and extracts lessons from the PR body. Worked example: PR #1280 daVinci-MagiHuman (9.8k LOC) decomposed into 10 stacked PRs.", "path": "decompose-pipeline-pr/SKILL.md", "status": "tested", "trust": "medium"}
{"name": "reseed-performance-baseline", "description": "Re-seed the HF performance-tracking baseline for an intentional runtime, dependency, or environment-caused benchmark shift. Use when performance CI fails because metrics such as latency, throughput, component time, or peak memory changed for an accepted reason and the rolling median baseline must be advanced by replicating one reviewed shifted source result into three success=true records, or five records when explicitly requested", "path": "reseed-performance-baseline/SKILL.md", "status": "draft", "trust": "low"}
@@ -0,0 +1,476 @@
---
name: reseed-performance-baseline
description: Re-seed the HF performance-tracking baseline for an intentional runtime, dependency, or environment-caused benchmark shift using one or more reviewed normalized performance JSONs. Use when performance CI fails because metrics such as latency, throughput, component time, or peak memory changed for an accepted reason and the rolling median baseline in FastVideo/performance-tracking must be advanced from a consistent batch of reviewed source results. The workflow backs up existing history under /tmp, validates all source JSONs for the same (model_id, gpu_type), rejects internally inconsistent source batches, uploads one success=true reseed record per accepted source JSON, and offers to clean local temp state after a successful upload.
---
# Re-seed Performance Baseline
## Purpose
Replace or advance the rolling performance baseline for a single
`(model_id, gpu_type)` pair in the HF dataset
`FastVideo/performance-tracking`.
Performance comparison uses the median of up to the last 5 successful records
for the same model and GPU. Failed records are useful audit history, but they
do not move the future baseline because `compare_baseline.py` loads records
with `successful_only=True`.
This skill now reseeds from a reviewed batch of one or more source performance
JSONs. It uploads one new `success=true` record per accepted source JSON; it
does not blindly replicate one measurement into 3 or 5 records. The effective
reseed size is therefore dynamic and equals the number of provided, validated,
internally consistent source JSONs.
If the operator provides fewer than 3 records, call out that the last-5 rolling
median may not move immediately. If the operator provides 3 consistent shifted
records, the rolling median usually moves immediately. If the operator provides
5 consistent shifted records, the last-5 window is effectively reset to the new
runtime profile.
These records are intentional operator-approved baseline resets, not ordinary
independent main-branch persistence. Mark them clearly with provenance fields
so the HF history remains auditable.
Use this skill when a performance test fails for an intentional and reviewed
reason, such as a torch/runtime/container upgrade that legitimately increases
peak memory or changes timings. This is the performance equivalent of
`reseed-ssim-references`: backup first, scope tightly, require explicit human
approval, then upload reviewed accepted baseline records.
## When to use
- A PR or main run failed the rolling performance comparison by more than the
allowed regression threshold, and maintainers agree the shift is caused by
an intentional runtime, dependency, hardware image, or benchmark environment
change rather than a FastVideo logic regression.
- One or more shifted source result JSONs have been reviewed and accepted, and
the operator wants to use those exact reviewed results to advance the rolling
baseline.
- The source batch is internally consistent: no provided source JSON regresses
against the batch median by more than the configured tolerance.
## When not to use
- The benchmark failure might be a real code regression. Fix or investigate
the code path first.
- The fixed benchmark thresholds in
`.buildkite/performance-benchmarks/tests/*.json` are too low. Those are a
separate gate from the rolling HF baseline and may need a code review change.
- There is no clear source run, commit, and rationale. Baseline history is a
production signal; do not edit it without provenance.
- The provided source JSONs disagree materially with each other. Rerun or
investigate instead of uploading a noisy reseed batch.
## Inputs
| Parameter | Required | Description |
|-----------|----------|-------------|
| `model_id` | Yes | Benchmark id, e.g. `wan-t2v-1.3b-2gpu`. This maps to the HF subdirectory after `sanitize(model_id)`. |
| `gpu_type` | Yes | Exact GPU device string from the performance record, e.g. the L40S device name emitted by CI. Baselines are GPU-specific. |
| `source_results` | Yes | One or more local paths or Buildkite artifact URLs for accepted shifted performance JSONs. Prefer normalized `normalized_perf_*.json` artifacts emitted by `compare_baseline.py`. Accept `source_result` as an alias only for a single JSON. |
| `max_intra_batch_regression` | No | Maximum allowed regression of any source JSON against the source batch median. Default: `PERF_MAX_REGRESSION` if set, otherwise `0.05` (5%). |
| `intent_rationale` | Yes | One-line explanation for why the baseline shift is legitimate. This is written into provenance and should be reused in the PR. |
Hardcoded defaults:
- HF repo: `FastVideo/performance-tracking` (`HF_REPO_ID` override is
supported by the code, but use the default unless the user explicitly asks).
- Local sync root: `/tmp/perf-tracking` (`PERFORMANCE_TRACKING_ROOT` override
is supported).
- Backup root: `/tmp/performance_reseed_backup`.
- Download scratch root for source artifact URLs: `/tmp/performance_reseed_source`.
- Baseline window: last 5 `success=true` records for the same
`(model_id, gpu_type)`.
- Reseed count: dynamic. Upload exactly one accepted seed record per validated
source JSON.
## Steps
### 1. Validate the target and source results
Normalize `source_results` to a list. If the user passes a single
`source_result`, treat it as a one-element `source_results` list and report
that a single record may not move the last-5 median immediately.
If any source result is a Buildkite artifact URL, download it first into a
local scratch directory under `/tmp/performance_reseed_source/` and use the
downloaded JSON path for the rest of the workflow. If the agent cannot access
the artifact because Buildkite authentication is missing, ask the user to
download the artifact manually and provide the local path.
Prefer the normalized Buildkite artifact emitted by `compare_baseline.py`:
```text
perf_reports/results/normalized_perf_*.json
```
That file is already in the HF tracking schema. Load each normalized JSON
directly:
```python
import json
with open(source_result, encoding="utf-8") as f:
record = json.load(f)
```
Stop if any normalized record's `model_id` or `gpu_type` does not match the
requested `model_id` and `gpu_type`.
The source records may have `success: false` when they came from failed
rolling baseline comparisons. That is expected; only the reviewed reseed
records become new `success: true` baseline records after explicit approval.
Sort validated source records by their original `timestamp` ascending before
preparing the seed records. If a source timestamp is missing or unparsable,
preserve input order for those records and print a warning. This makes the
fresh reseed timestamps deterministic and makes it clear which records enter
the last-5 window when more than 5 source JSONs are provided.
Check that `HF_API_KEY` is exported. The sync path may be public, but the
upload path requires write access.
### 1a. Check source batch consistency
Before syncing or preparing uploads, reject source batches that are internally
inconsistent. Use the same metric direction as `compare_baseline.py`:
- Lower is better: `latency`, `memory`, `text_encoder_time_s`, `dit_time_s`,
`vae_decode_time_s`.
- Higher is better: `throughput`.
For each metric with at least two non-null source values:
1. Compute the source batch median.
2. For lower-is-better metrics, compute `(source_value - batch_median) / batch_median`.
3. For `throughput`, compute `(batch_median - source_value) / batch_median`.
4. Stop if any source record regresses against the batch median by more than
`max_intra_batch_regression`.
Default `max_intra_batch_regression` to `PERF_MAX_REGRESSION` when set,
otherwise `0.05`. Print a table with per-source values, batch median, and
worst intra-batch regression.
This check prevents uploading a mixed batch where one JSON is materially
slower or faster than the others. If the batch fails this check, ask the user
to provide a cleaner batch or explicitly investigate the variance. Do not
silently drop outliers unless the user gives a concrete reviewed reason and a
new source list.
### 1b. How to obtain source results from CI
The performance CI exports normalized source results for failed rolling
baseline comparisons when `compare_baseline.py` ran. The preferred artifacts
come from:
```text
perf_reports/results/normalized_perf_*.json
```
The normal operator flow is:
1. Open the failed Buildkite performance job or several reruns of the same
benchmark after the accepted environment shift.
2. Download the `normalized_perf_*.json` artifacts for the target benchmark.
3. Pass all reviewed local paths or artifact URLs as `source_results`.
Do not scrape the Markdown performance summary to reconstruct JSON. The
normalized JSON artifacts are the only supported source of truth for reseed
metrics and provenance. Raw `fastvideo/tests/performance/results/perf_*.json`
artifacts are not accepted by this skill. If no normalized JSON artifact is
present, that run is not a valid source for baseline reseeding.
### 2. Sync and back up existing HF records under /tmp
Use `fastvideo/tests/performance/hf_store.py` helpers directly. Do **not** use
`compare_baseline.py` as a sync shortcut; on full main runs it can persist
records, while this step must only fetch and back up existing history.
The sync command pattern is:
```bash
export PERFORMANCE_TRACKING_ROOT="${PERFORMANCE_TRACKING_ROOT:-/tmp/perf-tracking}"
export HF_REPO_ID="${HF_REPO_ID:-FastVideo/performance-tracking}"
PYTHONPATH=fastvideo/tests/performance python -c 'from hf_store import sync_from_hf; import os; sync_from_hf(os.environ["PERFORMANCE_TRACKING_ROOT"], strict=True)'
```
Then back up only the sanitized model directory under `/tmp`:
```bash
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
MODEL_SAFE=$(PYTHONPATH=fastvideo/tests/performance python - <<'PY'
from hf_store import sanitize
print(sanitize("<model_id>"))
PY
)
BACKUP_DIR="/tmp/performance_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
mkdir -p "$BACKUP_DIR"
cp -R "${PERFORMANCE_TRACKING_ROOT}/${MODEL_SAFE}" "$BACKUP_DIR/" 2>/dev/null || true
```
Write provenance next to the backup:
```bash
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
model_id: <model_id>
gpu_type: <gpu_type>
source_results:
- <source_result_1>
- <source_result_2>
reseed_record_count: <len(source_results)>
max_intra_batch_regression: <threshold>
head_commit: $(git rev-parse HEAD)
timestamp_utc: $(date -u +%FT%TZ)
reason: <intent_rationale>
EOF
```
If the backup has no prior records, this is not a destructive reseed; it is a
first baseline seed. Continue, but report that baseline history was empty.
### 3. Compute old baseline and candidate shift
Load the last 5 successful records for the target:
```python
from hf_store import load_records_for_model
records = load_records_for_model(
"/tmp/perf-tracking",
"<model_id>",
"<gpu_type>",
last_n=5,
successful_only=True,
)
```
Print a small table showing old medians, source batch medians, candidate
medians after appending the proposed seed records, and source batch spread for:
- `latency`
- `throughput`
- `memory`
- `text_encoder_time_s`
- `dit_time_s`
- `vae_decode_time_s`
Also print how many successful old records exist. Make clear:
- 1 seed record usually does not move a last-5 median by itself.
- 3 consistent seed records usually move the last-5 median immediately.
- 5 consistent seed records effectively reset the last-5 window.
- The records are intentional approved baseline resets and must be labeled
that way.
### 4. Confirm intent
Require an explicit confirmation phrase before preparing the upload:
> About to RE-SEED performance baseline for `<model_id>` on `<gpu_type>`.
> This will upload `<N>` new `success=true` records to
> `FastVideo/performance-tracking/<sanitize(model_id)>/`, one per accepted
> source JSON.
>
> Reason: `<intent_rationale>`
> Source results: `<source_results>`
> Reseed record count: `<N>`
> Max intra-batch regression: `<threshold>`
> Note: these records come from a reviewed source batch and are intended to
> move the rolling median to the accepted runtime profile. They are not
> ordinary main-branch persistence.
> HEAD: `<git rev-parse --short=12 HEAD>`
> Backup: `<BACKUP_DIR>`
>
> Reply `confirm performance reseed` to proceed, anything else to abort.
Do not continue unless the user types exactly `confirm performance reseed`.
### 5. Create the accepted seed records
Create one seed record from each normalized source result. Do not copy the
source JSON wholesale.
Infer the baseline field allowlist from all existing HF records for the target
`(model_id, gpu_type)` after syncing, including both `success=true` and
`success=false` records. Use the union of non-provenance keys present in those
target records, preserving only fields that also exist in the normalized
source record or are explicitly set by the reseed workflow. Always include
`model_id`, `timestamp`, and `success` because the upload path and baseline
loader depend on them. Always set `timestamp` to a fresh reseed timestamp and
`success` to `true`. Do not include unrelated source-only fields that are
absent from existing HF records.
Exclude existing provenance or operator metadata from the inferred baseline
field allowlist. At minimum, exclude keys prefixed with `baseline_reseed` and
any fields known to be local-only audit metadata.
If there are no previous HF records for the target model/GPU, fall back to this
default baseline field list:
- `model_id`
- `timestamp`
- `commit_sha`
- `gpu_type`
- `latency`
- `throughput`
- `memory`
- `text_encoder_time_s`
- `dit_time_s`
- `vae_decode_time_s`
- `success`
Do not upload extra fields from the source artifact.
Optional provenance fields are allowed and useful:
- `baseline_reseed: true`
- `baseline_reseed_reason`
- `baseline_reseed_source_result`
- `baseline_reseed_source_timestamp`
- `baseline_reseed_source_success`
- `baseline_reseed_batch_size`
- `baseline_reseed_batch_index`
- `baseline_reseed_operator`
- `baseline_reseed_max_intra_batch_regression`
Use a fresh reseed timestamp for each seed record, not the original source
result timestamp. This is required because
`load_records_for_model(..., last_n=5)` keeps the last records after loading
the model directory; stale filenames/timestamps may not enter the last-5
window and therefore may not move the median. Preserve the original source
timestamp in `baseline_reseed_source_timestamp`.
Use the existing filename convention from `_write_tracking_record()`:
`<sanitize(timestamp)>_<sanitize(commit_sha)>.json` under the sanitized model
directory, but include a deterministic suffix such as `_reseed_01`,
`_reseed_02`, and so on before `.json` so multiple records from the same
batch do not overwrite each other.
If a source record already exists on HF with `success=false`, do not edit it
in place unless the user explicitly asked for an audit-preserving correction.
Prefer uploading new accepted seed records so failed history remains visible.
### 6. Pause before upload
Print:
- Backup directory path under `/tmp`.
- Prepared local record paths under `PERFORMANCE_TRACKING_ROOT`.
- HF paths that will receive the new records.
- Old rolling medians.
- Source batch medians, source batch spread, reseed count, and candidate
medians.
- Rationale.
Ask the user to reply exactly `upload`. Anything else aborts and leaves the
prepared records plus backup on disk.
### 7. Upload only the scoped records
Use the shared storage helper so the path and repo type match CI:
```python
from hf_store import upload_record
upload_record("<local_record_path>", record, strict=True)
```
Run it once per prepared record. Each upload goes to:
```text
FastVideo/performance-tracking/<sanitize(model_id)>/<record_filename>.json
```
Never bulk upload the whole tracking root. Never modify another model's
directory in the same operation.
### 8. Report outcome and offer cleanup
Report:
- Uploaded HF paths.
- Backup directory under `/tmp`.
- Local tracking root, usually `/tmp/perf-tracking`.
- Old baseline window count and medians.
- Source batch medians, source batch spread, reseed count, and candidate
medians.
- Expected effect based on reseed count.
- Any separate threshold changes still needed in
`.buildkite/performance-benchmarks/tests/*.json`.
Include the `intent_rationale` in the PR or follow-up comment so reviewers can
distinguish an accepted baseline shift from a hidden regression.
After the upload is verified, ask whether the user wants to clear temporary
local state. Explain what each directory is for:
- `PERFORMANCE_TRACKING_ROOT`, usually `/tmp/perf-tracking`: local synced
mirror of `FastVideo/performance-tracking` plus the prepared local seed
records used for scoped upload.
- `/tmp/performance_reseed_backup/<...>`: local backup of the target model's
pre-reseed HF history plus `PROVENANCE.txt`, kept so a bad reseed can be
audited or corrected.
- `/tmp/performance_reseed_source/<...>` when used: downloaded source JSON
artifacts from Buildkite URLs.
Ask:
> Reseed succeeded. Do you want me to delete the local temp tracking mirror,
> source downloads, and reseed backup under `/tmp`? These files are local
> safety/audit artifacts only; HF already has the uploaded records.
>
> Reply `cleanup reseed temp` to delete them, anything else to keep them.
Do not delete anything unless the user replies exactly
`cleanup reseed temp`. If cleanup is requested, remove only the specific
directories created for this reseed. Never remove unrelated `/tmp` contents.
## Failure modes and handling
- **`HF_API_KEY` unset.** Stop before upload. Do not create an untracked
process that appears to have reseeded but never reached HF.
- **Source result does not match target.** Stop. The wrong benchmark or GPU
would poison a separate baseline.
- **Source batch is internally inconsistent.** Stop if any source regresses
against the source batch median by more than `max_intra_batch_regression`.
Ask for cleaner sources or a reviewed explanation before continuing.
- **Too few source records to move the median.** Continue only after making
clear that one or two records may not immediately move the last-5 median.
- **The source results are noisy or suspicious.** Stop. Reseeding amplifies
those measurements into the baseline, so they must be reviewed first.
- **HF sync fails.** Stop for destructive reseeds. A stale or empty sync can
make the old baseline look missing.
- **Candidate still violates fixed thresholds.** Report that this skill only
handles the rolling HF baseline; update benchmark JSON thresholds in code
review if maintainers accept the new absolute limit.
- **The user aborts at either confirmation.** Leave the backup and prepared
records on disk. Nothing should be uploaded.
- **The user declines cleanup.** Keep `/tmp/perf-tracking`, the source
download directory if any, and `/tmp/performance_reseed_backup/<...>` in
place for audit/debugging.
- **A bad seed was uploaded.** Use the backup and HF history to identify the
uploaded file, then remove or supersede it with an explicitly reviewed
corrective record. Do not silently rewrite unrelated history.
## References
- `.agents/skills/reseed-ssim-references/SKILL.md` — safety pattern for
intentional baseline replacement.
- `fastvideo/tests/performance/compare_baseline.py` — normalization, rolling
median comparison, and persistence rules.
- `fastvideo/tests/performance/hf_store.py` — HF sync, record loading,
`sanitize()`, and `upload_record()`.
- `fastvideo/tests/performance/test_inference_performance.py` — source result
JSON schema.
- `.buildkite/performance-benchmarks/tests/*.json` — fixed absolute benchmark
thresholds, separate from rolling baseline comparisons.
## Changelog
| Date | Change |
|------|--------|
| 2026-05-03 | Initial version. Sister workflow to `reseed-ssim-references`, scoped to one performance `(model_id, gpu_type)` baseline seed with backup, confirmation, provenance, and `success=true` upload. |
| 2026-05-03 | Previous policy: replicate one approved shifted source result into 3 success records by default, or 5 only when explicitly requested. Add provenance marker for replicated-source reseeds. Superseded by the 2026-05-08 dynamic multi-source policy. |
| 2026-05-08 | Replace fixed 3/5 replication with dynamic multi-source reseeding: upload one seed record per reviewed source JSON, validate intra-batch consistency, move backup/source scratch under `/tmp`, and ask whether to clean temp state after successful upload. |
@@ -36,7 +36,10 @@
"thresholds": {
"L40S": {
"max_generation_time_s": 34.0,
"max_peak_memory_mb": 11000.0
"max_peak_memory_mb": 11000.0,
"max_text_encoder_time_s": 5.0,
"max_dit_time_s": 10.0,
"max_vae_decode_time_s": 10.0
},
"default": {
"max_generation_time_s": 120.0,
+14
View File
@@ -121,6 +121,19 @@ upload_performance_artifacts() {
fi
}
_upload_normalized_perf_results() {
local found=0
while IFS= read -r -d '' target; do
found=1
log "Found normalized performance result: $target. Uploading to Buildkite..."
buildkite-agent artifact upload "$target"
done < <(find "$LOCAL_DIR" -path "*/results/normalized_perf_*.json" -print0)
if [ "$found" -eq 0 ]; then
log "No normalized performance result artifacts found. This is expected when the rolling performance comparison did not run."
fi
}
_cleanup_modal_volume() {
log "Cleaning up perf_reports/ from Modal Volume..."
if modal volume rm hf-model-weights "perf_reports/" --recursive; then
@@ -139,6 +152,7 @@ upload_performance_artifacts() {
_download_reports || { _cleanup_local; return 1; }
_upload_dashboard
_upload_perf_summary
_upload_normalized_perf_results
_cleanup_modal_volume
_cleanup_local
}
+12 -5
View File
@@ -240,17 +240,24 @@ def run_lora_extraction_tests():
],
volumes={"/root/data": model_vol})
def run_performance_tests():
# Dashboard runs after compare_baseline regardless of regression result so
# the trend view is always available when investigating a failed gate.
# compare_baseline.py runs only after pytest passes, so normalized_perf_*.json
# artifacts are emitted for rolling-baseline failures, not fixed-threshold
# pytest failures. dashboard.py still runs on red CI for observability.
run_test(
"export HF_HOME='/root/data/.cache' && "
"export PERFORMANCE_TRACKING_ROOT='/tmp/perf-tracking' && "
"hf auth login --token $HF_API_KEY && "
"pytest ./fastvideo/tests/performance -vs && "
"{ python ./fastvideo/tests/performance/compare_baseline.py; "
"pytest ./fastvideo/tests/performance -vs; "
"PYTEST_RC=$?; "
"PERF_RC=0; "
"if [ $PYTEST_RC -eq 0 ]; then "
"python ./fastvideo/tests/performance/compare_baseline.py; "
"PERF_RC=$?; "
"fi; "
"python ./fastvideo/tests/performance/dashboard.py || true; "
"exit $PERF_RC; }")
"FINAL_RC=$PYTEST_RC; "
"if [ $FINAL_RC -eq 0 ]; then FINAL_RC=$PERF_RC; fi; "
"exit $FINAL_RC")
@app.function(gpu="L40S:1",
@@ -0,0 +1,97 @@
# Performance Redesign Commit Stack
## Summary
Break the redesign into a stacked sequence where each commit is reviewable and should keep the
performance job runnable. The stack starts with docs/schema/config, then turns on v2 result emission,
storage, comparison# Performance Redesign Commit Stack
## Summary
Break the redesign into a stacked sequence where each commit is reviewable and keeps the performance job
runnable. The stack starts with docs/schema/config, then adds v2 result emission, identity-scoped
storage, exact comparison, dashboarding, and deeper timing/rank instrumentation.
Important correction: CALIBRATION_NEEDED records should not automatically seed baselines. New hardware/
software/runtime cohorts are seeded manually through the reseed-performance workflow after review.
## Commit Stack
1. [docs]: add performance redesign plan
- Commit fastvideo/tests/performance/redesign_plan.md.
- No code behavior changes.
2. [perf]: add v2 performance schema and identity helpers
- Add helpers for statuses, metric policies, stable JSON hashing, recipe fingerprinting, hardware/
software profile IDs, and v1/v2 field flattening.
- Add unit tests for stable fingerprints, changed recipe fields, and software profile changes.
3. [perf]: version benchmark configs with workload and variant identity
- Update .buildkite/performance-benchmarks/tests/wan-t2v-1.3b.json with workload_id,
variant_id=canonical, benchmark_version=1, comparison policy, and quality policy.
- Keep benchmark_id and existing thresholds for compatibility.
4. [perf]: emit v2 benchmark results with legacy compatibility
- Update test_inference_performance.py to write v2 fields: identity, provenance, recipe, hardware,
software, metrics, runs, and quality status.
- Preserve existing flat fields so old comparison paths still work during the stack.
- Reset CUDA peak memory before each measured run and store median metrics in v2.
5. [perf]: support identity-scoped HF storage with legacy reads
- Update hf_store.py to write v2 records under <workload>/<variant>/<hardware>/<software>/....
- Keep legacy <model_id>/... reads and dataframe loading.
- Add load_records_for_identity(...) while leaving load_records_for_model(...) intact.
6. [perf]: compare v2 records by exact benchmark identity
- Update compare_baseline.py to prefer v2 identity comparison and fall back to v1 only for legacy
records.
- Implement statuses: PASS, REGRESSION, CALIBRATION_NEEDED, RECIPE_MISMATCH, INFRA_ERROR,
QUALITY_BLOCKED.
- Add metric-specific percent and absolute thresholds.
- Missing exact baseline returns CALIBRATION_NEEDED and does not persist as a successful baseline
seed.
- Only manually reseeded/promoted records are used to initialize a new comparable baseline.
7. [perf]: make scheduled-main provenance and persistence explicit
- Add scheduled-main provenance fields: trigger_type, schedule_id, build_id, build_url, branch,
commit_sha.
- Gate persistence with explicit scheduled-main env support while keeping current TEST_SCOPE=full &&
branch=main compatibility.
- Remove required PR provenance; allow optional ad-hoc change_request.
8. [perf]: update dashboard for v2 identities and statuses
- Group dashboard charts by workload, variant, hardware profile, software profile, and benchmark
version.
- Show status, quality status, recipe fingerprint, software profile, and legacy v1 charts separately.
- Make calibration-needed cohorts visible but distinct from accepted baselines.
9. [perf]: collect low-overhead denoising timing metrics
- Extend PipelineLoggingInfo and DenoisingStage instrumentation for denoise step, transformer
forward, and scheduler step timings.
- Prefer low-overhead CUDA-event timing when available; keep behavior disabled unless performance/
stage logging is enabled.
- Map these into v2 metrics when present.
10. [perf]: aggregate rank-level distributed performance metrics
- Update multiprocess executor responses to include rank-level timing and memory metadata.
- Store rank metrics under v2 runs; compare rank-max memory/timing where applicable.
- Preserve rank-0 legacy fields for compatibility.
11. [perf]: enforce candidate quality metadata
- Implement candidate/quality handling for variants.
- Candidate variants can collect metrics but are not promotable when required quality evidence is
missing.
- Update the performance reseed skill/workflow to target one exact v2 identity tuple.
- Require provenance for runtime upgrades and explicit marking of accepted baseline records.
- Manual reseed writes the records that make a new cohort comparable.
- Keep old v1 reseed notes only as legacy guidance.
- After commits 2-6: run unit tests for schema, HF storage, and comparison logic.
- After commit 4: run a local import/discovery check for benchmark configs and helper serialization.
- After commit 6: test v1 legacy records, v2 missing baseline, recipe mismatch, regression, and pass
paths.
- After commit 8: test dashboard grouping with synthetic v1 and v2 records.
scoped.
- Before merging the full stack: run the Modal performance job once in non-persistent/dry-run mode.
## Assumptions
- This is a stacked commit sequence; each commit is independently reviewable but may depend on previous
commits.
- v2 records are not compared to v1 records.
- CALIBRATION_NEEDED exits successfully by default but is visible in summaries.
- Scheduled-main calibration records are audit data, not automatic baseline seeds.
- New cohorts become comparable only after manual reseed/promotion.
- Promoted baseline support is optional until records are explicitly marked accepted.
+83 -31
View File
@@ -36,7 +36,23 @@ TRACKING_ROOT = os.environ.get(
"PERFORMANCE_TRACKING_ROOT",
"/tmp/perf-tracking",
)
PERF_REPORTS_DIR = os.environ.get("PERF_REPORTS_DIR", "/root/data/perf_reports")
MAX_REGRESSION = float(os.environ.get("PERF_MAX_REGRESSION", "0.05"))
METRICS = (
("latency", "Latency", 3),
("throughput", "Throughput", 3),
("memory", "Memory", 1),
("text_encoder_time_s", "Text Enc", 3),
("dit_time_s", "DiT", 3),
("vae_decode_time_s", "VAE Decode", 3),
)
LOWER_IS_BETTER_METRICS = {
"latency",
"memory",
"text_encoder_time_s",
"dit_time_s",
"vae_decode_time_s",
}
def _should_persist_tracking() -> bool:
@@ -44,7 +60,6 @@ def _should_persist_tracking() -> bool:
branch = os.environ.get("BUILDKITE_BRANCH", "")
return test_scope == "full" and branch == "main"
def _load_current_results() -> list[dict[str, Any]]:
pattern = os.path.join(RESULTS_DIR, "perf_*.json")
records: list[dict[str, Any]] = []
@@ -54,7 +69,14 @@ def _load_current_results() -> list[dict[str, Any]]:
return records
def _normalize_record(result: dict[str, Any]) -> dict[str, Any]:
def normalize_performance_result(result: dict[str, Any]) -> dict[str, Any]:
"""Normalize a raw perf_*.json result into the HF tracking schema.
The Buildkite artifact intentionally keeps the raw benchmark output from
test_inference_performance.py. Baseline comparison, main-branch persistence,
and manual baseline reseeds should all use this mapping so the stored HF
records do not drift from the artifact schema.
"""
benchmark_id = result.get("benchmark_id", "unknown")
model_id = benchmark_id
@@ -66,6 +88,9 @@ def _normalize_record(result: dict[str, Any]) -> dict[str, Any]:
latency = safe_float(result.get("avg_generation_time_s"))
throughput = safe_float(result.get("throughput_fps"))
memory = safe_float(result.get("max_peak_memory_mb"))
text_encoder_time = safe_float(result.get("text_encoder_time_s"))
dit_time = safe_float(result.get("dit_time_s"))
vae_decode_time = safe_float(result.get("vae_decode_time_s"))
return {
"model_id": model_id,
@@ -75,10 +100,17 @@ def _normalize_record(result: dict[str, Any]) -> dict[str, Any]:
"latency": latency,
"throughput": throughput,
"memory": memory,
"text_encoder_time_s": text_encoder_time,
"dit_time_s": dit_time,
"vae_decode_time_s": vae_decode_time,
"success": True,
}
def _normalize_record(result: dict[str, Any]) -> dict[str, Any]:
return normalize_performance_result(result)
def _write_tracking_record(record: dict[str, Any]) -> str:
model_dir = os.path.join(TRACKING_ROOT, sanitize(record["model_id"]))
os.makedirs(model_dir, exist_ok=True)
@@ -93,6 +125,24 @@ def _write_tracking_record(record: dict[str, Any]) -> str:
return out_path
def _write_normalized_artifact(record: dict[str, Any]) -> None:
try:
results_dir = os.path.join(PERF_REPORTS_DIR, "results")
os.makedirs(results_dir, exist_ok=True)
timestamp = sanitize(record["timestamp"])
model_id = sanitize(record["model_id"])
commit = sanitize(record["commit_sha"] or "unknown")
path = os.path.join(
results_dir,
f"normalized_perf_{model_id}_{timestamp}_{commit}.json",
)
with open(path, "w", encoding="utf-8") as f:
json.dump(record, f, indent=2)
print(f"Normalized performance result written to {path}")
except Exception as e:
print(f"Failed to write normalized performance result artifact: {e}")
def _baseline_metric(records: list[dict[str, Any]], key: str) -> float | None:
values = [safe_float(r.get(key)) for r in records]
values = [v for v in values if v is not None]
@@ -108,7 +158,9 @@ def _check_regressions(
) -> list[str]:
failures: list[str] = []
for metric in ("latency", "memory"):
for metric, _label, _precision in METRICS:
if metric not in LOWER_IS_BETTER_METRICS:
continue
baseline = _baseline_metric(baseline_records, metric)
curr = safe_float(current.get(metric))
if baseline is None or curr is None or baseline <= 0:
@@ -141,7 +193,7 @@ def _metric_delta_percent(
if curr is None or baseline is None or baseline <= 0:
return None
if metric in ("latency", "memory"):
if metric in LOWER_IS_BETTER_METRICS:
return (curr - baseline) / baseline * 100.0
if metric == "throughput":
return (baseline - curr) / baseline * 100.0
@@ -161,28 +213,27 @@ def _build_summary_row(
) -> dict[str, Any]:
"""Format a single benchmark result as a row for the Markdown table."""
latency_base = _baseline_metric(baseline_records, "latency")
throughput_base = _baseline_metric(baseline_records, "throughput")
memory_base = _baseline_metric(baseline_records, "memory")
metric_values: dict[str, dict[str, float | None]] = {}
regressions: list[float] = []
for metric, _label, _precision in METRICS:
curr = safe_float(record.get(metric))
baseline = _baseline_metric(baseline_records, metric)
regression = _metric_delta_percent(metric, record, baseline_records)
metric_values[metric] = {
"curr": curr,
"base": baseline,
"regression_pct": regression,
}
if regression is not None:
regressions.append(regression)
# Calculate percentages for the 'Worst Regression' column
latency_reg = _metric_delta_percent("latency", record, baseline_records)
throughput_reg = _metric_delta_percent("throughput", record, baseline_records)
memory_reg = _metric_delta_percent("memory", record, baseline_records)
regressions = [v for v in (latency_reg, throughput_reg, memory_reg) if v is not None]
worst_regression_pct = max(regressions) if regressions else None
return {
"model_id": record["model_id"],
"gpu_type": record["gpu_type"],
"baseline_n": len(baseline_records),
"latency_curr": safe_float(record.get("latency")),
"latency_base": latency_base,
"throughput_curr": safe_float(record.get("throughput")),
"throughput_base": throughput_base,
"memory_curr": safe_float(record.get("memory")),
"memory_base": memory_base,
"metrics": metric_values,
"worst_regression_pct": worst_regression_pct,
"failed": has_failed,
}
@@ -199,24 +250,24 @@ def _build_markdown_summary(
"",
("| Model | GPU | Baseline N | Latency (curr/base) | "
"Throughput (curr/base) | Memory (curr/base) | "
"Worst Regression | Status |"),
"|---|---|---:|---|---|---|---:|---|",
"Text Enc (curr/base) | DiT (curr/base) | "
"VAE Decode (curr/base) | Worst Regression | Status |"),
"|---|---|---:|---|---|---|---|---|---|---:|---|",
]
for row in summary_rows:
latency = (f"{_compact_value(row['latency_curr'])} / "
f"{_compact_value(row['latency_base'])}")
throughput = (f"{_compact_value(row['throughput_curr'])} / "
f"{_compact_value(row['throughput_base'])}")
memory = (f"{_compact_value(row['memory_curr'], 1)} / "
f"{_compact_value(row['memory_base'], 1)}")
metric_cells = []
for metric, _label, precision in METRICS:
values = row["metrics"][metric]
metric_cells.append(f"{_compact_value(values['curr'], precision)} / "
f"{_compact_value(values['base'], precision)}")
worst_reg = ("n/a" if row["worst_regression_pct"] is None else f"{row['worst_regression_pct']:.1f}%")
status = "FAIL" if row["failed"] else "PASS"
lines.append(f"| {row['model_id']} | {row['gpu_type']} | "
f"{row['baseline_n']} | "
f"{latency} | {throughput} | {memory} | "
f"{' | '.join(metric_cells)} | "
f"{worst_reg} | {status} |")
return "\n".join(lines) + "\n"
@@ -233,11 +284,10 @@ def _emit_markdown_summary(markdown: str, commit_sha: str) -> None:
# 2. Write to Modal volume for Buildkite to pick up in post-run hook
try:
perf_reports_dir = "/root/data/perf_reports"
os.makedirs(perf_reports_dir, exist_ok=True)
os.makedirs(PERF_REPORTS_DIR, exist_ok=True)
short_sha = commit_sha[:7] if commit_sha else "unknown"
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
report_path = os.path.join(perf_reports_dir, f"perf_{short_sha}_{timestamp}.md")
report_path = os.path.join(PERF_REPORTS_DIR, f"perf_{short_sha}_{timestamp}.md")
with open(report_path, "w", encoding="utf-8") as f:
f.write(markdown + "\n")
print(f"Performance report written to {report_path}")
@@ -286,6 +336,8 @@ def main() -> int:
record["success"] = not failures
all_failures.extend(failures)
_write_normalized_artifact(record)
# Strict upload: a silent failure would freeze the rolling baseline.
if persist_tracking:
current_path = _write_tracking_record(record)
+85 -10
View File
@@ -1,5 +1,6 @@
# SPDX-License-Identifier: Apache-2.0
import os
from html import escape
from datetime import datetime
import plotly.express as px
@@ -7,6 +8,15 @@ import pandas as pd
from hf_store import sync_from_hf, load_as_dataframe
METRICS = (
"latency",
"throughput",
"memory",
"text_encoder_time_s",
"dit_time_s",
"vae_decode_time_s",
)
# -----------------------------
# 1. Grouping
# -----------------------------
@@ -19,15 +29,36 @@ def group_data(df: pd.DataFrame):
# -----------------------------
# 2. Plot builder
# -----------------------------
def build_plots(df: pd.DataFrame) -> list:
def build_plots(df: pd.DataFrame) -> tuple[list, list[dict[str, object]]]:
figs = []
skipped_metrics: list[dict[str, object]] = []
for (model_id, gpu_type), g in group_data(df):
g = g.sort_values("timestamp")
# One chart per metric so the y-axes aren't on wildly different scales
for metric in ("latency", "throughput", "memory"):
if g[metric].isna().all():
for metric in METRICS:
if metric not in g.columns:
skipped_metrics.append({
"model_id": model_id,
"gpu_type": gpu_type,
"metric": metric,
"reason": "column missing from loaded records",
"records": len(g),
"non_null": 0,
})
continue
non_null = int(g[metric].notna().sum())
if non_null == 0:
skipped_metrics.append({
"model_id": model_id,
"gpu_type": gpu_type,
"metric": metric,
"reason": "no non-null values in loaded records",
"records": len(g),
"non_null": non_null,
})
continue
fig = px.line(
@@ -41,20 +72,57 @@ def build_plots(df: pd.DataFrame) -> list:
)
figs.append(fig)
return figs
return figs, skipped_metrics
def render_skipped_metrics(skipped_metrics: list[dict[str, object]]) -> str:
if not skipped_metrics:
return ""
rows = [
"<h3>Skipped Metric Plots</h3>",
"<table>",
("<thead><tr><th>Model</th><th>GPU</th><th>Metric</th>"
"<th>Records</th><th>Non-null</th><th>Reason</th></tr></thead>"),
"<tbody>",
]
for item in skipped_metrics:
rows.append(
"<tr>"
f"<td>{escape(str(item['model_id']))}</td>"
f"<td>{escape(str(item['gpu_type']))}</td>"
f"<td>{escape(str(item['metric']))}</td>"
f"<td>{item['records']}</td>"
f"<td>{item['non_null']}</td>"
f"<td>{escape(str(item['reason']))}</td>"
"</tr>"
)
rows.extend(["</tbody>", "</table>"])
return "\n".join(rows)
# -----------------------------
# 3. Render HTML dashboard
# -----------------------------
def render_html(figs: list, days: int) -> str:
def render_html(figs: list, skipped_metrics: list[dict[str, object]],
days: int) -> str:
html_parts = [
"<html>",
"<head><meta charset='utf-8'>",
"<style>body { font-family: sans-serif; margin: 2rem; }</style>",
("<style>"
"body { font-family: sans-serif; margin: 2rem; }"
"table { border-collapse: collapse; margin: 1rem 0 2rem; }"
"th, td { border: 1px solid #ddd; padding: 0.4rem 0.6rem; "
"text-align: left; }"
"th { background: #f5f5f5; }"
"</style>"),
"</head><body>",
f"<h2>Performance Dashboard (last {days} days)</h2>",
]
skipped_html = render_skipped_metrics(skipped_metrics)
if skipped_html:
html_parts.append(skipped_html)
for fig in figs:
html_parts.append(fig.to_html(full_html=False, include_plotlyjs="cdn"))
@@ -67,7 +135,7 @@ def render_html(figs: list, days: int) -> str:
def main() -> None:
days = int(os.environ.get("DASHBOARD_DAYS", "30"))
local_dir = sync_from_hf("/tmp/perf-tracking")
local_dir = sync_from_hf("/tmp/perf-tracking", reuse_existing=True)
df = load_as_dataframe(local_dir, days=days)
if df.empty:
@@ -79,8 +147,15 @@ def main() -> None:
f"{df['gpu_type'].nunique()} GPU type(s), "
f"date range: {df['timestamp'].min()} → {df['timestamp'].max()}")
figs = build_plots(df)
html = render_html(figs, days)
figs, skipped_metrics = build_plots(df)
if skipped_metrics:
print("Skipped metric plots:")
for item in skipped_metrics:
print(f" - {item['model_id']} | {item['gpu_type']} | "
f"{item['metric']}: {item['reason']} "
f"({item['non_null']}/{item['records']} non-null)")
print(f"Generated {len(figs)} metric plot(s)")
html = render_html(figs, skipped_metrics, days)
commit_sha = os.environ.get("BUILDKITE_COMMIT", "unknown")[:7]
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
@@ -97,4 +172,4 @@ def main() -> None:
print(f"Dashboard generated: {output_file}")
if __name__ == "__main__":
main()
main()
+58 -3
View File
@@ -25,6 +25,8 @@ from huggingface_hub import HfApi, snapshot_download
HF_REPO_ID: str = os.environ.get("HF_REPO_ID", "FastVideo/performance-tracking")
HF_TOKEN: str | None = os.environ.get("HF_API_KEY")
SYNC_MARKER = ".hf_sync_complete"
SYNC_REUSE_TTL_SECONDS = int(os.environ.get("PERFORMANCE_TRACKING_SYNC_REUSE_TTL_SECONDS", "3600"))
# ---------------------------------------------------------------------------
# Low-level helpers
@@ -51,7 +53,33 @@ def safe_float(value: Any) -> float | None:
# ---------------------------------------------------------------------------
def sync_from_hf(local_dir: str, *, strict: bool = False) -> str:
def _sync_marker_path(local_dir: str) -> str:
return os.path.join(local_dir, SYNC_MARKER)
def _sync_marker_is_fresh(marker_path: str) -> bool:
try:
with open(marker_path, encoding="utf-8") as marker:
marker_data = json.load(marker)
synced_at_raw = marker_data.get("synced_at")
if not synced_at_raw:
return False
synced_at = datetime.fromisoformat(synced_at_raw)
if synced_at.tzinfo is None:
synced_at = synced_at.replace(tzinfo=timezone.utc)
except (OSError, json.JSONDecodeError, ValueError, TypeError):
return False
age = datetime.now(timezone.utc) - synced_at
return age.total_seconds() <= SYNC_REUSE_TTL_SECONDS
def sync_from_hf(
local_dir: str,
*,
strict: bool = False,
reuse_existing: bool = False,
) -> str:
"""Download the HF dataset repo snapshot to *local_dir*.
Returns *local_dir* so callers can chain: ``load_records(sync_from_hf(...))``.
@@ -61,7 +89,24 @@ def sync_from_hf(local_dir: str, *, strict: bool = False) -> str:
unavailable. Callers that depend on the sync for correctness (e.g. the
main-branch baseline writer) must pass ``strict=True`` so that misconfig
or transient HF errors fail loud rather than silently reset the baseline.
When ``reuse_existing=True``, a previous successful sync in ``local_dir``
is reused only while its marker is fresh. This avoids duplicate HF
snapshot checks when compare and dashboard scripts run sequentially in the
same CI job, without silently reusing stale data in persistent local or
long-lived runner environments.
"""
marker_path = _sync_marker_path(local_dir)
if reuse_existing and os.path.exists(marker_path):
if _sync_marker_is_fresh(marker_path):
print(f"hf_store: reusing existing sync at {local_dir}")
return local_dir
os.remove(marker_path)
print(f"hf_store: existing sync at {local_dir} is stale; refreshing")
if not reuse_existing and os.path.exists(marker_path):
os.remove(marker_path)
if not HF_REPO_ID:
msg = "hf_store: HF_REPO_ID not set"
if strict:
@@ -78,6 +123,12 @@ def sync_from_hf(local_dir: str, *, strict: bool = False) -> str:
token=HF_TOKEN,
allow_patterns="*.json",
)
os.makedirs(local_dir, exist_ok=True)
with open(marker_path, "w", encoding="utf-8") as marker:
json.dump({
"repo_id": HF_REPO_ID,
"synced_at": datetime.now(timezone.utc).isoformat(),
}, marker)
except Exception as exc:
if strict:
raise
@@ -223,14 +274,18 @@ def load_records_for_model(
# DataFrame helpers (dashboard / analytics consumers)
# ---------------------------------------------------------------------------
_NUMERIC_COLS = ("latency", "throughput", "memory")
_NUMERIC_COLS = (
"latency", "throughput", "memory",
"text_encoder_time_s", "dit_time_s", "vae_decode_time_s",
)
def normalize_dataframe(df: pd.DataFrame) -> pd.DataFrame:
"""Apply standard type coercions to a raw records DataFrame.
- Parses ``timestamp`` to UTC-aware datetime.
- Coerces ``latency``, ``throughput``, ``memory`` to float.
- Coerces ``latency``, ``throughput``, ``memory``, ``text_encoder_time_s``,
``dit_time_s``, ``vae_decode_time_s`` to float.
- Adds a ``config_id`` column (first 7 chars of ``commit_sha``).
Returns the mutated DataFrame (also modifies in place for efficiency).
@@ -0,0 +1,429 @@
# FastVideo Scheduled Main Performance Regression Tracking Redesign
## Purpose
FastVideo needs a scheduled main-branch performance benchmark that detects real inference regressions without confusing intentional optimizations, runtime upgrades, hardware changes, or quality-changing recipe changes with ordinary code regressions.
The current design compares recent results by `(benchmark_id, gpu_type)` and mainly tracks end-to-end generation latency. That is too coarse. A PyTorch upgrade, attention backend change, fewer inference steps, different precision, or different parallelism setup can enter the same baseline even though the result is not comparable.
This redesign makes comparability explicit.
## Core Principle
A benchmark result is comparable only when all of these match:
- workload identity
- variant identity
- benchmark schema/version
- inference recipe fingerprint
- hardware profile
- software/runtime profile
If any required identity does not match, the system reports `CALIBRATION_NEEDED` or `RECIPE_MISMATCH`, not a regression.
## Result Statuses
Each scheduled run should produce one of these statuses:
- `PASS`: comparable baseline exists and no metric regressed beyond threshold.
- `REGRESSION`: comparable baseline exists and one or more metrics regressed.
- `CALIBRATION_NEEDED`: no comparable baseline exists for this hardware/software cohort.
- `RECIPE_MISMATCH`: same variant was used with a different recipe fingerprint.
- `INFRA_ERROR`: benchmark failed for infrastructure reasons.
- `QUALITY_BLOCKED`: faster/new variant exists but lacks required quality validation.
## Identity Model
Use this comparison key:
```text
workload_id + variant_id + benchmark_version + hardware_profile_id + software_profile_id + recipe_fingerprint
```
Definitions:
- `workload_id`: stable benchmark family, for example `wan-t2v-1.3b`.
- `variant_id`: intentional recipe family, for example `canonical`, `fast-4step`, `flash-attn`, `torch-sdpa`.
- `benchmark_version`: version of the benchmark schema and comparison policy.
- `recipe_fingerprint`: hash of inference settings that affect comparability.
- `hardware_profile_id`: normalized GPU/hardware cohort.
- `software_profile_id`: intentionally versioned runtime cohort used for comparison.
- `environment_fingerprint`: full audit hash, stored but not used as the primary comparison key.
## Recipe Fingerprint
The `recipe_fingerprint` should include:
- model path and model revision
- pipeline or preset name/version
- prompt set digest
- negative prompt digest
- height, width, frame count, fps
- seed
- number of inference steps
- scheduler settings
- guidance scale and embedded CFG scale
- attention backend
- SP/TP size and number of GPUs
- text encoder, DiT, and VAE precision
- VAE tiling/SP/offload settings
- output type
- any benchmark-specific overrides
If the same `variant_id` produces a different recipe fingerprint, the system should not compare it to the existing baseline. It should report `RECIPE_MISMATCH`.
## Runtime And Software Cohorts
A PyTorch, CUDA, Triton, FlashAttention, container, or driver upgrade can change performance without a FastVideo code regression. These must be handled as software cohort changes.
Use two runtime fields:
- `software_profile_id`: compare-affecting runtime cohort.
- `environment_fingerprint`: full audit trail.
`software_profile_id` should include only major performance-affecting runtime fields:
- Python version
- PyTorch version
- CUDA runtime version
- Triton version
- FlashAttention/SageAttention/xFormers versions when used
- container performance profile version
- optional CUDA driver major/minor if runner stability requires it
`environment_fingerprint` can include:
- full container image digest
- all package versions
- driver details
- OS/kernel details
- dependency lock hash
- full environment metadata
A new `software_profile_id` should produce `CALIBRATION_NEEDED` until maintainers manually seed or promote reviewed records through the reseed-performance workflow.
## Runtime Upgrade Workflow
For PyTorch/CUDA/container upgrades:
- Create a new `software_profile_id`.
- Do not compare new runtime results against old runtime baselines.
- Run a bridge report when possible:
same FastVideo commit, same benchmark recipe, same hardware, old runtime vs new runtime.
- Use the bridge report to attribute performance shifts to the runtime upgrade.
- Seed the new runtime cohort only after the shift is reviewed, using the manual reseed-performance workflow.
## Hardware Profile
`hardware_profile_id` should normalize:
- GPU name
- GPU count
- GPU memory size
- relevant interconnect/topology when available
- Modal/runner machine class if stable and meaningful
For multi-GPU benchmarks, collect rank-level metrics and compare rank-max values where appropriate.
## Measurement Model
Keep end-to-end latency, but do not make it the only primary regression signal.
Primary metrics:
- `pipeline_total_s`
- `text_encode_s`
- `denoise_total_s`
- `denoise_step_mean_s`
- `denoise_step_p50_s`
- `denoise_step_p90_s`
- `transformer_forward_mean_s`
- `scheduler_step_mean_s`
- `vae_decode_s`
- `postprocess_cpu_s`
- `peak_memory_mb`
- `throughput_fps`
Informational metrics:
- `request_total_s`
- video write/export time
- generated artifact size
- individual run timings
For canonical inference compute benchmarks, prefer `save_video=false` to avoid disk and video encoding noise. Video writing/export can be tracked in a separate benchmark if needed.
## Instrumentation
Use existing stage logging as the coarse stage timing source.
Extend instrumentation to capture:
- text encoding time
- denoising total time
- denoising per-step timings
- transformer forward timings
- scheduler step timings
- VAE decode time
- CPU postprocess time
- peak memory per measurement run
- rank-level memory/timing for distributed runs
Avoid adding heavy synchronization inside every denoising sub-step by default. Prefer CUDA events or a dedicated low-overhead perf instrumentation path. Coarse stage timing may use synchronization, but per-step timing should not distort the scheduled benchmark.
Reset CUDA peak memory stats before each measured run on every rank.
## Baseline Policy
For each exact comparable identity:
- Use scheduled `main` records only.
- Persist only from scheduled main runs.
- Keep failed records for audit.
- Exclude failed records from baseline medians.
- Use the last 5 successful records as the rolling baseline.
- Compare current median against rolling median.
- Treat `CALIBRATION_NEEDED` records as audit data only. They must not automatically seed or advance a comparable baseline.
- Initialize new comparable baselines manually from reviewed scheduled-main artifacts through the reseed-performance workflow.
To prevent slow drift, also keep a promoted baseline:
- `rolling_baseline`: last 5 successful scheduled records.
- `promoted_baseline`: explicitly accepted reference cohort/version.
A run should alert or fail if it exceeds thresholds against either baseline.
## Threshold Policy
Use metric-specific thresholds instead of one global threshold.
Suggested defaults:
- 5% for core compute metrics: denoise total, denoise per-step, transformer forward, VAE decode.
- 5% for peak memory.
- 8-10% for noisy end-to-end/request metrics.
- Minimum absolute delta floor:
- 250ms for stage totals.
- 50ms for per-step metrics.
- reasonable MB floor for memory.
A regression should require both:
```text
percent_delta > threshold_percent
absolute_delta > threshold_absolute
```
This avoids failing on tiny noisy changes.
## Optimization And Variant Workflow
Same-recipe code optimizations stay in the same variant.
Examples:
- Faster implementation of the same attention backend: same variant.
- Reduced inference steps: new variant.
- Different default attention backend: new variant.
- Different precision recipe: new variant.
- DMD/distilled recipe: new variant.
A faster quality-affecting recipe should be created as a candidate variant:
```text
workload_id = wan-t2v-1.3b
variant_id = fast-4step
status = candidate
```
Candidate lifecycle:
```text
candidate -> calibrating -> canonical
candidate -> deprecated
```
Promotion requires:
- quality validation
- stable scheduled performance records
- explicit maintainer approval
## Quality Policy
Performance tracking must not decide "same quality" by latency alone.
Candidate variants require quality evidence, such as:
- SSIM/latent similarity reference
- existing evaluation metric
- reviewed generated artifacts
- documented human approval when automated quality metrics are unavailable
If quality validation is missing, the benchmark should report `QUALITY_BLOCKED` for promotion, even if performance improved.
## Data Schema
Each raw result should include:
```text
result_schema_version
workload_id
variant_id
benchmark_version
recipe_fingerprint
hardware_profile_id
software_profile_id
environment_fingerprint
trigger_type
schedule_id
build_id
build_url
branch
commit_sha
timestamp
status
quality_status
recipe
hardware
software
metrics
runs
comparison
```
Do not make PR fields required. This system is for scheduled main benchmarks.
Optional ad-hoc provenance can be stored as:
```text
change_request.type
change_request.number
```
but it should not be part of the scheduled-main schema requirements.
## Storage Layout
Change HF storage from:
```text
<model_id>/<record>.json
```
to:
```text
<workload_id>/<variant_id>/<hardware_profile_id>/<software_profile_id>/<record>.json
```
Store full result JSONs under this path.
Keep legacy v1 records readable for dashboard history, but do not compare v2 records against v1 records unless an explicit migration script creates compatible v2 identities.
## Dashboard Requirements
Dashboard should group by:
- workload
- variant
- hardware profile
- software profile
- benchmark version
Dashboard should show:
- current status
- rolling baseline comparison
- promoted baseline comparison
- recipe fingerprint
- software profile
- environment fingerprint
- quality status
- metric trend charts
- calibration-needed cohorts
- recipe mismatch incidents
- runtime upgrade bridge reports
Legacy v1 data can be shown separately as historical context.
## Scheduled Main Behavior
Scheduled main runs should:
- run every configured interval
- execute benchmarks on `main`
- write raw result artifacts
- compare against exact comparable baselines
- persist successful, failed, calibration, and mismatch records for audit
- update rolling baseline only with successful comparable records
- never silently initialize a baseline as passing
When no baseline exists, status is `CALIBRATION_NEEDED`. That record is persisted for review, but it does not become a baseline seed unless a maintainer explicitly reseeds or promotes it.
## Implementation Plan
1. Update benchmark JSON configs to include:
`workload_id`, `variant_id`, `benchmark_version`, comparison policy, quality policy, and recipe fields.
2. Update `test_inference_performance.py` to:
emit v2 schema, collect runtime metadata, compute recipe/hardware/software IDs, collect median metrics, reset memory per run, and avoid PR-required fields.
3. Extend pipeline instrumentation to:
capture stage metrics and denoising internals with low overhead.
4. Update distributed executor paths to:
return rank-level timing and memory metrics for multi-GPU runs.
5. Update `compare_baseline.py` to:
load by the new identity, reject recipe mismatches, report calibration for missing cohorts, compare rolling and promoted baselines, apply metric-specific thresholds, and avoid treating calibration records as baseline seeds.
6. Update `hf_store.py` to:
support the new storage layout and preserve legacy reads.
7. Update `dashboard.py` to:
group by workload/variant/hardware/software and show status, quality, and runtime cohort changes.
8. Update the reseed workflow to:
reseed only one exact identity tuple, require provenance especially for runtime upgrades, and mark reviewed records as accepted baseline seeds.
9. Add unit and integration tests for:
identity, recipe mismatch, software cohort changes, regression detection, dashboard grouping, and v2 result serialization.
## Migration Plan
- Treat current records as legacy schema v1.
- Start v2 with the current Wan benchmark:
- `workload_id = wan-t2v-1.3b`
- `variant_id = canonical`
- `benchmark_version = 1`
- Seed v2 baselines manually from reviewed scheduled-main result artifacts through the reseed-performance workflow.
- Keep legacy charts for reference only.
- Do not compare v2 records against v1 records by default.
## Test Plan
Required tests:
- Same config produces stable recipe fingerprint.
- Changed `num_inference_steps` under same variant produces `RECIPE_MISMATCH`.
- Changed PyTorch/software profile produces `CALIBRATION_NEEDED`.
- Same recipe with slower denoise metric produces `REGRESSION`.
- Same recipe with faster metrics produces `PASS`.
- Candidate variant does not compare against canonical.
- Missing quality evidence blocks candidate promotion.
- Dashboard separates software cohorts and variants.
- Multi-GPU metrics use rank-level aggregation.
- Current Wan config writes valid v2 JSON with populated metrics.
## Defaults
- HF dataset remains `FastVideo/performance-tracking`.
- Only scheduled main runs persist canonical tracking records.
- Missing exact baseline means `CALIBRATION_NEEDED`.
- Calibration records are not baseline seeds by default.
- New cohorts become comparable only after manual reseed or promotion.
- Runtime upgrades create new software cohorts.
- Fewer-step optimizations create new variants.
- End-to-end latency remains visible but is not the sole regression signal.
- Video export timing is separate from canonical inference compute timing.
@@ -0,0 +1,575 @@
# FastVideo Multi-Mode Performance Regression Tracking Redesign
## Purpose
FastVideo needs a performance benchmark system that detects real inference regressions without confusing intentional optimizations, runtime upgrades, hardware changes, local developer environments, or quality-changing recipe changes with ordinary code regressions.
The current design compares recent results by `(benchmark_id, gpu_type)` and mainly tracks end-to-end generation latency. That is too coarse. A PyTorch upgrade, attention backend change, fewer inference steps, different precision, different GPU, or different parallelism setup can enter the same baseline even though the result is not comparable.
This redesign makes comparability explicit and supports three execution modes:
- scheduled main benchmarks
- pull-request CI gating
- local developer runs
## Core Principle
A benchmark result is comparable only when all of these match:
- workload identity
- variant identity
- benchmark schema/version
- inference recipe fingerprint
- hardware profile
- software/runtime profile
If any required identity does not match, the system reports `CALIBRATION_NEEDED` or `RECIPE_MISMATCH`, not a misleading regression.
## Result Statuses
Each performance comparison can produce one of these statuses:
- `PASS`: comparable baseline exists and no metric regressed beyond threshold.
- `REGRESSION`: comparable baseline exists and one or more metrics regressed.
- `CALIBRATION_NEEDED`: no comparable baseline exists for this hardware/software cohort.
- `RECIPE_MISMATCH`: same variant was used with a different recipe fingerprint.
- `INFRA_ERROR`: benchmark failed for infrastructure reasons.
- `QUALITY_BLOCKED`: faster/new variant exists but lacks required quality validation.
## Execution Modes
The same benchmark configs and v2 result schema should support three modes.
### `scheduled_main`
Canonical scheduled run on `main`.
Behavior:
- Runs every configured interval.
- Executes benchmarks on `main`.
- Writes raw result artifacts.
- Compares against exact comparable accepted baselines.
- Persists successful, failed, calibration, and mismatch records for audit.
- Updates rolling baseline only with successful comparable accepted records.
- Never silently initializes a baseline as passing.
- Does not automatically seed new hardware/software cohorts.
- New cohorts become comparable only after manual reseed or promotion.
Failure policy:
- Static per-GPU threshold failure fails the build.
- Comparable rolling/promoted baseline regression fails the build.
- Missing exact baseline returns `CALIBRATION_NEEDED`; this should be visible and non-regression, but it requires maintainer action before the cohort is trusted.
### `pr_ci`
Pull-request gate.
Behavior:
- Runs the same benchmark configs as scheduled main.
- Compares against existing accepted exact-match baselines.
- Never uploads, seeds, promotes, or mutates baseline records.
- Emits raw artifacts, Markdown summary, and dashboard artifacts for review.
- Uses optional PR provenance only as metadata, never as baseline identity.
Failure policy:
- Static per-GPU threshold failure fails the PR gate.
- Comparable rolling/promoted baseline regression fails the PR gate.
- `RECIPE_MISMATCH` fails the PR gate because the PR changed benchmark comparability without declaring a new variant.
- `CALIBRATION_NEEDED` on standard CI hardware/software should fail or alert as CI configuration drift, because CI should have accepted baselines.
- `CALIBRATION_NEEDED` on an intentionally new benchmark variant should be treated as a review-required state, not as a successful regression pass.
### `local`
Developer diagnostic mode.
Behavior:
- Runs the same benchmark configs and emits the same v2 result schema.
- Never uploads, seeds, promotes, or mutates HF baselines.
- Compares only when an exact matching accepted baseline is available.
- Can use synced HF history or a local baseline directory override.
- Different GPU, PyTorch build, CUDA runtime, or attention library versions produce a different hardware/software profile and therefore do not compare against unrelated baselines.
Failure policy:
- Static threshold failures are reported clearly.
- Comparable baseline regressions are reported clearly.
- `CALIBRATION_NEEDED` is a non-misleading diagnostic, not proof of a project regression.
- Local output should include the local hardware/software identity so the developer can understand why comparison did or did not happen.
## Maintainer Surfaces
Every performance build should emit a Markdown summary.
The Markdown summary should include:
- execution mode
- overall status
- workload, variant, benchmark version
- recipe fingerprint
- hardware profile
- software profile
- static threshold result
- rolling baseline comparison
- promoted baseline comparison
- worst metric regressions
- calibration-needed cohorts
- recipe mismatch details
- reseed or promotion guidance when applicable
The Plotly dashboard remains the long-form surface for history and trend analysis.
The dashboard should group by:
- workload
- variant
- hardware profile
- software profile
- benchmark version
The dashboard should show:
- current status
- rolling baseline comparison
- promoted baseline comparison
- recipe fingerprint
- software profile
- environment fingerprint
- quality status
- metric trend charts
- calibration-needed cohorts
- recipe mismatch incidents
- runtime upgrade bridge reports
Legacy v1 data can be shown separately as historical context.
## Identity Model
Use this comparison key:
```text
workload_id + variant_id + benchmark_version + hardware_profile_id + software_profile_id + recipe_fingerprint
```
Definitions:
- `workload_id`: stable benchmark family, for example `wan-t2v-1.3b`.
- `variant_id`: intentional recipe family, for example `canonical`, `fast-4step`, `flash-attn`, `torch-sdpa`.
- `benchmark_version`: version of the benchmark schema and comparison policy.
- `recipe_fingerprint`: hash of inference settings that affect comparability.
- `hardware_profile_id`: normalized GPU/hardware cohort.
- `software_profile_id`: intentionally versioned runtime cohort used for comparison.
- `environment_fingerprint`: full audit hash, stored but not used as the primary comparison key.
## Recipe Fingerprint
The `recipe_fingerprint` should include:
- model path and model revision
- pipeline or preset name/version
- prompt set digest
- negative prompt digest
- height, width, frame count, fps
- seed
- number of inference steps
- scheduler settings
- guidance scale and embedded CFG scale
- attention backend
- SP/TP size and number of GPUs
- text encoder, DiT, and VAE precision
- VAE tiling/SP/offload settings
- output type
- any benchmark-specific overrides
If the same `variant_id` produces a different recipe fingerprint, the system should not compare it to the existing baseline. It should report `RECIPE_MISMATCH`.
## Runtime And Software Cohorts
A PyTorch, CUDA, Triton, FlashAttention, container, or driver upgrade can change performance without a FastVideo code regression. These must be handled as software cohort changes.
Use two runtime fields:
- `software_profile_id`: compare-affecting runtime cohort.
- `environment_fingerprint`: full audit trail.
`software_profile_id` should include only major performance-affecting runtime fields:
- Python version
- PyTorch version
- CUDA runtime version
- Triton version
- FlashAttention/SageAttention/xFormers versions when used
- container performance profile version
- optional CUDA driver major/minor if runner stability requires it
`environment_fingerprint` can include:
- full container image digest
- all package versions
- driver details
- OS/kernel details
- dependency lock hash
- full environment metadata
A new `software_profile_id` should produce `CALIBRATION_NEEDED` until maintainers manually seed or promote reviewed records through the reseed-performance workflow.
## Runtime Upgrade Workflow
For PyTorch/CUDA/container upgrades:
- Create a new `software_profile_id`.
- Do not compare new runtime results against old runtime baselines.
- Run a bridge report when possible:
same FastVideo commit, same benchmark recipe, same hardware, old runtime vs new runtime.
- Use the bridge report to attribute performance shifts to the runtime upgrade.
- Seed the new runtime cohort only after the shift is reviewed, using the manual reseed-performance workflow.
## Hardware Profile
`hardware_profile_id` should normalize:
- GPU name
- GPU count
- GPU memory size
- relevant interconnect/topology when available
- Modal/runner machine class if stable and meaningful
For multi-GPU benchmarks, collect rank-level metrics and compare rank-max values where appropriate.
## Measurement Model
Keep end-to-end latency, but do not make it the only primary regression signal.
Primary metrics:
- `pipeline_total_s`
- `text_encode_s`
- `denoise_total_s`
- `denoise_step_mean_s`
- `denoise_step_p50_s`
- `denoise_step_p90_s`
- `transformer_forward_mean_s`
- `scheduler_step_mean_s`
- `vae_decode_s`
- `postprocess_cpu_s`
- `peak_memory_mb`
- `throughput_fps`
Informational metrics:
- `request_total_s`
- video write/export time
- generated artifact size
- individual run timings
For canonical inference compute benchmarks, prefer `save_video=false` to avoid disk and video encoding noise. Video writing/export can be tracked in a separate benchmark if needed.
## Instrumentation
Use existing stage logging as the coarse stage timing source.
Extend instrumentation to capture:
- text encoding time
- denoising total time
- denoising per-step timings
- transformer forward timings
- scheduler step timings
- VAE decode time
- CPU postprocess time
- peak memory per measurement run
- rank-level memory/timing for distributed runs
Avoid adding heavy synchronization inside every denoising sub-step by default. Prefer CUDA events or a dedicated low-overhead perf instrumentation path. Coarse stage timing may use synchronization, but per-step timing should not distort the scheduled benchmark.
Reset CUDA peak memory stats before each measured run on every rank.
## Baseline Policy
For each exact comparable identity:
- Use accepted scheduled-main records as the canonical baseline source.
- PR CI and local runs may compare against accepted baselines but must never mutate them.
- Persist only from scheduled main runs.
- Keep failed records for audit.
- Exclude failed records from baseline medians.
- Use the last 5 successful accepted records as the rolling baseline.
- Compare current median against rolling median.
- Treat `CALIBRATION_NEEDED` records as audit data only. They must not automatically seed or advance a comparable baseline.
- Initialize new comparable baselines manually from reviewed scheduled-main artifacts through the reseed-performance workflow.
To prevent slow drift, also keep a promoted baseline:
- `rolling_baseline`: last 5 successful accepted scheduled records.
- `promoted_baseline`: explicitly accepted reference cohort/version.
A run should alert or fail according to execution mode if it exceeds thresholds against either baseline.
## Threshold Policy
Use metric-specific thresholds instead of one global threshold.
Suggested defaults:
- 5% for core compute metrics: denoise total, denoise per-step, transformer forward, VAE decode.
- 5% for peak memory.
- 8-10% for noisy end-to-end/request metrics.
- Minimum absolute delta floor:
- 250ms for stage totals.
- 50ms for per-step metrics.
- reasonable MB floor for memory.
A regression should require both:
```text
percent_delta > threshold_percent
absolute_delta > threshold_absolute
```
This avoids failing on tiny noisy changes.
## Optimization And Variant Workflow
Same-recipe code optimizations stay in the same variant.
Examples:
- Faster implementation of the same attention backend: same variant.
- Reduced inference steps: new variant.
- Different default attention backend: new variant.
- Different precision recipe: new variant.
- DMD/distilled recipe: new variant.
A faster quality-affecting recipe should be created as a candidate variant:
```text
workload_id = wan-t2v-1.3b
variant_id = fast-4step
status = candidate
```
Candidate lifecycle:
```text
candidate -> calibrating -> canonical
candidate -> deprecated
```
Promotion requires:
- quality validation
- stable scheduled performance records
- explicit maintainer approval
## Quality Policy
Performance tracking must not decide "same quality" by latency alone.
Candidate variants require quality evidence, such as:
- SSIM/latent similarity reference
- existing evaluation metric
- reviewed generated artifacts
- documented human approval when automated quality metrics are unavailable
If quality validation is missing, the benchmark should report `QUALITY_BLOCKED` for promotion, even if performance improved.
## Data Schema
Each raw result should include:
```text
result_schema_version
execution_mode
workload_id
variant_id
benchmark_version
recipe_fingerprint
hardware_profile_id
software_profile_id
environment_fingerprint
trigger_type
schedule_id
build_id
build_url
branch
commit_sha
timestamp
status
quality_status
recipe
hardware
software
metrics
runs
comparison
```
PR fields are optional metadata, not required identity fields.
Optional ad-hoc or PR provenance can be stored as:
```text
change_request.type
change_request.number
change_request.url
```
Local-only provenance can be stored as:
```text
local_run.user
local_run.hostname
local_run.baseline_source
local_run.upload_enabled
```
Local upload should default to disabled.
## Storage Layout
Change HF storage from:
```text
<model_id>/<record>.json
```
to:
```text
<workload_id>/<variant_id>/<hardware_profile_id>/<software_profile_id>/<record>.json
```
Store full result JSONs under this path.
Keep legacy v1 records readable for dashboard history, but do not compare v2 records against v1 records unless an explicit migration script creates compatible v2 identities.
## Scheduled Main Behavior
Scheduled main runs should:
- run every configured interval
- execute benchmarks on `main`
- write raw result artifacts
- compare against exact comparable baselines
- persist successful, failed, calibration, and mismatch records for audit
- update rolling baseline only with successful comparable accepted records
- never silently initialize a baseline as passing
When no baseline exists, status is `CALIBRATION_NEEDED`. That record is persisted for review, but it does not become a baseline seed unless a maintainer explicitly reseeds or promotes it.
## PR CI Behavior
PR CI runs should:
- execute the same benchmark configs as scheduled main
- compare against exact comparable accepted baselines
- enforce static per-GPU thresholds
- enforce rolling/promoted regression checks
- emit Markdown summary and dashboard artifacts
- never persist, seed, or promote records
PR CI should fail on:
- static threshold failure
- comparable regression
- recipe mismatch
- infrastructure error
- missing baseline for standard CI identity, unless explicitly marked as an expected new benchmark/variant calibration case
## Local Developer Behavior
Local runs should:
- execute the same benchmark configs
- emit the same v2 raw result schema
- compare only against exact matching accepted baselines
- never upload or seed baselines by default
- allow a local baseline directory override
- report different GPU/runtime profiles as `CALIBRATION_NEEDED`, not as regressions
- show environment details needed to understand why comparison did or did not occur
## Implementation Plan
1. Update benchmark JSON configs to include:
`workload_id`, `variant_id`, `benchmark_version`, comparison policy, quality policy, and recipe fields.
2. Add execution mode detection:
`scheduled_main`, `pr_ci`, and `local`.
3. Update `test_inference_performance.py` to:
emit v2 schema, collect runtime metadata, compute recipe/hardware/software IDs, collect median metrics, reset memory per run, and avoid PR-required fields.
4. Extend pipeline instrumentation to:
capture stage metrics and denoising internals with low overhead.
5. Update distributed executor paths to:
return rank-level timing and memory metrics for multi-GPU runs.
6. Update `compare_baseline.py` to:
load by the new identity, reject recipe mismatches, report calibration for missing cohorts, compare rolling and promoted baselines, apply metric-specific thresholds, and avoid treating calibration records as baseline seeds.
7. Update `hf_store.py` to:
support the new storage layout and preserve legacy reads.
8. Update `dashboard.py` to:
group by workload/variant/hardware/software and show status, quality, runtime cohort changes, and execution mode.
9. Update Markdown summary rendering to:
always emit a compact maintainer-facing summary for scheduled main, PR CI, and local diagnostic runs.
10. Update the reseed workflow to:
reseed only one exact identity tuple, require provenance especially for runtime upgrades, and mark reviewed records as accepted baseline seeds.
11. Add unit and integration tests for:
identity, recipe mismatch, software cohort changes, regression detection, execution mode behavior, dashboard grouping, summary rendering, and v2 result serialization.
## Migration Plan
- Treat current records as legacy schema v1.
- Start v2 with the current Wan benchmark:
- `workload_id = wan-t2v-1.3b`
- `variant_id = canonical`
- `benchmark_version = 1`
- Seed v2 baselines manually from reviewed scheduled-main result artifacts through the reseed-performance workflow.
- Keep legacy charts for reference only.
- Do not compare v2 records against v1 records by default.
## Test Plan
Required tests:
- Same config produces stable recipe fingerprint.
- Changed `num_inference_steps` under same variant produces `RECIPE_MISMATCH`.
- Changed PyTorch/software profile produces `CALIBRATION_NEEDED`.
- Same recipe with slower denoise metric produces `REGRESSION`.
- Same recipe with faster metrics produces `PASS`.
- Candidate variant does not compare against canonical.
- Missing quality evidence blocks candidate promotion.
- Dashboard separates software cohorts, variants, and execution modes.
- Multi-GPU metrics use rank-level aggregation.
- Current Wan config writes valid v2 JSON with populated metrics.
- PR CI exact-match regression fails.
- PR CI missing baseline for standard CI identity fails or alerts as configuration drift.
- Local missing baseline returns diagnostic `CALIBRATION_NEEDED`.
- Local different software profile does not compare against scheduled-main baseline.
- Scheduled main still does not auto-seed calibration records.
- Markdown summary renders useful output for scheduled main, PR CI, and local modes.
## Defaults
- HF dataset remains `FastVideo/performance-tracking`.
- Accepted scheduled-main records are the canonical baseline source.
- PR CI gates against accepted exact-match baselines but never mutates them.
- Local mode is diagnostic and upload-disabled by default.
- Missing exact baseline means `CALIBRATION_NEEDED`.
- Calibration records are not baseline seeds by default.
- New cohorts become comparable only after manual reseed or promotion.
- Runtime upgrades create new software cohorts.
- Fewer-step optimizations create new variants.
- End-to-end latency remains visible but is not the sole regression signal.
- Video export timing is separate from canonical inference compute timing.
@@ -10,6 +10,7 @@ import glob
import json
import os
import time
from collections.abc import Mapping
from datetime import datetime, timezone
import torch
@@ -21,6 +22,13 @@ from fastvideo.worker.multiproc_executor import MultiprocExecutor
logger = init_logger(__name__)
STAGE_METRIC_MAP: dict[str, str] = {
"TextEncodingStage": "text_encoder_time_s",
"DenoisingStage": "dit_time_s",
"DmdDenoisingStage": "dit_time_s",
"DecodingStage": "vae_decode_time_s",
}
# -- Config discovery -------------------------------------------------------
_BENCHMARKS_DIR = os.path.join(
@@ -73,16 +81,53 @@ def _shutdown_executor(generator):
generator.executor.shutdown()
def _extract_component_times(result: dict) -> dict[str, float | None]:
component_times: dict[str, float | None] = {
"text_encoder_time_s": None,
"dit_time_s": None,
"vae_decode_time_s": None,
}
logging_info = result.get("logging_info")
if logging_info is None:
return component_times
if isinstance(logging_info, Mapping):
stages: dict = logging_info.get("stages", {}) or {}
else:
stages: dict = getattr(logging_info, "stages", {}) or {}
if not stages:
return component_times
logger.info("Discovered pipeline stages: %s", list(stages.keys()))
for stage_name, stage_data in stages.items():
metric_key = STAGE_METRIC_MAP.get(stage_name)
if metric_key is None:
logger.debug("Unmapped stage '%s' (%.3fs)",
stage_name,
stage_data.get("execution_time", 0))
continue
elapsed = stage_data.get("execution_time")
if elapsed is None:
continue
existing = component_times[metric_key]
component_times[metric_key] = (
elapsed if existing is None else existing + elapsed)
return component_times
def _avg_component(all_component_times: list[dict], key: str) -> float | None:
vals = [r[key] for r in all_component_times if r.get(key) is not None]
return round(sum(vals) / len(vals), 3) if vals else None
def _run_generation(generator, prompt, generation_kwargs):
"""Run a single generation, return (elapsed_s, peak_memory_mb)."""
"""Run a single generation, return (elapsed_s, peak_memory_mb, component_times)."""
torch.cuda.synchronize()
start = time.perf_counter()
result = generator.generate_video(prompt, **generation_kwargs)
torch.cuda.synchronize()
elapsed = time.perf_counter() - start
peak_memory_mb = result.get("peak_memory_mb", 0.0) or 0.0
return elapsed, peak_memory_mb
component_times = _extract_component_times(result)
return elapsed, peak_memory_mb, component_times
def _write_results(results):
"""Write JSON results to the results directory."""
@@ -102,15 +147,7 @@ def _write_results(results):
# -- Test -------------------------------------------------------------------
@pytest.mark.parametrize(
"cfg",
_BENCHMARK_CONFIGS,
ids=[c["benchmark_id"] for c in _BENCHMARK_CONFIGS],
)
def test_inference_performance(cfg):
"""Measure generation latency and peak GPU memory,
assert against device-aware thresholds."""
def _run_benchmark(cfg):
run_config = cfg.get("run_config", {})
required_gpus = run_config.get("required_gpus", 1)
available = torch.cuda.device_count()
@@ -152,12 +189,18 @@ def test_inference_performance(cfg):
times = []
peak_memories = []
all_component_times = []
for i in range(num_measure):
logger.info("Measurement run %d/%d", i + 1, num_measure)
elapsed, peak_mb = _run_generation(generator, prompt, gen_kwargs)
elapsed, peak_mb, component_times = _run_generation(
generator,
prompt,
gen_kwargs,
)
logger.info(" Time: %.2fs, Peak memory: %.0fMB", elapsed, peak_mb)
times.append(elapsed)
peak_memories.append(peak_mb)
all_component_times.append(component_times)
finally:
_shutdown_executor(generator)
@@ -186,6 +229,11 @@ def test_inference_performance(cfg):
"commit": os.environ.get("BUILDKITE_COMMIT", ""),
"pr_number": os.environ.get("BUILDKITE_PULL_REQUEST", ""),
"timestamp": datetime.now(timezone.utc).isoformat(),
"text_encoder_time_s": _avg_component(all_component_times,
"text_encoder_time_s"),
"dit_time_s": _avg_component(all_component_times, "dit_time_s"),
"vae_decode_time_s": _avg_component(all_component_times,
"vae_decode_time_s"),
}
logger.info(
@@ -203,3 +251,38 @@ def test_inference_performance(cfg):
assert max_peak_memory <= max_mem, (
f"Peak memory {max_peak_memory:.0f}MB exceeds "
f"threshold {max_mem:.0f}MB for {device_name}")
component_thresholds = {
"text_encoder_time_s": thresholds.get("max_text_encoder_time_s"),
"dit_time_s": thresholds.get("max_dit_time_s"),
"vae_decode_time_s": thresholds.get("max_vae_decode_time_s"),
}
for metric, max_val in component_thresholds.items():
if max_val is None:
continue
actual = results[metric]
if actual is not None:
assert actual <= max_val, (
f"{metric} {actual:.3f}s exceeds threshold {max_val:.3f}s "
f"for {device_name}")
@pytest.mark.parametrize(
"cfg",
_BENCHMARK_CONFIGS,
ids=[c["benchmark_id"] for c in _BENCHMARK_CONFIGS],
)
def test_inference_performance(cfg):
"""Measure generation latency, peak GPU memory, and component-level timings
(text encoder, DiT, VAE decode). Assert each against device-aware thresholds.
"""
original_env = os.environ.get("FASTVIDEO_STAGE_LOGGING")
os.environ["FASTVIDEO_STAGE_LOGGING"] = "1"
try:
_run_benchmark(cfg)
finally:
if original_env is None:
os.environ.pop("FASTVIDEO_STAGE_LOGGING", None)
else:
os.environ["FASTVIDEO_STAGE_LOGGING"] = original_env