Compare commits
46
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c2e89f22d3 | ||
|
|
e47a3c5aad | ||
|
|
d40fbfc534 | ||
|
|
126a52ce32 | ||
|
|
419e1c68f1 | ||
|
|
938bc3c972 | ||
|
|
381a7aae66 | ||
|
|
7fa4fb50bb | ||
|
|
de3cd6aab2 | ||
|
|
a64e5e62ab | ||
|
|
f4200cc3f3 | ||
|
|
363230b372 | ||
|
|
1dcdacc35a | ||
|
|
0a7a4a21f6 | ||
|
|
c30d731992 | ||
|
|
f8eaff4292 | ||
|
|
76e7048b3d | ||
|
|
b1386a7a78 | ||
|
|
664d6b3c23 | ||
|
|
0688c5131a | ||
|
|
0b605d0a41 | ||
|
|
8f5ebe2aeb | ||
|
|
84803076a0 | ||
|
|
5cc337fdb6 | ||
|
|
2b8e5a56a8 | ||
|
|
e117167eb9 | ||
|
|
4f9af56b78 | ||
|
|
33efe7a673 | ||
|
|
b07cc3f9d4 | ||
|
|
29eb4109cc | ||
|
|
d07a7691fe | ||
|
|
960485519f | ||
|
|
412c95b1a0 | ||
|
|
f6275f8005 | ||
|
|
7b872cc41e | ||
|
|
37418946c8 | ||
|
|
95fd29e0cb | ||
|
|
e17cd2633c | ||
|
|
e0dc5f2b0c | ||
|
|
70ee5d230c | ||
|
|
24ced500f5 | ||
|
|
4ddcdf541f | ||
|
|
0e3529869c | ||
|
|
e1e0d91c00 | ||
|
|
145a3f166b | ||
|
|
88a5a933ab |
Executable
+96
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env bash
|
||||
# Sync .agents/skills/ into .claude/skills/ via per-skill symlinks.
|
||||
#
|
||||
# Why: Claude Code only scans .claude/skills/ and ~/.claude/skills/ for
|
||||
# user-invocable skills (no skillsPath config exists — see
|
||||
# https://code.claude.com/docs/en/skills.md). This repo's skills live
|
||||
# in .agents/skills/ so they travel with the repo and stay under git.
|
||||
# Run this once after cloning (or after adding/removing a skill) to
|
||||
# expose them to Claude Code without maintaining a parallel tree.
|
||||
#
|
||||
# Usage:
|
||||
# .agents/scripts/sync-skills.sh
|
||||
#
|
||||
# Idempotent and safe to re-run. Prunes stale symlinks whose source
|
||||
# has been removed from .agents/skills/. Leaves hand-written
|
||||
# .claude/skills/<name>/ directories untouched (only symlinks are
|
||||
# managed).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(git -C "$(dirname "$0")" rev-parse --show-toplevel)"
|
||||
SRC_DIR="$REPO_ROOT/.agents/skills"
|
||||
DST_DIR="$REPO_ROOT/.claude/skills"
|
||||
|
||||
if [[ ! -d "$SRC_DIR" ]]; then
|
||||
echo "Error: $SRC_DIR does not exist." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p "$DST_DIR"
|
||||
|
||||
linked=0
|
||||
unchanged=0
|
||||
skipped=0
|
||||
pruned=0
|
||||
|
||||
link_skill() {
|
||||
local name="$1"
|
||||
local src="$SRC_DIR/$name"
|
||||
local dst="$DST_DIR/$name"
|
||||
# Relative target keeps symlinks portable across clones.
|
||||
local rel="../../.agents/skills/$name"
|
||||
|
||||
if [[ -L "$dst" ]]; then
|
||||
if [[ "$(readlink "$dst")" == "$rel" ]]; then
|
||||
unchanged=$((unchanged + 1))
|
||||
return
|
||||
fi
|
||||
rm "$dst"
|
||||
elif [[ -e "$dst" ]]; then
|
||||
echo "Skipped (not a symlink): .claude/skills/$name" >&2
|
||||
skipped=$((skipped + 1))
|
||||
return
|
||||
fi
|
||||
|
||||
ln -s "$rel" "$dst"
|
||||
echo "Linked: .claude/skills/$name -> $rel"
|
||||
linked=$((linked + 1))
|
||||
}
|
||||
|
||||
prune_stale() {
|
||||
local link="$1"
|
||||
local target
|
||||
target="$(readlink "$link")"
|
||||
case "$target" in
|
||||
../../.agents/skills/*) ;;
|
||||
*) return ;;
|
||||
esac
|
||||
local name="${target##*/}"
|
||||
if [[ ! -d "$SRC_DIR/$name" ]]; then
|
||||
rm "$link"
|
||||
echo "Pruned stale: .claude/skills/$(basename "$link")"
|
||||
pruned=$((pruned + 1))
|
||||
fi
|
||||
}
|
||||
|
||||
for src in "$SRC_DIR"/*/; do
|
||||
[[ -d "$src" ]] || continue
|
||||
name="$(basename "$src")"
|
||||
# Only treat directories that actually contain a SKILL.md as skills.
|
||||
[[ -f "$src/SKILL.md" ]] || continue
|
||||
link_skill "$name"
|
||||
done
|
||||
|
||||
shopt -s nullglob
|
||||
for link in "$DST_DIR"/*; do
|
||||
[[ -L "$link" ]] || continue
|
||||
prune_stale "$link"
|
||||
done
|
||||
shopt -u nullglob
|
||||
|
||||
printf "\nSummary: %d linked, %d unchanged, %d pruned" "$linked" "$unchanged" "$pruned"
|
||||
if [[ "$skipped" -gt 0 ]]; then
|
||||
printf ", %d skipped (non-symlink collision)" "$skipped"
|
||||
fi
|
||||
printf "\n"
|
||||
@@ -5,3 +5,4 @@
|
||||
{"name": "evaluate-video-quality", "description": "Evaluate generated video quality using available metrics (SSIM, loss trajectory, caption consistency)", "path": "evaluate-video-quality/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "index-related-work", "description": "Ingest a paper or repository into the related work index", "path": "index-related-work/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "search-related-work", "description": "Query the related work index for relevant papers, repos, or comparisons", "path": "search-related-work/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "seed-ssim-references", "description": "Run a new or updated fastvideo/tests/ssim/ test on Modal, pull generated videos, and upload them to FastVideo/ssim-reference-videos so the test has a regression baseline", "path": "seed-ssim-references/SKILL.md", "status": "draft", "trust": "low"}
|
||||
|
||||
@@ -0,0 +1,250 @@
|
||||
---
|
||||
name: seed-ssim-references
|
||||
description: Seed HF reference videos for a single newly-added SSIM test. Runs the test on Modal L40S, downloads the generated mp4s via `modal volume get`, pauses for the user to eyeball quality, then uploads only that test's files to `FastVideo/ssim-reference-videos`. Use when a new `fastvideo/tests/ssim/test_*_similarity.py` has just been added and has no references on HF yet.
|
||||
---
|
||||
|
||||
# Seed SSIM Reference Videos
|
||||
|
||||
## Purpose
|
||||
|
||||
A brand-new SSIM test in `fastvideo/tests/ssim/` fails forever until its
|
||||
reference videos exist on the HF dataset (`FastVideo/ssim-reference-videos`).
|
||||
This skill:
|
||||
|
||||
1. Runs the test on Modal's L40S pool to generate the videos.
|
||||
2. Downloads them to the local repo via `modal volume get`.
|
||||
3. Pauses so the user can eyeball the mp4s and confirm quality.
|
||||
4. Uploads only the new test's files to HF, with a guard that refuses to
|
||||
overwrite anything already present.
|
||||
|
||||
The skill is run **manually**, once per new test. Before invoking it, the user
|
||||
has already sanity-tested the new test locally — it launches `VideoGenerator`
|
||||
and writes an mp4 without crashing. The skill does not re-test locally; it
|
||||
goes straight to Modal L40S (which is what CI uses).
|
||||
|
||||
## When to use
|
||||
|
||||
- A new `test_*_similarity.py` file has been added in `fastvideo/tests/ssim/`
|
||||
and the HF dataset has no `reference_videos/default/L40S_reference_videos/<model_id>/`
|
||||
subtree for it yet.
|
||||
|
||||
## When not to use
|
||||
|
||||
- Regular CI runs — once refs exist, `pytest fastvideo/tests/ssim/` downloads
|
||||
them automatically.
|
||||
- Re-seeding an existing test. That requires `--force` on the upload step, and
|
||||
is out of scope here; treat as a separate, deliberate operation.
|
||||
|
||||
## Inputs
|
||||
|
||||
The skill has **one required input**: the path to the new SSIM test file.
|
||||
Prompt the user for it if they didn't supply it.
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `test_file` | Yes | e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`. The skill's first action is to ask for this if missing. |
|
||||
|
||||
Everything else is fixed:
|
||||
|
||||
- Modal runner GPU: **L40S** (hardcoded in `fastvideo/tests/modal/ssim_test.py`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
- Quality tier: `default` (the tier CI runs). The `full_quality` tier is not
|
||||
seeded by this skill.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (dataset).
|
||||
- Multi-model test files: all model ids in `*_MODEL_TO_PARAMS` are seeded
|
||||
together; the Modal run produces one mp4 per (model, prompt, backend) and
|
||||
the upload scopes by `--model-id`, looping if there is more than one.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The user has confirmed:
|
||||
|
||||
- `modal` CLI authenticated.
|
||||
- `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`) exported with write
|
||||
access to `FastVideo/ssim-reference-videos`.
|
||||
- The test file runs locally end-to-end (generates an mp4; SSIM assertion
|
||||
failure due to missing reference is expected and fine).
|
||||
|
||||
Fail fast if the token env var is missing.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Ask for the test file
|
||||
|
||||
If the user didn't name one, ask: *"Which SSIM test file do you want to seed
|
||||
references for? (e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`)"*.
|
||||
|
||||
Validate:
|
||||
|
||||
- Path exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
|
||||
- File defines a `*_MODEL_TO_PARAMS` dict — grep it to extract the set of
|
||||
model ids. Those ids drive step 5.
|
||||
|
||||
If either check fails, stop and tell the user what's wrong.
|
||||
|
||||
### 2. Run the test on Modal L40S
|
||||
|
||||
Pick a subdir name so repeated runs don't collide:
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
|
||||
```
|
||||
|
||||
Then launch the Modal run:
|
||||
|
||||
```bash
|
||||
modal run fastvideo/tests/modal/ssim_test.py \
|
||||
--git-repo="$(git config --get remote.origin.url)" \
|
||||
--git-commit="$(git rev-parse HEAD)" \
|
||||
--hf-api-key="$HF_API_KEY" \
|
||||
--test-files="<test_file>" \
|
||||
--sync-generated-to-volume \
|
||||
--generated-volume-subdir="$SUBDIR" \
|
||||
--skip-reference-download \
|
||||
--no-fail-fast
|
||||
```
|
||||
|
||||
Flag rationale:
|
||||
- `--skip-reference-download`: no refs exist yet, so conftest must not try to
|
||||
pull them.
|
||||
- `--no-fail-fast`: lets the test finish generation before `_assert_similarity`
|
||||
raises `FileNotFoundError: Reference video folder does not exist`. The
|
||||
expected failure is what we want — the mp4 has already been written.
|
||||
- `--sync-generated-to-volume` + `--generated-volume-subdir`: copies the
|
||||
generated mp4s to the `hf-model-weights` Modal volume under
|
||||
`ssim_generated_videos/default/<SUBDIR>/generated_videos/` so we can pull
|
||||
them locally.
|
||||
|
||||
The Modal run will end with a nonzero exit (expected) and print a
|
||||
`modal volume get hf-model-weights ssim_generated_videos/default/<SUBDIR>/generated_videos ./generated_videos_modal/default`
|
||||
command. Capture that `<SUBDIR>` — you need it for step 3.
|
||||
|
||||
### 3. Download generated videos locally
|
||||
|
||||
```bash
|
||||
modal volume get --force hf-model-weights \
|
||||
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
|
||||
./generated_videos_modal/default
|
||||
```
|
||||
|
||||
`--force` is required when the parent `./generated_videos_modal/default`
|
||||
already exists; without it, `modal volume get` errors with `[Errno 21] Is a
|
||||
directory`. Safe to pass on the first run too.
|
||||
|
||||
After this, the mp4s live at
|
||||
`./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
The extra `generated_videos/` level comes from the volume layout in
|
||||
`_sync_generated_videos_to_volume` (`ssim_test.py`) — the command copies
|
||||
`<repo>/fastvideo/tests/ssim/generated_videos/<tier>` to
|
||||
`ssim_generated_videos/<tier>/<SUBDIR>/generated_videos/`, and `modal volume
|
||||
get` preserves that trailing `generated_videos/` segment.
|
||||
|
||||
### 4. PAUSE — user reviews quality
|
||||
|
||||
Print the list of downloaded mp4s and their paths, then stop. Tell the user:
|
||||
|
||||
> "Generated videos downloaded to `./generated_videos_modal/default/generated_videos/L40S_reference_videos/`. Please open them and confirm the quality looks correct. Reply **`upload`** to continue, or anything else to abort."
|
||||
|
||||
Do not proceed until the user explicitly says `upload`. If they abort, leave
|
||||
everything on disk so they can inspect further — no cleanup.
|
||||
|
||||
### 5. Copy into the local reference layout
|
||||
|
||||
Scoped copy — only the new test's mp4s. Loop over each `<model_id>` extracted
|
||||
in step 1:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
|
||||
```
|
||||
|
||||
(The `--generated-dir` points at the device-folder root inside the
|
||||
downloaded tree; `copy-local` walks all `<model>/<backend>/*.mp4`
|
||||
underneath it. Since the Modal run was scoped to a single test file via
|
||||
`--test-files`, only that test's model(s) are present — so the copy is
|
||||
implicitly per-test.)
|
||||
|
||||
Result: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
|
||||
### 6. Upload to HF — scoped per model_id, with overwrite guard
|
||||
|
||||
For each `<model_id>`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>"
|
||||
```
|
||||
|
||||
The upload command:
|
||||
|
||||
- Uploads **only** `reference_videos/default/L40S_reference_videos/<model_id>/`.
|
||||
- **Refuses** if any file already exists at that path on HF (this is the
|
||||
guard — seeding a new test should never clobber existing refs). To override,
|
||||
the user must re-run with `--force`. If the guard fires, stop and report
|
||||
exactly which files exist; do not silently `--force`.
|
||||
|
||||
Reads the HF token from `HF_API_KEY` / `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`.
|
||||
|
||||
### 7. Report success
|
||||
|
||||
List what was uploaded (paths in repo) and remind the user to push any
|
||||
related code changes. Do **not** auto-verify by re-running Modal — the user
|
||||
can run `pytest fastvideo/tests/ssim/<test_file>` later to confirm end-to-end;
|
||||
it will auto-download the refs they just uploaded.
|
||||
|
||||
## Failure modes and how to handle them
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before step 2. The Modal run needs it (passed
|
||||
via `--hf-api-key`), and step 6 needs it for upload.
|
||||
- **Modal run fails before generation.** No mp4s on the volume — nothing to
|
||||
download. Fix the test locally (`pytest fastvideo/tests/ssim/<test_file>`)
|
||||
and retry from step 2.
|
||||
- **`./generated_videos_modal/default/L40S_reference_videos/` missing after
|
||||
`modal volume get`.** The run didn't produce videos (most likely the test
|
||||
crashed before writing, or `REQUIRED_GPUS` exceeded the partition capacity
|
||||
— see Modal logs).
|
||||
- **Upload guard fires (files already exist).** The test name / model id
|
||||
collides with something already on HF. Verify the user actually wants to
|
||||
replace existing refs; if so, re-run the upload with `--force`. If not,
|
||||
rename the model id in `*_MODEL_TO_PARAMS` and re-seed.
|
||||
- **Quality looks wrong in step 4.** Abort. The mp4s stay on disk for
|
||||
inspection. The fix is usually in the test's params (resolution, steps,
|
||||
seed) — edit the test, then re-run the skill.
|
||||
|
||||
## Design notes (for future skill maintainers)
|
||||
|
||||
- The skill deliberately runs on Modal, **not** locally, because the CI
|
||||
runner is L40S. Seeding from a different GPU SKU produces refs that CI's
|
||||
L40S runs can't match (SSIM drifts across SKUs).
|
||||
- The skill is default-tier only. `full_quality` refs are seeded by a
|
||||
separate, deliberate operation — they double runtime and aren't what CI
|
||||
gates on.
|
||||
- The overwrite guard in `reference_videos_cli.py upload` is default-on
|
||||
specifically because this skill exists. Re-seeding is a distinct operation
|
||||
that requires explicit `--force`.
|
||||
|
||||
## References
|
||||
|
||||
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator; see
|
||||
`--sync-generated-to-volume`, `--generated-volume-subdir`,
|
||||
`--skip-reference-download`, `--no-fail-fast`.
|
||||
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
|
||||
(with `--model-id`, `--force`), `download`, `ensure` subcommands.
|
||||
- `fastvideo/tests/ssim/README.md` — reference layout, HF repo conventions.
|
||||
- `fastvideo/tests/ssim/inference_similarity_utils.py` —
|
||||
`run_text_to_video_similarity_test` + `_build_init_kwargs`: what each test
|
||||
config passes to `VideoGenerator.from_pretrained`.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-04-17 | Initial version (Modal sync-to-volume flow). |
|
||||
| 2026-04-21 | Rewrite: single-test scope, explicit user-review pause, per-`model_id` upload, HF overwrite guard. Dropped `scripts/seed_ssim.sh`. |
|
||||
| 2026-04-21 | Post-first-run fixes: `modal volume get` needs `--force` when parent exists; download tree has an extra `generated_videos/` level so `--generated-dir` must reflect it. |
|
||||
@@ -4,21 +4,24 @@ description: How to develop, validate, and register a new evaluation metric
|
||||
|
||||
# Evaluation Development SOP
|
||||
|
||||
Standard procedure for adding new video quality evaluation metrics to the
|
||||
FastVideo agent toolkit.
|
||||
Standard procedure for adding new video quality evaluation metrics to
|
||||
the FastVideo agent toolkit.
|
||||
|
||||
## When to Use
|
||||
## When to use
|
||||
|
||||
- You need a metric that doesn't exist in `.agents/memory/evaluation-registry/README.md`.
|
||||
- You need a metric that does not exist in
|
||||
`.agents/memory/evaluation-registry/README.md`.
|
||||
- An existing metric needs significant changes to its methodology.
|
||||
- You're exploring a new evaluation approach.
|
||||
- You are exploring a new evaluation approach.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Research
|
||||
|
||||
- Search `.agents/memory/related-work/` for existing evaluation approaches.
|
||||
- Check the `evaluation_registry.md` for current metrics and their limitations.
|
||||
- Search `.agents/memory/related-work/` for existing evaluation
|
||||
approaches.
|
||||
- Check `.agents/memory/evaluation-registry/README.md` for current
|
||||
metrics and their limitations.
|
||||
- Review literature: FVD, CLIP-Score, human preference, etc.
|
||||
|
||||
### 2. Prototype
|
||||
@@ -29,21 +32,25 @@ FastVideo agent toolkit.
|
||||
|
||||
### 3. Validate
|
||||
|
||||
- **Known-good test**: Metric should score high on reference-quality videos.
|
||||
- **Known-bad test**: Metric should score low on degraded/unrelated videos.
|
||||
- **Sensitivity test**: Small quality differences should produce meaningful
|
||||
score differences.
|
||||
- **Known-good test**: metric should score high on reference-quality
|
||||
videos.
|
||||
- **Known-bad test**: metric should score low on degraded or unrelated
|
||||
videos.
|
||||
- **Sensitivity test**: small quality differences should produce
|
||||
meaningful score differences.
|
||||
- Document thresholds and their justification.
|
||||
|
||||
### 4. Register
|
||||
|
||||
Update `.agents/memory/evaluation-registry/README.md`:
|
||||
|
||||
- Add the metric with status `Active`.
|
||||
- Document location, thresholds, and trust level.
|
||||
|
||||
### 5. Integrate
|
||||
|
||||
Update `.agents/skills/evaluate-video-quality.md`:
|
||||
Update `.agents/skills/evaluate-video-quality/SKILL.md`:
|
||||
|
||||
- Add the new metric as a section.
|
||||
- Include code examples and interpretation guide.
|
||||
|
||||
@@ -52,3 +59,35 @@ Update `.agents/skills/evaluate-video-quality.md`:
|
||||
- Move the exploration log content into the skill.
|
||||
- Clean up the exploration file or mark it as `promoted`.
|
||||
- If anything went wrong during development, create a lesson.
|
||||
|
||||
## Where the metrics live
|
||||
|
||||
The eval suite is `fastvideo/eval/`. New metrics register themselves
|
||||
via `@register("<group>.<name>")` and are auto-discovered when
|
||||
`fastvideo.eval.metrics` is imported.
|
||||
|
||||
- **Native metrics** (SSIM, PSNR, LPIPS, optical flow, VLM): add a
|
||||
file under the appropriate group dir
|
||||
(`fastvideo/eval/metrics/common/`, `optical_flow/`, `videoscore2/`,
|
||||
`physics_iq/`).
|
||||
- **Metrics that wrap upstream research code**: follow the vbench
|
||||
pattern in `fastvideo/eval/metrics/vbench/`. The contract is:
|
||||
- Upstream lives as a git submodule under
|
||||
`fastvideo/third_party/eval/<bench>/`, pinned to a SHA in repo-root
|
||||
`.gitmodules`.
|
||||
- The metric package's `__init__.py` inserts the submodule path on
|
||||
`sys.path` and installs runtime compat shims (attribute-level
|
||||
monkey-patches) for any modern-dep drift. Do not modify upstream
|
||||
files on disk, and do not ship a `setup.sh`.
|
||||
- See `fastvideo/eval/README.md` for the worked vbench example.
|
||||
- Full porting guide:
|
||||
[`docs/contributing/eval-metrics.md`](../../docs/contributing/eval-metrics.md).
|
||||
|
||||
## Out of scope of the initial eval port
|
||||
|
||||
The following land in follow-up PRs:
|
||||
|
||||
- **MIND** metrics (depends on a separate `vipe` submodule).
|
||||
- **VBench-2.0** sibling package.
|
||||
- Native conversion of **FVD** under `fastvideo/eval/metrics/fvd/`.
|
||||
- The training-time `EvalCallback`.
|
||||
|
||||
+1
-1
@@ -105,7 +105,7 @@ pull_request_rules:
|
||||
- files~=^fastvideo/pipelines/samplers/
|
||||
- files~=^fastvideo/entrypoints/
|
||||
- files~=^fastvideo/worker/
|
||||
- files~=^fastvideo/configs/sample/
|
||||
- files~=^fastvideo/api/sampling_param
|
||||
- files~=^fastvideo/configs/pipelines/
|
||||
- files~=^examples/inference/
|
||||
- -closed
|
||||
|
||||
@@ -4,3 +4,6 @@
|
||||
[submodule "fastvideo-kernel/include/cutlass"]
|
||||
path = fastvideo-kernel/include/cutlass
|
||||
url = https://github.com/NVIDIA/cutlass.git
|
||||
[submodule "fastvideo/third_party/eval/vbench"]
|
||||
path = fastvideo/third_party/eval/vbench
|
||||
url = https://github.com/Vchitect/VBench.git
|
||||
|
||||
@@ -62,9 +62,9 @@ This page contains the complete API reference for the FastVideo library.
|
||||
show_root_toc_entry: true
|
||||
heading_level: 4
|
||||
|
||||
#### fastvideo.configs.sample
|
||||
#### fastvideo.api.sampling_param
|
||||
|
||||
::: fastvideo.configs.sample
|
||||
::: fastvideo.api.sampling_param
|
||||
options:
|
||||
show_source: true
|
||||
show_root_heading: true
|
||||
|
||||
@@ -1,318 +0,0 @@
|
||||
# Attention QAT
|
||||
|
||||
Attention QAT in FastVideo covers two related, but different, backends:
|
||||
|
||||
- `ATTN_QAT_INFER`: the inference-oriented CUDA kernel path
|
||||
- `ATTN_QAT_TRAIN`: the training-oriented Triton attention path
|
||||
|
||||
Both are selected with `FASTVIDEO_ATTENTION_BACKEND`, but they are not
|
||||
interchangeable. The main practical split is:
|
||||
|
||||
- use `ATTN_QAT_INFER` for standalone inference with the dedicated inference
|
||||
kernel
|
||||
- use `ATTN_QAT_TRAIN` for finetuning, validation during training, or when you
|
||||
specifically want to reproduce the training-side attention path
|
||||
|
||||
## Quick Start
|
||||
|
||||
If your goal is "run Wan 2.1 14B with Attention QAT inference weights", this is
|
||||
the shortest path:
|
||||
|
||||
1. Build the in-repo kernel package so FastVideo can import `attn_qat_infer`.
|
||||
2. Download the Wan 2.1 14B QAT checkpoint.
|
||||
3. Edit the provided inference example to point at the 14B base model and the
|
||||
downloaded QAT safetensors.
|
||||
4. Run the example with `ATTN_QAT_INFER`.
|
||||
|
||||
### Step 1. Build the kernel package
|
||||
|
||||
Before using either Attention QAT backend, build the in-repo
|
||||
`fastvideo-kernel` package from source:
|
||||
|
||||
```bash
|
||||
git submodule update --init --recursive
|
||||
cd fastvideo-kernel
|
||||
./build.sh
|
||||
```
|
||||
|
||||
After a successful build:
|
||||
|
||||
- `ATTN_QAT_TRAIN` should be able to import `fastvideo_kernel`
|
||||
- `ATTN_QAT_INFER` should be able to import `attn_qat_infer`
|
||||
|
||||
`ATTN_QAT_INFER` currently targets the Blackwell CUDA path under
|
||||
`fastvideo-kernel/attn_qat_infer/` and requires CUDA 12.8+.
|
||||
|
||||
### Step 2. Download the Wan 2.1 14B QAT checkpoint
|
||||
|
||||
FastVideo includes a helper script:
|
||||
|
||||
- `examples/inference/optimizations/download_14B_qat.sh`
|
||||
|
||||
By default it downloads:
|
||||
|
||||
- Hugging Face repo: `FastVideo/14B_qat_400`
|
||||
- local directory: `checkpoints/14B_qat_400`
|
||||
|
||||
Prerequisites:
|
||||
|
||||
- `huggingface_hub` installed, for example:
|
||||
`uv pip install huggingface_hub`
|
||||
- access to the model repo if it is private or gated:
|
||||
`huggingface-cli login`
|
||||
|
||||
Run the downloader:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh
|
||||
```
|
||||
|
||||
To download into a custom directory:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
|
||||
```
|
||||
|
||||
The script prints a ready-to-copy `init_weights_from_safetensors=...` value at
|
||||
the end.
|
||||
|
||||
### Step 3. Edit the provided inference example
|
||||
|
||||
The example to start from is:
|
||||
|
||||
- `examples/inference/optimizations/attn_qat_inference_example.py`
|
||||
|
||||
Open that file and update these two values:
|
||||
|
||||
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
|
||||
`Wan-AI/Wan2.1-T2V-14B-Diffusers`
|
||||
2. Replace
|
||||
`init_weights_from_safetensors="safetensors_path"` with the directory that
|
||||
contains the downloaded `.safetensors` files
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
init_weights_from_safetensors="checkpoints/14B_qat_400",
|
||||
)
|
||||
```
|
||||
|
||||
Important:
|
||||
|
||||
- the checked-in example currently uses the `1.3B` base model until you edit it
|
||||
- do not load the 14B QAT weights on top of the `1.3B` base model; the weights
|
||||
and model config will not match
|
||||
|
||||
### Step 4. Run the inference example
|
||||
|
||||
```bash
|
||||
python examples/inference/optimizations/attn_qat_inference_example.py
|
||||
```
|
||||
|
||||
Generated videos are written to `video_samples/` by default.
|
||||
|
||||
## Backend Overview
|
||||
|
||||
| Backend | Best for | Package requirement | Primary kernel location |
|
||||
|---------|----------|---------------------|-------------------------|
|
||||
| `ATTN_QAT_TRAIN` | finetuning, training-time validation, reproducing the training path | `fastvideo_kernel` | `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` |
|
||||
| `ATTN_QAT_INFER` | standalone inference with the dedicated CUDA kernel | `attn_qat_infer` from the in-repo `fastvideo-kernel` checkout | `fastvideo-kernel/attn_qat_infer/` |
|
||||
|
||||
FastVideo routes backend selection through:
|
||||
|
||||
- `fastvideo/envs.py`
|
||||
- `fastvideo/platforms/cuda.py`
|
||||
- `fastvideo/attention/backends/attn_qat_train.py`
|
||||
- `fastvideo/attention/backends/attn_qat_infer.py`
|
||||
|
||||
The legacy training pipeline also contains explicit Attention QAT integration:
|
||||
|
||||
- `fastvideo/training/training_pipeline.py`
|
||||
|
||||
That pipeline forces generator loading through `ATTN_QAT_TRAIN` when
|
||||
`FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN` or `--generator_4bit_attn` is
|
||||
enabled.
|
||||
|
||||
## Inference Workflows
|
||||
|
||||
For standalone inference, prefer `ATTN_QAT_INFER` when the CUDA kernel is
|
||||
available. Use `ATTN_QAT_TRAIN` for inference only if you intentionally want to
|
||||
exercise the training-side attention path for debugging or parity checks.
|
||||
|
||||
### Minimal Python example
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1,
|
||||
)
|
||||
|
||||
generator.generate_video(
|
||||
"A cinematic close-up of rain on a neon street at night.",
|
||||
output_path="video_samples",
|
||||
save_video=True,
|
||||
)
|
||||
```
|
||||
|
||||
### Loading custom safetensors during inference
|
||||
|
||||
FastVideo supports loading custom transformer weights through
|
||||
`init_weights_from_safetensors`.
|
||||
|
||||
This value can point to either:
|
||||
|
||||
- a directory containing one or more `.safetensors` files
|
||||
- a single `.safetensors` file
|
||||
|
||||
For Wan 2.1 14B QAT inference, the common pattern is:
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
init_weights_from_safetensors="checkpoints/14B_qat_400",
|
||||
)
|
||||
```
|
||||
|
||||
### CLI example
|
||||
|
||||
You can also force the backend from the command line:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
|
||||
fastvideo generate \
|
||||
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
|
||||
--num-gpus 1 \
|
||||
--sp-size 1 \
|
||||
--tp-size 1 \
|
||||
--height 480 \
|
||||
--width 832 \
|
||||
--num-frames 77 \
|
||||
--num-inference-steps 50 \
|
||||
--guidance-scale 6.0 \
|
||||
--prompt "A cinematic close-up of rain on a neon street at night." \
|
||||
--output-path outputs_video/
|
||||
```
|
||||
|
||||
If you want to use custom QAT transformer weights from the CLI, pass the same
|
||||
custom weight override that the Python API uses:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
|
||||
fastvideo generate \
|
||||
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
|
||||
--init-weights-from-safetensors checkpoints/14B_qat_400 \
|
||||
--num-gpus 1 \
|
||||
--output-path outputs_video/ \
|
||||
--prompt "A cinematic close-up of rain on a neon street at night."
|
||||
```
|
||||
|
||||
## Training Workflows
|
||||
|
||||
Today the checked-in Attention QAT training launchers use the legacy training
|
||||
pipeline in `fastvideo/training/wan_training_pipeline.py`.
|
||||
|
||||
### Ready-made launchers
|
||||
|
||||
Use the provided SLURM scripts directly:
|
||||
|
||||
```bash
|
||||
sbatch examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh
|
||||
sbatch examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh
|
||||
```
|
||||
|
||||
Both scripts already set:
|
||||
|
||||
```bash
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
```
|
||||
|
||||
Before launching, update the script-local values that depend on your
|
||||
environment:
|
||||
|
||||
- `WANDB_API_KEY`
|
||||
- `MODEL_PATH`
|
||||
- `DATA_DIR`
|
||||
- `VALIDATION_DATASET_FILE`
|
||||
- output directory and SLURM resource requests
|
||||
|
||||
### What the launchers run
|
||||
|
||||
The training scripts eventually invoke:
|
||||
|
||||
```bash
|
||||
torchrun fastvideo/training/wan_training_pipeline.py ...
|
||||
```
|
||||
|
||||
If you are adapting the workflow to your own cluster or running outside SLURM,
|
||||
the main Attention QAT requirement is still:
|
||||
|
||||
```bash
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
```
|
||||
|
||||
Then launch the normal Wan training pipeline with your preferred `torchrun`
|
||||
arguments and training flags.
|
||||
|
||||
## Where The Code Lives
|
||||
|
||||
Use these paths when you want to trace or modify the Attention QAT flow:
|
||||
|
||||
| Location | Purpose |
|
||||
|----------|---------|
|
||||
| `fastvideo/attention/backends/attn_qat_train.py` | FastVideo wrapper that imports and calls the Triton training kernel |
|
||||
| `fastvideo/attention/backends/attn_qat_infer.py` | FastVideo wrapper that imports and calls the inference kernel |
|
||||
| `fastvideo-kernel/CMakeLists.txt` | Kernel build definition that compiles the `attn_qat_infer` inference extensions |
|
||||
| `fastvideo/platforms/cuda.py` | Chooses the concrete attention backend at runtime |
|
||||
| `fastvideo/envs.py` | Documents supported `FASTVIDEO_ATTENTION_BACKEND` values |
|
||||
| `fastvideo/training/training_pipeline.py` | Training-time forcing logic for the generator attention backend |
|
||||
| `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` | Triton implementation for `ATTN_QAT_TRAIN` |
|
||||
| `fastvideo-kernel/attn_qat_infer/api.py` | Python API entrypoint for the inference kernel |
|
||||
| `fastvideo-kernel/benchmarks/benchmark_*.py` | Kernel-side benchmark scripts for FlashAttn2, SageAttention3, FP4, and comparison plots |
|
||||
| `fastvideo-kernel/attn_qat_infer/blackwell/api.cu` | CUDA implementation behind `ATTN_QAT_INFER` |
|
||||
| `fastvideo-kernel/tests/test_attn_qat_train.py` | Kernel-level test coverage for the training path |
|
||||
| `examples/inference/optimizations/attn_qat_inference_example.py` | Ready-to-edit inference example for custom Attention QAT weights |
|
||||
| `examples/inference/optimizations/download_14B_qat.sh` | Helper script for downloading the Wan 2.1 14B QAT checkpoint |
|
||||
| `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 1.3B Attention QAT finetune launcher |
|
||||
| `examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 14B Attention QAT finetune launcher |
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- If `ATTN_QAT_TRAIN` fails to import, verify that `fastvideo-kernel` built
|
||||
successfully and exposes `fastvideo_kernel`.
|
||||
- If `ATTN_QAT_INFER` fails to import, verify that the local build exposes the
|
||||
`attn_qat_infer` package.
|
||||
- If the Wan 2.1 14B example fails after you changed only the checkpoint path,
|
||||
make sure you also changed the base model to
|
||||
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
|
||||
- If you hit issues with CPU memory pressure or obscure CUDA argument errors in
|
||||
the example script, try setting `pin_cpu_memory=False`.
|
||||
- If you want a known-safe fallback for debugging, use
|
||||
`FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA`.
|
||||
|
||||
## Related Pages
|
||||
|
||||
- [Attention Overview](../index.md)
|
||||
- [Inference Optimizations](../../inference/optimizations.md)
|
||||
- [Debugging](../../utilities/debugging.md)
|
||||
@@ -5,8 +5,6 @@ FastVideo provides highly optimized custom attention kernels to accelerate video
|
||||
## Supported Kernels
|
||||
|
||||
* **[Video Sparse Attention (VSA)](vsa/index.md)**: Sparse attention mechanism selecting top-k blocks.
|
||||
* **[Attention QAT](attn_qat/index.md)**: Dedicated guide for Attention QAT
|
||||
inference, training, checkpoint loading, and troubleshooting.
|
||||
* **[Sliding Tile Attention (STA)](sta/index.md)**: STA kernel support is kept in
|
||||
`fastvideo-kernel`; full FastVideo STA pipeline workflow is archived in
|
||||
`sta_do_not_delete`.
|
||||
|
||||
@@ -173,7 +173,7 @@ Applied by Mergify based on which paths you modified. Multiple scope labels can
|
||||
| Label | File paths that trigger it |
|
||||
|-------|---------------------------|
|
||||
| `scope: training` | `fastvideo/train/`, `fastvideo/training/`, `fastvideo/distillation/`, `examples/train/`, `examples/training/`, `examples/distill/` |
|
||||
| `scope: inference` | `fastvideo/pipelines/basic/`, `fastvideo/pipelines/stages/`, `fastvideo/pipelines/samplers/`, `fastvideo/entrypoints/`, `fastvideo/worker/`, `fastvideo/configs/sample/`, `fastvideo/configs/pipelines/`, `examples/inference/` |
|
||||
| `scope: inference` | `fastvideo/pipelines/basic/`, `fastvideo/pipelines/stages/`, `fastvideo/pipelines/samplers/`, `fastvideo/entrypoints/`, `fastvideo/worker/`, `fastvideo/api/sampling_param.py`, `fastvideo/configs/pipelines/`, `examples/inference/` |
|
||||
| `scope: attention` | `fastvideo/attention/` |
|
||||
| `scope: kernel` | `fastvideo-kernel/`, `csrc/` |
|
||||
| `scope: data` | `fastvideo/dataset/`, `fastvideo/pipelines/preprocess/`, `examples/preprocessing/` |
|
||||
|
||||
@@ -44,7 +44,7 @@ FastVideo maps a Diffusers-style repo into a pipeline like:
|
||||
- `fastvideo/configs/models/*`: arch configs and `param_names_mapping` for
|
||||
weight name translation.
|
||||
- `fastvideo/configs/pipelines/*`: pipeline wiring (component classes + names).
|
||||
- `fastvideo/configs/sample/*`: default runtime sampling parameters.
|
||||
- `fastvideo/api/sampling_param.py`: runtime sampling parameters.
|
||||
- `fastvideo/pipelines/basic/*`: end-to-end pipeline logic built from stages.
|
||||
- `model_index.json`: the HF repo entrypoint that maps component names to
|
||||
classes and weight files.
|
||||
@@ -55,7 +55,7 @@ Minimal usage example (based on `examples/inference/basic/basic.py`):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # or official_weights/<model_name>/
|
||||
generator = VideoGenerator.from_pretrained(model_id, num_gpus=1)
|
||||
@@ -319,7 +319,8 @@ Purpose:
|
||||
|
||||
- `fastvideo/configs/pipelines/` describes pipeline wiring and model module
|
||||
names.
|
||||
- `fastvideo/configs/sample/` defines default runtime parameters.
|
||||
- `fastvideo/api/sampling_param.py` defines runtime sampling parameters.
|
||||
Defaults come from profiles in `fastvideo/pipelines/basic/<family>/profiles.py`.
|
||||
|
||||
Action:
|
||||
|
||||
@@ -474,7 +475,7 @@ FastVideo integration.
|
||||
3. Pipeline wiring.
|
||||
- Pipeline: `fastvideo/pipelines/basic/wan/wan_pipeline.py`
|
||||
- Pipeline config: `fastvideo/configs/pipelines/wan.py`
|
||||
- Sampling defaults: `fastvideo/configs/sample/wan.py`
|
||||
- Sampling defaults: `fastvideo/pipelines/basic/wan/profiles.py`
|
||||
|
||||
4. Minimal example.
|
||||
- Script: `examples/inference/basic/basic.py`
|
||||
|
||||
@@ -0,0 +1,559 @@
|
||||
# Porting Eval Metrics into `fastvideo.eval`
|
||||
|
||||
This guide is for contributors adding new evaluation metrics to
|
||||
FastVideo's eval suite. To run the existing metrics, see
|
||||
[`fastvideo/eval/README.md`](../../fastvideo/eval/README.md).
|
||||
|
||||
## When to use this guide
|
||||
|
||||
Use this guide when you are:
|
||||
|
||||
- Adding a new metric (native or wrapping a third-party library).
|
||||
- Porting a benchmark (e.g. VBench, MIND, EvalCrafter) whose Python
|
||||
code needs to be importable from a pinned upstream.
|
||||
- Adding a new metric group (audio, vlm, etc.).
|
||||
|
||||
## TL;DR
|
||||
|
||||
Metrics are auto-discovered from
|
||||
`fastvideo/eval/metrics/<group>/<name>/metric.py`. Each declares itself
|
||||
with `@register("<group>.<name>")` and subclasses `BaseMetric`. Three
|
||||
recipes:
|
||||
|
||||
1. **Native metric** (pure-PyTorch, no submodule). Drop a file,
|
||||
declare deps, implement `compute(sample)`.
|
||||
2. **Library-wrapped metric** (CLIP, torch.hub, transformers, pyiqa).
|
||||
Same as above, plus route the library's cache through
|
||||
`get_cache_dir()` if it has a `download_root=` / `cache_dir=`
|
||||
kwarg.
|
||||
3. **Upstream-submodule-wrapped metric** (vbench-style). Pin upstream
|
||||
as a git submodule under `fastvideo/third_party/eval/<bench>/`. The
|
||||
adapter `__init__.py` does the `sys.path` insert and any runtime
|
||||
compat shims for modern dep versions. Patches live as Python in
|
||||
that file rather than as on-disk patches to the submodule.
|
||||
|
||||
The full recipes are below.
|
||||
|
||||
---
|
||||
|
||||
## 0) Layout and auto-discovery
|
||||
|
||||
```
|
||||
fastvideo/eval/metrics/
|
||||
├── base.py # BaseMetric + lifecycle contract
|
||||
├── common/ # group: SSIM, PSNR, LPIPS
|
||||
├── optical_flow/ # group: gt_optical_flow, synthetic_optical_flow
|
||||
├── vlm/ # group: VideoScore-2
|
||||
├── physics_iq/ # group + sub-metrics
|
||||
└── vbench/ # group: 16 sub-metrics
|
||||
├── __init__.py # sys.path bootstrap + runtime compat shims
|
||||
├── _grit_helper.py # shared upstream-touching helpers
|
||||
└── <sub_metric>/metric.py
|
||||
```
|
||||
|
||||
Auto-discovery (`fastvideo/eval/metrics/__init__.py`) walks each group
|
||||
dir and imports every `metric.py` it finds, which fires the
|
||||
`@register` decorators. Names starting with `_` are skipped. Use that
|
||||
prefix for shared helpers or vendored code that should not register
|
||||
itself.
|
||||
|
||||
---
|
||||
|
||||
## 1) The `BaseMetric` contract
|
||||
|
||||
Every metric subclasses `fastvideo.eval.metrics.base.BaseMetric` and
|
||||
declares:
|
||||
|
||||
```python
|
||||
class YourMetric(BaseMetric):
|
||||
name: str = "common.your_metric" # must match @register
|
||||
requires_reference: bool = True # needs sample["reference"]
|
||||
higher_is_better: bool = True # for ranking / aggregates
|
||||
dependencies: list[str] = [] # importable module names;
|
||||
# registry surfaces a clean
|
||||
# ImportError if missing
|
||||
needs_gpu: bool = False
|
||||
backbone: str | None = None # e.g. "clip_vit_l14"
|
||||
```
|
||||
|
||||
You must implement:
|
||||
|
||||
```python
|
||||
def compute(self, sample: dict) -> list[MetricResult]:
|
||||
"""sample['video'] is (1, T, C, H, W). Return a one-element list.
|
||||
|
||||
The leading 1 is preserved for forward-compat with batched eval;
|
||||
today :class:`EvalWorker` always invokes metrics with B=1.
|
||||
"""
|
||||
```
|
||||
|
||||
You may override:
|
||||
|
||||
- `setup(self) -> None`. Eager model loading. Called once by
|
||||
`create_evaluator`. Idempotent (re-entrant). Use the `if self._model
|
||||
is not None: return` pattern.
|
||||
- `to(self, device)`. Move the metric and its submodels to `device`.
|
||||
|
||||
If a required input is missing (e.g. an fps-aware metric called
|
||||
without `fps`), return `self._skip(sample, reason)` instead of
|
||||
raising.
|
||||
|
||||
---
|
||||
|
||||
## 2) Recipe A: native metric (no external deps)
|
||||
|
||||
Smallest case. Pixel math, simple closed-form.
|
||||
|
||||
```python
|
||||
# fastvideo/eval/metrics/common/your_metric/metric.py
|
||||
from __future__ import annotations
|
||||
import torch
|
||||
from fastvideo.eval.metrics.base import BaseMetric
|
||||
from fastvideo.eval.registry import register
|
||||
from fastvideo.eval.types import MetricResult
|
||||
|
||||
|
||||
@register("common.your_metric")
|
||||
class YourMetric(BaseMetric):
|
||||
name = "common.your_metric"
|
||||
requires_reference = True
|
||||
higher_is_better = True
|
||||
needs_gpu = False
|
||||
dependencies: list[str] = [] # nothing extra
|
||||
|
||||
def compute(self, sample: dict) -> list[MetricResult]:
|
||||
gen, ref = sample["video"], sample["reference"] # (B,T,C,H,W) each
|
||||
per_video = ((gen - ref) ** 2).mean(dim=(1, 2, 3, 4)).sqrt()
|
||||
return [
|
||||
MetricResult(name=self.name, score=float(s), details={})
|
||||
for s in per_video
|
||||
]
|
||||
```
|
||||
|
||||
That is the whole recipe. Drop the file and the registry picks it up.
|
||||
|
||||
---
|
||||
|
||||
## 3) Recipe B: library-wrapped metric (CLIP, torch.hub, transformers, pyiqa)
|
||||
|
||||
If your metric loads a backbone from a Python package, route the
|
||||
library at the eval cache so users get one knob (`FASTVIDEO_EVAL_CACHE`)
|
||||
to redirect everything.
|
||||
|
||||
### Cache routing rules
|
||||
|
||||
| Library | How to route | Location after redirect |
|
||||
|---|---|---|
|
||||
| `clip.load("ViT-X")` | pass `download_root=str(get_cache_dir() / "clip")` | `${FASTVIDEO_EVAL_CACHE}/clip/` |
|
||||
| `torch.hub.load(...)` | nothing; `TORCH_HOME` is redirected at `fastvideo.eval` import time | `${FASTVIDEO_EVAL_CACHE}/torch/hub/` |
|
||||
| `transformers.from_pretrained(...)` | nothing; leave HF's default cache (`~/.cache/huggingface/hub/`) so users dedupe with other ML projects | `~/.cache/huggingface/hub/` |
|
||||
| `huggingface_hub.snapshot_download` / `hf_hub_download` | use `ensure_checkpoint(...)` (it wraps these with filelock) | same as above |
|
||||
| `pyiqa.create_metric(...)` | no env var or kwarg honored; document in metric docstring | pyiqa-internal |
|
||||
| `lpips`, `ptlflow` | torch.hub-based, auto-redirected | `${FASTVIDEO_EVAL_CACHE}/torch/hub/` |
|
||||
| Raw URL (no HF Hub) | use `ensure_checkpoint(name, source="https://...")` | `${FASTVIDEO_EVAL_CACHE}/models/<name>` |
|
||||
| Dataset asset (raw video/mask/image) auto-fetched from a public bucket | download into `get_cache_dir() / "datasets" / "<bench>"`, mirroring upstream's relative layout. Vendor any small manifest (CSV/JSON ≤1 MB) under the metric folder so the dataset can be used without external setup. | `${FASTVIDEO_EVAL_CACHE}/datasets/<bench>/` |
|
||||
|
||||
### Dataset assets: vendor the manifest, auto-fetch the rest
|
||||
|
||||
If your metric ships with its own paired-reference dataset (Physics-IQ
|
||||
is the canonical example), follow this layout:
|
||||
|
||||
- **Manifest** (CSV/JSON ≤1 MB): vendor it under
|
||||
`fastvideo/eval/metrics/<bench>/_vendored/<manifest>.<ext>`, with a
|
||||
sibling `_vendored/LICENSE` recording attribution and provenance.
|
||||
The `_vendored/` subdir is the project-wide convention for
|
||||
upstream-provenance files: it is auto-skipped by metric discovery
|
||||
(the `_` prefix) and by codespell (one `*/_vendored/*` glob in
|
||||
`[tool.codespell].skip`), so dropping in a new vendored file
|
||||
requires no further config. Read the manifest from the dataset
|
||||
module via a `Path(__file__)`-relative resolver. Mirror
|
||||
`_VENDORED_DESCRIPTIONS_CSV` in
|
||||
`fastvideo/eval/datasets/physics_iq.py`.
|
||||
- **Heavy assets** (videos, masks, images): do not vendor. Auto-fetch
|
||||
on first miss into `get_cache_dir() / "datasets" / "<bench>"`,
|
||||
mirroring upstream's relative directory layout one-for-one so a
|
||||
pre-downloaded mirror at any path works as a drop-in
|
||||
`dataset_root=`. Use atomic `.part` then final-rename to be safe
|
||||
under concurrent SLURM ranks.
|
||||
- **Bucket override**: expose `FASTVIDEO_<BENCH>_BUCKET_URL` so users
|
||||
with internal mirrors can redirect.
|
||||
- **Opt-out**: accept `auto_download: bool = True` in the dataset
|
||||
constructor; on `False`, raise `FileNotFoundError` instead of
|
||||
fetching. This covers air-gapped runs and CI.
|
||||
|
||||
The end-state is `get_dataset("<bench>")` with no kwargs.
|
||||
|
||||
### Example: CLIP backbone + LAION head
|
||||
|
||||
```python
|
||||
# fastvideo/eval/metrics/your_group/your_metric/metric.py
|
||||
from __future__ import annotations
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
from fastvideo.eval.metrics.base import BaseMetric
|
||||
from fastvideo.eval.registry import register
|
||||
from fastvideo.eval.types import MetricResult
|
||||
|
||||
|
||||
@register("your_group.your_metric")
|
||||
class YourMetric(BaseMetric):
|
||||
name = "your_group.your_metric"
|
||||
requires_reference = False
|
||||
needs_gpu = True
|
||||
dependencies = ["clip"] # "openai-clip" PyPI; importable as `clip`
|
||||
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
self._clip = None
|
||||
self._head = None
|
||||
|
||||
def setup(self) -> None:
|
||||
if self._clip is not None:
|
||||
return
|
||||
import clip
|
||||
from fastvideo.eval.models import ensure_checkpoint, get_cache_dir
|
||||
|
||||
# Backbone: route CLIP's cache through our root.
|
||||
self._clip, _ = clip.load(
|
||||
"ViT-L/14",
|
||||
device=self.device,
|
||||
download_root=str(get_cache_dir() / "clip"),
|
||||
)
|
||||
self._clip.eval()
|
||||
|
||||
# URL-fetched head: ensure_checkpoint downloads to
|
||||
# ${FASTVIDEO_EVAL_CACHE}/models/ with filelock + atomic rename.
|
||||
ckpt = ensure_checkpoint(
|
||||
"your_head.pth",
|
||||
source="https://example.com/path/to/your_head.pth",
|
||||
)
|
||||
self._head = nn.Linear(768, 1)
|
||||
self._head.load_state_dict(
|
||||
torch.load(ckpt, map_location="cpu", weights_only=True)
|
||||
)
|
||||
self._head.to(self.device).eval()
|
||||
|
||||
def to(self, device):
|
||||
super().to(device)
|
||||
if self._clip is not None:
|
||||
self._clip = self._clip.to(self.device)
|
||||
if self._head is not None:
|
||||
self._head = self._head.to(self.device)
|
||||
return self
|
||||
|
||||
def compute(self, sample: dict) -> list[MetricResult]:
|
||||
...
|
||||
```
|
||||
|
||||
### Do not redirect other `~/.cache/...` dirs
|
||||
|
||||
If a third-party library hard-codes `~/.cache/<lib>/` and offers no
|
||||
override, document the exception in the metric's docstring. Forcing
|
||||
redirection by setting `os.environ` or patching `os.path.expanduser`
|
||||
is fragile and breaks user expectations of where the library's cache
|
||||
lives.
|
||||
|
||||
---
|
||||
|
||||
## 4) Recipe C: upstream-submodule-wrapped metric (vbench pattern)
|
||||
|
||||
Use this when the upstream benchmark ships Python code (`vbench/`,
|
||||
`MIND/`, etc.) that is not pip-installable cleanly. See
|
||||
`fastvideo/eval/metrics/vbench/__init__.py` for the worked example.
|
||||
|
||||
### 4.1 Pin the upstream as a submodule
|
||||
|
||||
```bash
|
||||
git submodule add <upstream-url> fastvideo/third_party/eval/<bench>
|
||||
cd fastvideo/third_party/eval/<bench>
|
||||
git checkout <pinned-sha>
|
||||
cd -
|
||||
git add .gitmodules fastvideo/third_party/eval/<bench>
|
||||
```
|
||||
|
||||
The submodule pulls under the standard `git submodule update --init
|
||||
--recursive` flow that users already run for kernel deps.
|
||||
|
||||
### 4.2 Bootstrap on `sys.path`
|
||||
|
||||
```python
|
||||
# fastvideo/eval/metrics/<bench>/__init__.py
|
||||
from __future__ import annotations
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# fastvideo/eval/metrics/<bench>/__init__.py → ../../../../third_party/eval/<bench>
|
||||
_UPSTREAM = Path(__file__).resolve().parents[3] / "third_party" / "eval" / "<bench>"
|
||||
if _UPSTREAM.is_dir() and str(_UPSTREAM) not in sys.path:
|
||||
sys.path.insert(0, str(_UPSTREAM))
|
||||
```
|
||||
|
||||
We do not `pip install` the upstream because its egg-link/.pth would
|
||||
just re-do this `sys.path.insert`, and skipping the install also skips
|
||||
the upstream's `setup.py` (which often gates on a specific CUDA
|
||||
version).
|
||||
|
||||
### 4.3 Modern-dep compat: runtime shims
|
||||
|
||||
Upstream code pinned to e.g. `transformers==4.33.2`, `numpy<2`
|
||||
typically breaks against modern versions in 3-4 known places (API
|
||||
renames). Fix those at import time, in the same `__init__.py`:
|
||||
|
||||
```python
|
||||
def _install_compat_shims() -> None:
|
||||
# Example: transformers.modeling_utils API moved.
|
||||
try:
|
||||
import transformers.modeling_utils as _mu
|
||||
import transformers.pytorch_utils as _pu
|
||||
for _n in ("apply_chunking_to_forward",
|
||||
"find_pruneable_heads_and_indices",
|
||||
"prune_linear_layer"):
|
||||
if not hasattr(_mu, _n) and hasattr(_pu, _n):
|
||||
setattr(_mu, _n, getattr(_pu, _n))
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# Example: numpy.lib.function_base.disp removed in numpy>=2.
|
||||
try:
|
||||
import types, numpy.lib as _nl
|
||||
if not hasattr(_nl, "function_base"):
|
||||
_stub = types.ModuleType("numpy.lib.function_base")
|
||||
_stub.disp = lambda *a, **k: None
|
||||
sys.modules["numpy.lib.function_base"] = _stub
|
||||
_nl.function_base = _stub
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
_install_compat_shims()
|
||||
```
|
||||
|
||||
For function-level patches that cannot be expressed as attribute
|
||||
writes (e.g. wrapping a model factory function), use a
|
||||
`sys.meta_path` finder that wraps the loader. See
|
||||
`_install_modeling_finetune_hook()` in
|
||||
`fastvideo/eval/metrics/vbench/__init__.py` for the pattern (about 30
|
||||
lines).
|
||||
|
||||
Why shims rather than `git apply` patches: patches go stale when the
|
||||
upstream SHA changes; shims are versioned Python code in our repo,
|
||||
they are grep-able, and they only run if the targeted module is
|
||||
imported.
|
||||
|
||||
### 4.4 Per-sub-metric files
|
||||
|
||||
Each sub-metric is a normal `BaseMetric` subclass that imports from
|
||||
the upstream:
|
||||
|
||||
```python
|
||||
# fastvideo/eval/metrics/<bench>/<sub>/metric.py
|
||||
from fastvideo.eval.metrics.base import BaseMetric
|
||||
from fastvideo.eval.registry import register
|
||||
from fastvideo.eval.types import MetricResult
|
||||
|
||||
|
||||
@register("<bench>.<sub>")
|
||||
class YourSubMetric(BaseMetric):
|
||||
...
|
||||
|
||||
def setup(self) -> None:
|
||||
if self._model is not None:
|
||||
return
|
||||
# The sys.path bootstrap fired when fastvideo.eval.metrics.<bench>
|
||||
# was imported (which auto-discovery does before importing this
|
||||
# sub-package). Upstream imports just work:
|
||||
from <bench>.something import SomeModel
|
||||
...
|
||||
```
|
||||
|
||||
### 4.5 Conditional registration when the upstream is missing
|
||||
|
||||
If a user installed `fastvideo[eval]` but did not run `git submodule
|
||||
update --init`, `<bench>.*` metrics should not register. The
|
||||
auto-discovery walker imports each sub-package's `metric` module; have
|
||||
that import bail out cleanly:
|
||||
|
||||
```python
|
||||
# fastvideo/eval/metrics/<bench>/__init__.py — at the bottom
|
||||
_AVAILABLE = (_UPSTREAM / "<bench>" / "__init__.py").is_file()
|
||||
```
|
||||
|
||||
```python
|
||||
# fastvideo/eval/metrics/<bench>/<sub>/__init__.py
|
||||
from fastvideo.eval.metrics.<bench> import _AVAILABLE
|
||||
if _AVAILABLE:
|
||||
from .metric import YourSubMetric # noqa
|
||||
```
|
||||
|
||||
`fastvideo eval list` then reflects what the user actually has rather
|
||||
than what they could have.
|
||||
|
||||
### 4.6 Leave the upstream alone unless it blocks the metric
|
||||
|
||||
The upstream is pinned. If a metric works against the pinned SHA,
|
||||
leave the upstream files untouched. If it actively breaks against
|
||||
modern deps (the import-drift cases above), shim it. Avoid
|
||||
fastvideo-side forks of upstream code; they make patches go stale and
|
||||
parity drift.
|
||||
|
||||
---
|
||||
|
||||
## 5) Model checkpoints: `ensure_checkpoint`
|
||||
|
||||
Use `ensure_checkpoint(name, source, filename=None)` for any
|
||||
non-package weights. It resolves a local path, downloading on miss,
|
||||
with filelock safety across processes and SLURM ranks.
|
||||
|
||||
| `source` form | What happens |
|
||||
|---|---|
|
||||
| `"/abs/path/to/file.pth"` | passthrough, returned unchanged |
|
||||
| `"https://..."` | downloaded to `${FASTVIDEO_EVAL_CACHE}/models/<name>` via `huggingface_hub.http_get`, atomic rename, filelock |
|
||||
| `"org/repo"` (no `filename`) | `snapshot_download(repo_id)` → `~/.cache/huggingface/hub/` |
|
||||
| `"org/repo"` (with `filename`) | `hf_hub_download(repo_id, filename)` → `~/.cache/huggingface/hub/` |
|
||||
|
||||
`name` is only used as the local filename for URL sources. HF sources
|
||||
ignore it (HF manages its own cache key by content hash).
|
||||
|
||||
```python
|
||||
from fastvideo.eval.models import ensure_checkpoint
|
||||
|
||||
# URL: name matters
|
||||
ckpt = ensure_checkpoint(
|
||||
"amt-s.pth",
|
||||
source="https://huggingface.co/lalala125/AMT/resolve/main/amt-s.pth",
|
||||
)
|
||||
|
||||
# HF single file: name is decorative
|
||||
ckpt = ensure_checkpoint(
|
||||
"raft-things.pth", # ignored; HF cache uses repo+sha
|
||||
source="OpenGVLab/VBench_Used_Models",
|
||||
filename="raft-things.pth",
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 6) Declaring `dependencies`
|
||||
|
||||
Set `dependencies = ["pkg1", "pkg2"]` on your metric class with
|
||||
importable module names (not PyPI distribution names). The registry
|
||||
checks each via `importlib.util.find_spec` at instantiation time and
|
||||
raises a clean `ImportError` pointing the user at the right install
|
||||
extra:
|
||||
|
||||
```python
|
||||
class YourMetric(BaseMetric):
|
||||
dependencies = ["clip", "timm"] # importable as `import clip`, `import timm`
|
||||
```
|
||||
|
||||
If a dep is in `[project.optional-dependencies.eval-<group>]`, you do
|
||||
not need to do anything more. If it is a new dep, add it to that
|
||||
group in `pyproject.toml`.
|
||||
|
||||
---
|
||||
|
||||
## 7) Common gotchas
|
||||
|
||||
- The standard `git submodule update --init --recursive` is enough
|
||||
for a benchmark; do not write a `setup.sh`. Modern-dep compat goes
|
||||
into your `__init__.py` as runtime shims.
|
||||
- Do not modify upstream files on disk. The submodule should always
|
||||
match its pinned SHA. Compat lives in our `__init__.py`.
|
||||
- Do not pip-install the upstream. The egg-link is a glorified
|
||||
`sys.path.insert`, which we do directly in `__init__.py`.
|
||||
- Do not call `torch.hub.set_dir(...)` from your metric. It is done
|
||||
globally in `fastvideo/eval/__init__.py`.
|
||||
- Do not put cache-redirection env vars in your metric's `setup()`.
|
||||
By the time `setup()` runs, the library has likely already cached
|
||||
the default-location decision. Set env vars at package-init time.
|
||||
- Skip rather than raise when an input is missing. Use
|
||||
`self._skip(sample, reason)` for any expected-missing input. It
|
||||
returns a list of `MetricResult(score=None)` so other metrics in
|
||||
the same evaluator continue.
|
||||
- Watch for upstream re-registration conflicts. If the upstream uses
|
||||
a global registry (detectron2's `META_ARCH_REGISTRY`, MMCV, etc.),
|
||||
loading the same model twice in the same process will throw. The
|
||||
evaluator already loads each metric once; if you write a custom
|
||||
setup-then-call pattern, mirror that single-load discipline.
|
||||
|
||||
---
|
||||
|
||||
## 8) Training-time eval: keep evaluators hot, free caches between calls
|
||||
|
||||
When wiring eval into a training loop, the working pattern is:
|
||||
|
||||
1. Construct the `Evaluator` once and attach it to the pipeline
|
||||
(`self._eval = create_evaluator(...)`). Do not recreate it per
|
||||
validation round; that re-pays the model load cost.
|
||||
2. Save validation videos to disk (the diffusion path already does
|
||||
this). Pass paths to `evaluator.evaluate`, not in-memory tensors
|
||||
that share GPU memory with the training model.
|
||||
3. Run validation only on rank 0 of each sequence-parallel group.
|
||||
Gather paths from other ranks and let rank 0 score everything.
|
||||
4. After every `evaluate(...)` call, call
|
||||
`evaluator.release_cuda_memory()` in a `finally` block. That runs
|
||||
`gc.collect()` + `torch.cuda.empty_cache()` +
|
||||
`torch.cuda.ipc_collect()`. The eval model stays loaded; only
|
||||
transient activation buffers from the just-finished call get
|
||||
freed:
|
||||
|
||||
```python
|
||||
for video_path in batch:
|
||||
try:
|
||||
scores = self._eval.evaluate(video=load_video(video_path))
|
||||
finally:
|
||||
self._eval.release_cuda_memory()
|
||||
```
|
||||
|
||||
5. If memory pressure spikes (rare on H200), call
|
||||
`evaluator.unload()` to drop every metric reference and let the
|
||||
GPU memory be GC'd. `unload` is reversible:
|
||||
`evaluator.reload()` rebuilds the same metrics with the original
|
||||
config (re-paying the model load cost). Calling `evaluate`
|
||||
between `unload` and `reload` raises a clear `RuntimeError`.
|
||||
|
||||
For most metrics (sub-1 GB backbones, e.g. CLIP/DINO/RAFT/AMT) the
|
||||
eval model can stay co-resident with the training model in
|
||||
`transformer.eval()` mode without any swap. For larger ones
|
||||
(VideoScore2 at 14 GB), measure first; if it fits on the rank-0 GPU
|
||||
during validation (training model in eval mode means no
|
||||
grads/optimizer updates), keep it hot. If not, `unload` between
|
||||
rounds.
|
||||
|
||||
## 9) Local verification
|
||||
|
||||
Native and library-wrapped metrics: a single-GPU smoke is enough.
|
||||
|
||||
```python
|
||||
import torch
|
||||
from fastvideo.eval import create_evaluator
|
||||
|
||||
ev = create_evaluator(metrics=["<group>.<your_metric>"], device="cuda")
|
||||
video = torch.randn(1, 49, 3, 256, 256, device="cuda").clamp(0, 1)
|
||||
print(ev.evaluate(video=video))
|
||||
```
|
||||
|
||||
Submodule-wrapped metrics: also do a parity check against the
|
||||
upstream once. Clone upstream into a separate venv, run the same
|
||||
video through both, and expect an exact match on bit-deterministic
|
||||
metrics and ≤1% drift on backbone-heavy ones (driven by
|
||||
transformers/torch version differences).
|
||||
|
||||
For quick parity in CI: pin a tiny test video, record expected
|
||||
scores ± tolerance, and add a calibration test under
|
||||
`fastvideo/tests/eval/`.
|
||||
|
||||
---
|
||||
|
||||
## 10) When not to add a metric
|
||||
|
||||
- **Set-vs-set distribution metrics** (FVD, FID-style) do not fit
|
||||
`BaseMetric.compute(sample)` cleanly; they need a population.
|
||||
Adding them requires a stateful accumulator interface that does
|
||||
not exist yet. Open an issue first.
|
||||
- **Metrics requiring a single-GPU model larger than available
|
||||
memory.** Eval is not the place for tensor-parallel sharding;
|
||||
metrics are expected to fit on one GPU.
|
||||
- **Metrics that need `mmcv` with a conflicting CUDA ABI.** Document
|
||||
the affected sub-metrics as unsupported and skip them. Building
|
||||
isolation infrastructure (subprocess engine, per-metric venv) is
|
||||
out of scope.
|
||||
@@ -1,7 +1,7 @@
|
||||
status_definitions:
|
||||
kept: "Public field remains on a public adapter surface with the same meaning."
|
||||
moved: "Public field remains supported but normalizes into a different nested path."
|
||||
profile_owned: "Public field remains supported only through a model/profile-specific surface."
|
||||
preset_owned: "Public field remains supported only through a model/preset-specific surface."
|
||||
compatibility_only: "Legacy public field remains adapter-only during migration and is not part of the canonical typed schema."
|
||||
private_only: "Field should only be handled by private adapters and is not a public FastVideo compatibility promise."
|
||||
internal_only: "Field is runtime/config plumbing and should not be part of the new public typed inference API."
|
||||
@@ -29,24 +29,23 @@ surfaces:
|
||||
vae_cpu_offload: generator.engine.offload.vae
|
||||
pin_cpu_memory: generator.engine.offload.pin_cpu_memory
|
||||
enable_torch_compile: generator.engine.compile.enabled
|
||||
torch_compile_kwargs: generator.engine.compile.kwargs
|
||||
torch_compile_kwargs: generator.engine.compile.backend,fullgraph,mode,dynamic,extras
|
||||
disable_autocast: generator.engine.disable_autocast
|
||||
enable_stage_verification: generator.engine.enable_stage_verification
|
||||
prompt_txt: request.inputs.prompt_path
|
||||
override_text_encoder_safetensors: generator.pipeline.components.text_encoder_weights
|
||||
override_text_encoder_quant: generator.engine.quantization.text_encoder_quant
|
||||
transformer_quant: generator.engine.quantization.transformer_quant
|
||||
override_transformer_cls_name: generator.pipeline.components.override_transformer_cls_name
|
||||
init_weights_from_safetensors: generator.pipeline.components.transformer_weights
|
||||
init_weights_from_safetensors_2: generator.pipeline.components.transformer_2_weights
|
||||
override_pipeline_cls_name: generator.pipeline.components.override_pipeline_cls_name
|
||||
boundary_ratio: request.sampling.boundary_ratio
|
||||
profile_owned:
|
||||
ltx2_vae_tiling: generator.pipeline.profile_overrides.ltx2.vae_tiling
|
||||
ltx2_vae_spatial_tile_size_in_pixels: generator.pipeline.profile_overrides.ltx2.vae.spatial_tile_size_in_pixels
|
||||
ltx2_vae_spatial_tile_overlap_in_pixels: generator.pipeline.profile_overrides.ltx2.vae.spatial_tile_overlap_in_pixels
|
||||
ltx2_vae_temporal_tile_size_in_frames: generator.pipeline.profile_overrides.ltx2.vae.temporal_tile_size_in_frames
|
||||
ltx2_vae_temporal_tile_overlap_in_frames: generator.pipeline.profile_overrides.ltx2.vae.temporal_tile_overlap_in_frames
|
||||
ltx2_vae_tiling: generator.pipeline.vae_tiling
|
||||
preset_owned:
|
||||
ltx2_vae_spatial_tile_size_in_pixels: generator.pipeline.preset_overrides.ltx2.vae.spatial_tile_size_in_pixels
|
||||
ltx2_vae_spatial_tile_overlap_in_pixels: generator.pipeline.preset_overrides.ltx2.vae.spatial_tile_overlap_in_pixels
|
||||
ltx2_vae_temporal_tile_size_in_frames: generator.pipeline.preset_overrides.ltx2.vae.temporal_tile_size_in_frames
|
||||
ltx2_vae_temporal_tile_overlap_in_frames: generator.pipeline.preset_overrides.ltx2.vae.temporal_tile_overlap_in_frames
|
||||
ltx2_initial_latent_path: request.extensions.ltx2.initial_latent_path
|
||||
compatibility_only:
|
||||
mode: "Legacy multi-mode FastVideoArgs switch; typed inference config should not expose execution mode."
|
||||
@@ -70,16 +69,16 @@ surfaces:
|
||||
pipeline_config_base:
|
||||
moved:
|
||||
pipeline_config_path: generator.pipeline.components.pipeline_config_path
|
||||
profile_owned:
|
||||
embedded_cfg_scale: generator.pipeline.profile_overrides.embedded_cfg_scale
|
||||
flow_shift: generator.pipeline.profile_overrides.flow_shift
|
||||
flow_shift_sr: generator.pipeline.profile_overrides.flow_shift_sr
|
||||
is_causal: generator.pipeline.profile_overrides.is_causal
|
||||
vae_tiling: generator.pipeline.profile_overrides.vae_tiling
|
||||
vae_sp: generator.pipeline.profile_overrides.vae_sp
|
||||
dmd_denoising_steps: generator.pipeline.profile_overrides.dmd_denoising_steps
|
||||
ti2v_task: generator.pipeline.profile_overrides.ti2v_task
|
||||
boundary_ratio: generator.pipeline.profile_overrides.boundary_ratio
|
||||
preset_owned:
|
||||
embedded_cfg_scale: generator.pipeline.preset_overrides.embedded_cfg_scale
|
||||
flow_shift: generator.pipeline.preset_overrides.flow_shift
|
||||
flow_shift_sr: generator.pipeline.preset_overrides.flow_shift_sr
|
||||
is_causal: generator.pipeline.preset_overrides.is_causal
|
||||
vae_tiling: generator.pipeline.preset_overrides.vae_tiling
|
||||
vae_sp: generator.pipeline.preset_overrides.vae_sp
|
||||
dmd_denoising_steps: generator.pipeline.preset_overrides.dmd_denoising_steps
|
||||
ti2v_task: generator.pipeline.preset_overrides.ti2v_task
|
||||
boundary_ratio: generator.pipeline.preset_overrides.boundary_ratio
|
||||
compatibility_only:
|
||||
model_path: "Redundant with generator.model_path."
|
||||
disable_autocast: "Duplicated by generator.engine.disable_autocast during migration."
|
||||
@@ -98,7 +97,7 @@ surfaces:
|
||||
postprocess_text_funcs: "Internal text postprocessing hooks."
|
||||
|
||||
pipeline_config_extensions:
|
||||
profile_owned:
|
||||
preset_owned:
|
||||
conditioning_strategy:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
@@ -310,8 +309,8 @@ surfaces:
|
||||
compatibility_only:
|
||||
batch_size: "Gen3C inference-only tuning field pending typed batching design."
|
||||
gradient_checkpointing: "Gen3C inference-only compatibility field pending typed batching design."
|
||||
guidance_scale: "Gen3C pipeline-level default pending profile/default-request cleanup."
|
||||
num_inference_steps: "Gen3C pipeline-level default pending profile/default-request cleanup."
|
||||
guidance_scale: "Gen3C pipeline-level default pending preset/default-request cleanup."
|
||||
num_inference_steps: "Gen3C pipeline-level default pending preset/default-request cleanup."
|
||||
internal_only:
|
||||
audio_decoder_config: "Legacy internal component config object."
|
||||
audio_decoder_precision: "Precision override pending dedicated component precision design."
|
||||
@@ -355,88 +354,36 @@ surfaces:
|
||||
return_frames: request.output.return_frames
|
||||
return_trajectory_latents: request.runtime.return_trajectory_latents
|
||||
return_trajectory_decoded: request.runtime.return_trajectory_decoded
|
||||
profile_owned:
|
||||
continuation_state: request.state
|
||||
return_continuation_state: request.output.return_state
|
||||
preset_owned:
|
||||
t_thresh: request.stage_overrides.refine.t_thresh
|
||||
spatial_refine_only: request.stage_overrides.refine.spatial_refine_only
|
||||
num_cond_frames: request.stage_overrides.refine.num_cond_frames
|
||||
trajectory_type: request.extensions.gen3c.trajectory_type
|
||||
movement_distance: request.extensions.gen3c.movement_distance
|
||||
camera_rotation: request.extensions.gen3c.camera_rotation
|
||||
prompt_attention_mask: request.extensions.hyworld.prompt_attention_mask
|
||||
negative_attention_mask: request.extensions.hyworld.negative_attention_mask
|
||||
camera_states: request.extensions.hunyuangamecraft.camera_states
|
||||
camera_trajectory: request.extensions.hunyuangamecraft.camera_trajectory
|
||||
action_list: request.extensions.hunyuangamecraft.action_list
|
||||
action_speed_list: request.extensions.hunyuangamecraft.action_speed_list
|
||||
gt_latents: request.extensions.hunyuangamecraft.gt_latents
|
||||
conditioning_mask: request.extensions.hunyuangamecraft.conditioning_mask
|
||||
ltx2_cfg_scale_video: request.extensions.ltx2.cfg_scale_video
|
||||
ltx2_cfg_scale_audio: request.extensions.ltx2.cfg_scale_audio
|
||||
ltx2_modality_scale_video: request.extensions.ltx2.modality_scale_video
|
||||
ltx2_modality_scale_audio: request.extensions.ltx2.modality_scale_audio
|
||||
ltx2_rescale_scale: request.extensions.ltx2.rescale_scale
|
||||
ltx2_stg_scale_video: request.extensions.ltx2.stg_scale_video
|
||||
ltx2_stg_scale_audio: request.extensions.ltx2.stg_scale_audio
|
||||
ltx2_stg_blocks_video: request.extensions.ltx2.stg_blocks_video
|
||||
ltx2_stg_blocks_audio: request.extensions.ltx2.stg_blocks_audio
|
||||
internal_only:
|
||||
data_type: "Derived from the request shape and not a public input."
|
||||
|
||||
sampling_param_extensions:
|
||||
moved: {}
|
||||
profile_owned:
|
||||
action_list:
|
||||
target: request.extensions.hunyuangamecraft.action_list
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
action_speed_list:
|
||||
target: request.extensions.hunyuangamecraft.action_speed_list
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
camera_states:
|
||||
target: request.extensions.hunyuangamecraft.camera_states
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
camera_trajectory:
|
||||
target: request.extensions.hunyuangamecraft.camera_trajectory
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
conditioning_mask:
|
||||
target: request.extensions.hunyuangamecraft.conditioning_mask
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
gt_latents:
|
||||
target: request.extensions.hunyuangamecraft.gt_latents
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
prompt_attention_mask:
|
||||
target: request.extensions.hyworld.prompt_attention_mask
|
||||
sources: [fastvideo.configs.sample.hyworld.HYWorld_SamplingParam]
|
||||
negative_attention_mask:
|
||||
target: request.extensions.hyworld.negative_attention_mask
|
||||
sources: [fastvideo.configs.sample.hyworld.HYWorld_SamplingParam]
|
||||
ltx2_cfg_scale_audio:
|
||||
target: request.extensions.ltx2.cfg_scale_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_cfg_scale_video:
|
||||
target: request.extensions.ltx2.cfg_scale_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_modality_scale_audio:
|
||||
target: request.extensions.ltx2.modality_scale_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_modality_scale_video:
|
||||
target: request.extensions.ltx2.modality_scale_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_rescale_scale:
|
||||
target: request.extensions.ltx2.rescale_scale
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_blocks_audio:
|
||||
target: request.extensions.ltx2.stg_blocks_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_blocks_video:
|
||||
target: request.extensions.ltx2.stg_blocks_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_scale_audio:
|
||||
target: request.extensions.ltx2.stg_scale_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_scale_video:
|
||||
target: request.extensions.ltx2.stg_scale_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
sampling_param_extensions: {}
|
||||
|
||||
openai_image_request:
|
||||
kept:
|
||||
@@ -489,232 +436,14 @@ cli:
|
||||
notes:
|
||||
- "CLI parity is checked against the actual generate/serve parser dest sets."
|
||||
- "The inventory tracks parser dest names, excluding argparse's implicit help action."
|
||||
- "The refactored inference CLI is config-only: subcommands expose only --config, and any additional CLI input must use dotted override paths."
|
||||
generate:
|
||||
explicit_local_fields:
|
||||
- config
|
||||
expected_dests:
|
||||
- VSA_sparsity
|
||||
- boundary_ratio
|
||||
- bsa_cdf_threshold
|
||||
- bsa_chunk_k
|
||||
- bsa_chunk_q
|
||||
- bsa_sparsity
|
||||
- config
|
||||
- disable_autocast
|
||||
- dist_timeout
|
||||
- distributed_executor_backend
|
||||
- dit_config.prefix
|
||||
- dit_config.quant_config
|
||||
- dit_cpu_offload
|
||||
- dit_layerwise_offload
|
||||
- dit_precision
|
||||
- dmd_denoising_steps
|
||||
- embedded_cfg_scale
|
||||
- enable_bsa
|
||||
- enable_stage_verification
|
||||
- enable_torch_compile
|
||||
- flow_shift
|
||||
- fps
|
||||
- guidance_rescale
|
||||
- guidance_scale
|
||||
- height
|
||||
- hsdp_replicate_dim
|
||||
- hsdp_shard_dim
|
||||
- image_encoder_cpu_offload
|
||||
- image_encoder_precision
|
||||
- image_path
|
||||
- inference_mode
|
||||
- init_weights_from_safetensors
|
||||
- init_weights_from_safetensors_2
|
||||
- lora_nickname
|
||||
- lora_path
|
||||
- lora_target_modules
|
||||
- ltx2_initial_latent_path
|
||||
- ltx2_vae_spatial_tile_overlap_in_pixels
|
||||
- ltx2_vae_spatial_tile_size_in_pixels
|
||||
- ltx2_vae_temporal_tile_overlap_in_frames
|
||||
- ltx2_vae_temporal_tile_size_in_frames
|
||||
- ltx2_vae_tiling
|
||||
- master_port
|
||||
- moba_config_path
|
||||
- mode
|
||||
- model_path
|
||||
- negative_prompt
|
||||
- num_cond_frames
|
||||
- num_frames
|
||||
- num_gpus
|
||||
- num_inference_steps
|
||||
- num_videos_per_prompt
|
||||
- output_path
|
||||
- output_type
|
||||
- output_video_name
|
||||
- override_pipeline_cls_name
|
||||
- override_text_encoder_quant
|
||||
- override_text_encoder_safetensors
|
||||
- override_transformer_cls_name
|
||||
- pin_cpu_memory
|
||||
- pipeline_config_path
|
||||
- preprocess.dataloader_num_workers
|
||||
- preprocess.dataset_output_dir
|
||||
- preprocess.dataset_path
|
||||
- preprocess.dataset_type
|
||||
- preprocess.do_temporal_sample
|
||||
- preprocess.drop_short_ratio
|
||||
- preprocess.flush_frequency
|
||||
- preprocess.max_height
|
||||
- preprocess.max_width
|
||||
- preprocess.model_path
|
||||
- preprocess.num_frames
|
||||
- preprocess.preprocess_video_batch_size
|
||||
- preprocess.samples_per_file
|
||||
- preprocess.seed
|
||||
- preprocess.speed_factor
|
||||
- preprocess.train_fps
|
||||
- preprocess.training_cfg_rate
|
||||
- preprocess.video_length_tolerance_range
|
||||
- preprocess.video_loader_type
|
||||
- preprocess.with_audio
|
||||
- prompt
|
||||
- prompt_path
|
||||
- prompt_txt
|
||||
- refine_from
|
||||
- return_frames
|
||||
- return_trajectory_decoded
|
||||
- return_trajectory_latents
|
||||
- revision
|
||||
- save_video
|
||||
- seed
|
||||
- sp_size
|
||||
- spatial_refine_only
|
||||
- t_thresh
|
||||
- text_encoder_configs
|
||||
- text_encoder_cpu_offload
|
||||
- text_encoder_precisions
|
||||
- torch_compile_kwargs
|
||||
- transformer_quant
|
||||
- tp_size
|
||||
- trust_remote_code
|
||||
- use_fsdp_inference
|
||||
- vae_config.blend_num_frames
|
||||
- vae_config.load_decoder
|
||||
- vae_config.load_encoder
|
||||
- vae_config.tile_sample_min_height
|
||||
- vae_config.tile_sample_min_num_frames
|
||||
- vae_config.tile_sample_min_width
|
||||
- vae_config.tile_sample_stride_height
|
||||
- vae_config.tile_sample_stride_num_frames
|
||||
- vae_config.tile_sample_stride_width
|
||||
- vae_config.use_parallel_tiling
|
||||
- vae_config.use_temporal_tiling
|
||||
- vae_config.use_tiling
|
||||
- vae_cpu_offload
|
||||
- vae_precision
|
||||
- vae_sp
|
||||
- vae_tiling
|
||||
- video_path
|
||||
- width
|
||||
- workload_type
|
||||
serve:
|
||||
explicit_local_fields:
|
||||
- config
|
||||
- host
|
||||
- output_dir
|
||||
- port
|
||||
expected_dests:
|
||||
- VSA_sparsity
|
||||
- bsa_cdf_threshold
|
||||
- bsa_chunk_k
|
||||
- bsa_chunk_q
|
||||
- bsa_sparsity
|
||||
- config
|
||||
- disable_autocast
|
||||
- dist_timeout
|
||||
- distributed_executor_backend
|
||||
- dit_config.prefix
|
||||
- dit_config.quant_config
|
||||
- dit_cpu_offload
|
||||
- dit_layerwise_offload
|
||||
- dit_precision
|
||||
- dmd_denoising_steps
|
||||
- embedded_cfg_scale
|
||||
- enable_bsa
|
||||
- enable_stage_verification
|
||||
- enable_torch_compile
|
||||
- flow_shift
|
||||
- host
|
||||
- hsdp_replicate_dim
|
||||
- hsdp_shard_dim
|
||||
- image_encoder_cpu_offload
|
||||
- image_encoder_precision
|
||||
- inference_mode
|
||||
- init_weights_from_safetensors
|
||||
- init_weights_from_safetensors_2
|
||||
- lora_nickname
|
||||
- lora_path
|
||||
- lora_target_modules
|
||||
- ltx2_initial_latent_path
|
||||
- ltx2_vae_spatial_tile_overlap_in_pixels
|
||||
- ltx2_vae_spatial_tile_size_in_pixels
|
||||
- ltx2_vae_temporal_tile_overlap_in_frames
|
||||
- ltx2_vae_temporal_tile_size_in_frames
|
||||
- ltx2_vae_tiling
|
||||
- master_port
|
||||
- mode
|
||||
- model_path
|
||||
- num_gpus
|
||||
- output_dir
|
||||
- output_type
|
||||
- override_pipeline_cls_name
|
||||
- override_text_encoder_quant
|
||||
- override_text_encoder_safetensors
|
||||
- override_transformer_cls_name
|
||||
- pin_cpu_memory
|
||||
- pipeline_config_path
|
||||
- port
|
||||
- preprocess.dataloader_num_workers
|
||||
- preprocess.dataset_output_dir
|
||||
- preprocess.dataset_path
|
||||
- preprocess.dataset_type
|
||||
- preprocess.do_temporal_sample
|
||||
- preprocess.drop_short_ratio
|
||||
- preprocess.flush_frequency
|
||||
- preprocess.max_height
|
||||
- preprocess.max_width
|
||||
- preprocess.model_path
|
||||
- preprocess.num_frames
|
||||
- preprocess.preprocess_video_batch_size
|
||||
- preprocess.samples_per_file
|
||||
- preprocess.seed
|
||||
- preprocess.speed_factor
|
||||
- preprocess.train_fps
|
||||
- preprocess.training_cfg_rate
|
||||
- preprocess.video_length_tolerance_range
|
||||
- preprocess.video_loader_type
|
||||
- preprocess.with_audio
|
||||
- prompt_txt
|
||||
- revision
|
||||
- sp_size
|
||||
- text_encoder_cpu_offload
|
||||
- text_encoder_precisions
|
||||
- torch_compile_kwargs
|
||||
- transformer_quant
|
||||
- tp_size
|
||||
- trust_remote_code
|
||||
- use_fsdp_inference
|
||||
- vae_config.blend_num_frames
|
||||
- vae_config.load_decoder
|
||||
- vae_config.load_encoder
|
||||
- vae_config.tile_sample_min_height
|
||||
- vae_config.tile_sample_min_num_frames
|
||||
- vae_config.tile_sample_min_width
|
||||
- vae_config.tile_sample_stride_height
|
||||
- vae_config.tile_sample_stride_num_frames
|
||||
- vae_config.tile_sample_stride_width
|
||||
- vae_config.use_parallel_tiling
|
||||
- vae_config.use_temporal_tiling
|
||||
- vae_config.use_tiling
|
||||
- vae_cpu_offload
|
||||
- vae_precision
|
||||
- vae_sp
|
||||
- vae_tiling
|
||||
- workload_type
|
||||
|
||||
+6
-10
@@ -12,7 +12,7 @@ FastVideo maps a Diffusers-style repo into a pipeline like this:
|
||||
- `fastvideo/configs/models/*`: arch configs and `param_names_mapping` for
|
||||
weight name translation.
|
||||
- `fastvideo/configs/pipelines/*`: pipeline wiring (component classes + names).
|
||||
- `fastvideo/configs/sample/*`: default runtime sampling parameters.
|
||||
- `fastvideo/api/sampling_param.py`: runtime sampling parameters.
|
||||
- `fastvideo/pipelines/basic/*`: end-to-end pipelines.
|
||||
- `fastvideo/pipelines/stages/*`: reusable pipeline stages.
|
||||
- `fastvideo/models/loader/*`: component loaders for Diffusers-style repos.
|
||||
@@ -26,7 +26,7 @@ Minimal usage (from `examples/inference/basic/basic.py`):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # or official_weights/<model_name>/
|
||||
generator = VideoGenerator.from_pretrained(model_id, num_gpus=1)
|
||||
@@ -49,8 +49,9 @@ runtime parameters consistent:
|
||||
- `fastvideo/configs/models/`: architecture definitions, layer shapes, and
|
||||
`param_names_mapping` rules for key renaming.
|
||||
- `fastvideo/configs/pipelines/`: pipeline wiring and required components.
|
||||
- `fastvideo/configs/sample/`: default sampling parameters (steps, frames,
|
||||
guidance scale, resolution, fps).
|
||||
- `fastvideo/api/sampling_param.py`: sampling parameters (steps, frames,
|
||||
guidance scale, resolution, fps). Defaults come from profiles in
|
||||
`fastvideo/pipelines/basic/<family>/profiles.py`.
|
||||
- `fastvideo/registry.py`: unified registry for pipeline config + sampling
|
||||
defaults and model metadata resolution, defined via explicit
|
||||
`register_configs(...)` blocks (no separate dict registries).
|
||||
@@ -142,7 +143,7 @@ How this maps to FastVideo:
|
||||
- `T5TokenizerFast` -> loaded via HF in `fastvideo/models/loader/`
|
||||
- `UniPCMultistepScheduler` -> loaded via Diffusers scheduler utilities
|
||||
- Pipeline defaults -> `fastvideo/configs/pipelines/wan.py`
|
||||
- Sampling defaults -> `fastvideo/configs/sample/wan.py`
|
||||
- Sampling defaults -> `fastvideo/pipelines/basic/wan/profiles.py`
|
||||
|
||||
## Pipeline system
|
||||
|
||||
@@ -167,11 +168,6 @@ How this maps to FastVideo:
|
||||
|
||||
- Attention backends live in `fastvideo/attention/` and can be selected via
|
||||
`FASTVIDEO_ATTENTION_BACKEND`.
|
||||
- SageAttention3 is split into two selectable backends:
|
||||
`SAGE_ATTN_THREE` for the regular upstream package and
|
||||
`ATTN_QAT_INFER` for the FastVideoKernel-backed inference variant.
|
||||
- `ATTN_QAT_TRAIN` is a separate FastVideoKernel Triton backend for the QAT attention
|
||||
path.
|
||||
- `LocalAttention` is used for cross-attention and most attention layers.
|
||||
- `DistributedAttention` is used for full-sequence self-attention in the DiT.
|
||||
- Tensor-parallel layers live in `fastvideo/layers/`.
|
||||
|
||||
@@ -0,0 +1,177 @@
|
||||
# Streaming WebSocket Server Contract
|
||||
|
||||
The streaming server (`fastvideo/entrypoints/streaming/server.py`) speaks
|
||||
a JSON-over-WebSocket protocol with binary fMP4 chunks for media. This
|
||||
document is the authoritative spec for the message catalogue and the
|
||||
session state machine. Any change to either must update this document
|
||||
in the same PR that touches `protocol.py` or `session.py`.
|
||||
|
||||
## Endpoint
|
||||
|
||||
| Path | Protocol | Purpose |
|
||||
|---|---|---|
|
||||
| `WS /v1/stream` | WebSocket (JSON + binary) | Per-session realtime streaming |
|
||||
| `GET /health` | HTTP | Liveness probe (`status`, `stream_mode`, active `sessions`) |
|
||||
|
||||
The server is launched by `fastvideo serve --config <serve.yaml>` when
|
||||
the config carries a `streaming:` block. Without that block the same CLI
|
||||
launches the OpenAI stateless HTTP server instead.
|
||||
|
||||
## Connection lifecycle
|
||||
|
||||
Every WebSocket connection holds exactly one `Session`. Sessions move
|
||||
through the states in `SessionState` (`fastvideo/entrypoints/streaming/session.py`).
|
||||
|
||||
```
|
||||
┌──────────────┐
|
||||
│ INITIALIZING │ ← WebSocket accepted, before init frame
|
||||
└──────┬───────┘
|
||||
│ session_init_v2 received
|
||||
┌──────────────┼──────────────┐
|
||||
▼ ▼ ▼
|
||||
QUEUED GPU_BINDING REJECTED
|
||||
│ │ ↑
|
||||
│ slot ready │ │ max-sessions hit
|
||||
▼ ▼ │ or invalid init
|
||||
┌────────┐ │
|
||||
│ ACTIVE │ ────────┘
|
||||
└────┬───┘
|
||||
segment loop │
|
||||
│
|
||||
┌───────────┼───────────┐
|
||||
▼ ▼ ▼
|
||||
COMPLETE ERROR TIMEOUT
|
||||
(clean leave) (any failure) (idle / segment_cap reached)
|
||||
```
|
||||
|
||||
Terminal states (`COMPLETE`, `ERROR`, `TIMEOUT`, `REJECTED`) are sinks —
|
||||
no transitions out. The transition matrix is enforced in
|
||||
`session.py::_VALID_TRANSITIONS`; bad transitions raise.
|
||||
|
||||
`SessionManager` enforces the per-process budgets pulled from
|
||||
`StreamingConfig`:
|
||||
|
||||
- `session_timeout_seconds` — idle reaper drops sessions that haven't
|
||||
advanced; non-terminal sessions transition to `TIMEOUT`.
|
||||
- `generation_segment_cap` — a session that hits the cap transitions to
|
||||
`COMPLETE` after the last segment ships.
|
||||
|
||||
## Message catalogue
|
||||
|
||||
Every JSON frame carries `{"type": <str>, ...}`. Pydantic models in
|
||||
`protocol.py` are the source of truth; this table is the human-readable
|
||||
view.
|
||||
|
||||
### Client → server
|
||||
|
||||
| `type` | Required fields | Purpose |
|
||||
|---|---|---|
|
||||
| `session_init_v2` | — | Opening frame. Carries preset, curated prompts, optional initial image, feature toggles, optional `continuation_state` to resume from a snapshot. |
|
||||
| `segment_prompt_source` | `prompt` | Request the next segment using the supplied prompt; optional sampling overrides (`seed`, `num_inference_steps`, `guidance_scale`, `negative_prompt`). |
|
||||
| `seed_prompts_updated` | `seed_prompts` | Replace the session's seed-prompt list; takes effect on the next segment. |
|
||||
| `enhancement_updated` | `enabled` | Toggle prompt enhancement for subsequent segments. |
|
||||
| `auto_extension_updated` | `enabled` | Toggle automatic per-segment prompt extension. |
|
||||
| `loop_generation_updated` | `enabled` | Toggle loop-generation mode. |
|
||||
| `generation_paused_updated` | `paused` | Pause/resume segment generation; queued requests defer. |
|
||||
| `snapshot_state` | — | Request the current `ContinuationState` for export; server replies with `continuation_state_snapshot`. |
|
||||
|
||||
The opening frame must be `session_init_v2`. Any other first frame is
|
||||
rejected with an `error` (code `invalid_message`) and the WebSocket is
|
||||
closed.
|
||||
|
||||
### Server → client
|
||||
|
||||
| `type` | Carries | When emitted |
|
||||
|---|---|---|
|
||||
| `queue_status` | `position`, `queue_depth` | After `session_init_v2` accepted, before GPU binding. |
|
||||
| `gpu_assigned` | GPU id, model id | Once a generator slot is bound. |
|
||||
| `ltx2_stream_start` | session-level metadata | Once the session enters `ACTIVE`. |
|
||||
| `ltx2_segment_start` | `segment_idx`, `prompt`, prompt source | When a `segment_prompt_source` request begins generation. |
|
||||
| `step_complete` | `segment_idx`, denoise timings | After the segment's denoising loop finishes (before media emission). |
|
||||
| `media_init` | `segment_idx`, mime, stream id | First frame of fMP4 output for the segment. |
|
||||
| binary frame | fMP4 fragment bytes | Subsequent media chunks; the protocol enforces that `media_init` precedes any binary frames. |
|
||||
| `media_segment_complete` | `segment_idx`, chunk count, byte count | Last media chunk for the segment. |
|
||||
| `ltx2_segment_complete` | `segment_idx`, segment summary | Segment fully shipped; ready for the next `segment_prompt_source`. |
|
||||
| `ltx2_stream_complete` | session summary | Session reached `generation_segment_cap` or client requested clean shutdown. |
|
||||
| `session_timeout` | reason | Session hit `session_timeout_seconds`; immediately followed by close. |
|
||||
| `continuation_state_snapshot` | `kind`, `payload` | Reply to `snapshot_state`. The payload is the same shape produced by `LTX2ContinuationState.to_continuation_state(...)`. |
|
||||
| `error` | `code`, `message` | Any validation/runtime error. Non-fatal errors keep the connection open; fatal errors precede a `close`. |
|
||||
|
||||
## Continuation state
|
||||
|
||||
The session optionally accepts a `continuation_state` dict inside the
|
||||
opening `session_init_v2` frame. When present, the server hydrates it
|
||||
into a `ContinuationState(kind, payload)` envelope and feeds it as the
|
||||
`request.state` on the first segment's `GenerationRequest` — letting a
|
||||
client resume after a disconnect, migrate sessions across processes,
|
||||
or replay a prior session.
|
||||
|
||||
After every segment, if the runtime returns a fresh state, the server
|
||||
persists it to the `SessionStore` so a `snapshot_state` request can
|
||||
export it. The store and serialization contracts live with the model
|
||||
family (e.g. `fastvideo/pipelines/basic/ltx2/continuation.py` for LTX-2).
|
||||
|
||||
## Example flow
|
||||
|
||||
```
|
||||
client server
|
||||
────── ──────
|
||||
WS /v1/stream ─────── connect ─────────────────────────►
|
||||
◄────── (accept)
|
||||
|
||||
{"type": "session_init_v2",
|
||||
"preset": "ltx2_two_stage",
|
||||
"curated_prompts": ["a fox in snow", "the fox jumps"],
|
||||
"initial_image": {...},
|
||||
"stream_mode": "av_fmp4"} ─────────────────────────────►
|
||||
|
||||
(validate, queue, bind)
|
||||
◄──── {"type": "queue_status",
|
||||
"position": 0, "queue_depth": 0}
|
||||
◄──── {"type": "gpu_assigned",
|
||||
"gpu_id": 0, "model_id": "..."}
|
||||
◄──── {"type": "ltx2_stream_start", ...}
|
||||
|
||||
{"type": "segment_prompt_source",
|
||||
"prompt": "a fox in snow",
|
||||
"source": "curated"} ───────────────────────────────────►
|
||||
(run pipeline)
|
||||
◄──── {"type": "ltx2_segment_start",
|
||||
"segment_idx": 1, ...}
|
||||
◄──── {"type": "step_complete",
|
||||
"segment_idx": 1, "timings": {...}}
|
||||
◄──── {"type": "media_init",
|
||||
"segment_idx": 1,
|
||||
"mime": "video/mp4", ...}
|
||||
◄──── <binary fMP4 init segment>
|
||||
◄──── <binary fMP4 fragment>
|
||||
◄──── <binary fMP4 fragment>
|
||||
◄──── {"type": "media_segment_complete",
|
||||
"segment_idx": 1, "chunks": 12}
|
||||
◄──── {"type": "ltx2_segment_complete",
|
||||
"segment_idx": 1, ...}
|
||||
|
||||
{"type": "segment_prompt_source",
|
||||
"prompt": "the fox jumps"} ─────────────────────────────►
|
||||
(segment 2 …)
|
||||
|
||||
{"type": "snapshot_state"} ──────────────────────────────►
|
||||
◄──── {"type": "continuation_state_snapshot",
|
||||
"kind": "ltx2.v1",
|
||||
"payload": {"schema_version": 1, ...}}
|
||||
|
||||
(close) ──────────────────────────────────────────────────►
|
||||
(session → COMPLETE)
|
||||
```
|
||||
|
||||
## Backward / forward compatibility
|
||||
|
||||
- Adding a new client message: append a Pydantic model to `protocol.py`
|
||||
with a unique `type`; add the discriminator entry to `ClientMessage`;
|
||||
add a row to the table above. Old clients that don't send the new
|
||||
message remain compatible.
|
||||
- Adding a new server message: emit only when a new feature flag is
|
||||
enabled (or always emit, since clients ignore unknown types).
|
||||
- Changing an existing message: bump the `type` (e.g. `session_init_v2`
|
||||
→ `session_init_v3`) and accept both for one release cycle. Never
|
||||
silently change field semantics under the same `type`.
|
||||
@@ -16,7 +16,8 @@ Both models are trained on **61×448×832** resolution but support generating vi
|
||||
First install [VSA](../attention/vsa/index.md). Set `MODEL_BASE` to your own model path and run:
|
||||
|
||||
```bash
|
||||
bash scripts/inference/v1_inference_wan_dmd.sh
|
||||
FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN \
|
||||
fastvideo generate --config scripts/inference/inference_wan_VSA_DMD_1_3B.yaml
|
||||
```
|
||||
|
||||
## 🗂️ Dataset
|
||||
@@ -85,3 +86,25 @@ sbatch examples/distill/Wan2.2-TI2V-5B-Diffusers/Data-free/distill_dmd_t2v_5B.sh
|
||||
- Learning rate: 2e-5
|
||||
- Training steps: 3000 (~12 hours)
|
||||
- HSDP shard dim: 1
|
||||
|
||||
## 🧭 Note on `real_score_guidance_scale`
|
||||
|
||||
The teacher CFG used inside the DMD loss follows the DMD2 reference
|
||||
implementation and uses the parameterization
|
||||
|
||||
```
|
||||
x = x_cond + w * (x_cond - x_uncond)
|
||||
```
|
||||
|
||||
rather than the Ho & Salimans form `x_uncond + w * (x_cond - x_uncond)`. The
|
||||
two are mathematically equivalent up to a constant offset:
|
||||
|
||||
| `real_score_guidance_scale` (`w`) | Equivalent standard CFG (`w + 1`) | Output |
|
||||
|-----------------------------------|-----------------------------------|-----------------------|
|
||||
| `-1` | `0` | unconditional |
|
||||
| `0` | `1` | conditional |
|
||||
| `3.5` (default) | `4.5` | strong guidance |
|
||||
|
||||
So `real_score_guidance_scale` should be read as the **extra** guidance
|
||||
strength added on top of the conditional prediction. When porting values
|
||||
from a paper that uses the Ho & Salimans form, subtract 1.
|
||||
|
||||
@@ -33,7 +33,7 @@ The following two classes `PipelineConfig` and `SamplingParam` are used to confi
|
||||
|
||||
### SamplingParam
|
||||
|
||||
::: fastvideo.configs.sample.base.SamplingParam
|
||||
::: fastvideo.api.sampling_param.SamplingParam
|
||||
options:
|
||||
show_root_heading: true
|
||||
show_source: false
|
||||
|
||||
@@ -128,19 +128,14 @@ Concrete hierarchy: `DiTConfig` → `DiTArchConfig`, `VAEConfig` →
|
||||
- `dump_to_json()` / `load_from_json()` — JSON persistence. Callable
|
||||
fields and `arch_config` are excluded from dumps.
|
||||
|
||||
### SamplingParam (`fastvideo/configs/sample/`)
|
||||
### SamplingParam (`fastvideo/api/sampling_param.py`)
|
||||
|
||||
Generation parameters separate from pipeline config. Each model family
|
||||
provides defaults:
|
||||
provides defaults via a profile (see `fastvideo/pipelines/basic/<family>/profiles.py`):
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class WanT2V_1_3B_SamplingParam(SamplingParam):
|
||||
height: int = 480
|
||||
width: int = 832
|
||||
num_frames: int = 81
|
||||
guidance_scale: float = 3.0
|
||||
num_inference_steps: int = 50
|
||||
sp = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
# sp.height == 480, sp.width == 832, sp.num_frames == 81, etc.
|
||||
```
|
||||
|
||||
## Component Loading
|
||||
@@ -430,9 +425,9 @@ User: generator.generate_video(prompt, ...)
|
||||
`fastvideo/configs/pipelines/<model>.py`. Set DiT/VAE/encoder configs,
|
||||
flow_shift, precision defaults.
|
||||
|
||||
2. **Sampling param** — Create a `SamplingParam` subclass in
|
||||
`fastvideo/configs/sample/<model>.py`. Set default height, width,
|
||||
num_frames, guidance_scale, num_inference_steps.
|
||||
2. **Sampling param profile** — Create a profile in
|
||||
`fastvideo/pipelines/basic/<model>/profiles.py` with default height,
|
||||
width, num_frames, guidance_scale, num_inference_steps.
|
||||
|
||||
3. **Register configs** — In `fastvideo/registry.py`, add a
|
||||
`register_configs()` call inside `_register_configs()` with
|
||||
@@ -455,6 +450,6 @@ User: generator.generate_video(prompt, ...)
|
||||
`fastvideo/pipelines/stages/`, implement `forward()`, optionally
|
||||
implement `verify_input()`/`verify_output()`.
|
||||
|
||||
7. **Verify** — Run `fastvideo generate --model-path <path> --prompt
|
||||
"test" --num-inference-steps 2` to confirm the pipeline loads and
|
||||
generates output.
|
||||
7. **Verify** — Run `fastvideo generate --config <config.yaml>` with a
|
||||
minimal nested config to confirm the pipeline loads and generates
|
||||
output.
|
||||
|
||||
+42
-81
@@ -1,71 +1,29 @@
|
||||
# FastVideo CLI Inference
|
||||
|
||||
The FastVideo CLI exposes the same core inference controls as the Python API.
|
||||
The FastVideo CLI is config-first. Inference runs are driven by a nested JSON or
|
||||
YAML config, with optional dotted-path overrides on the command line. The
|
||||
contract matches training: use an explicit subcommand plus `--config`, then add
|
||||
any dotted overrides you need.
|
||||
|
||||
## Basic Usage
|
||||
|
||||
Use either:
|
||||
|
||||
1. `--model-path` + `--prompt`
|
||||
2. `--model-path` + `--prompt-txt` (batch prompts, one line per prompt)
|
||||
3. `--config` (JSON/YAML)
|
||||
|
||||
```bash
|
||||
fastvideo generate --model-path Wan-AI/Wan2.1-T2V-1.3B-Diffusers \
|
||||
--prompt "A cat playing with a ball of yarn"
|
||||
fastvideo generate --config config.yaml
|
||||
fastvideo serve --config serve.yaml
|
||||
```
|
||||
|
||||
```bash
|
||||
fastvideo generate --model-path Wan-AI/Wan2.1-T2V-1.3B-Diffusers \
|
||||
--prompt-txt prompts.txt
|
||||
```
|
||||
|
||||
You cannot provide both `--prompt` and `--prompt-txt` in the same run.
|
||||
|
||||
## View All Arguments
|
||||
|
||||
```bash
|
||||
fastvideo generate --help
|
||||
```
|
||||
|
||||
Arguments come from:
|
||||
The subcommands intentionally expose only `--config`. Any per-run CLI changes
|
||||
must use dotted override paths such as:
|
||||
|
||||
- FastVideo runtime args (`FastVideoArgs`)
|
||||
- Sampling args (`SamplingParam`)
|
||||
- Pipeline config args (`PipelineConfig`)
|
||||
|
||||
## Common Arguments
|
||||
|
||||
### Parallelism
|
||||
|
||||
- `--num-gpus`
|
||||
- `--sp-size`
|
||||
- `--tp-size`
|
||||
|
||||
### Sampling
|
||||
|
||||
- `--num-frames`
|
||||
- `--height` / `--width`
|
||||
- `--num-inference-steps`
|
||||
- `--guidance-scale`
|
||||
- `--seed`
|
||||
- `--negative-prompt`
|
||||
|
||||
### Output
|
||||
|
||||
- `--output-path`
|
||||
- `--save-video` / `--no-save-video`
|
||||
- `--return-frames`
|
||||
|
||||
### Offloading and Performance
|
||||
|
||||
- `--dit-layerwise-offload`
|
||||
- `--use-fsdp-inference`
|
||||
- `--text-encoder-cpu-offload`
|
||||
- `--image-encoder-cpu-offload`
|
||||
- `--vae-cpu-offload`
|
||||
- `--enable-torch-compile`
|
||||
- `--torch-compile-kwargs`
|
||||
- `--generator.engine.num_gpus 2`
|
||||
- `--request.sampling.seed 42`
|
||||
- `--server.port 9000`
|
||||
|
||||
## Using Config Files
|
||||
|
||||
@@ -73,50 +31,53 @@ Arguments come from:
|
||||
fastvideo generate --config config.yaml
|
||||
```
|
||||
|
||||
Config files can be JSON or YAML. CLI flags override config-file values.
|
||||
Config files can be JSON or YAML. Dotted CLI overrides take precedence over
|
||||
config-file values.
|
||||
|
||||
Example `config.yaml`:
|
||||
|
||||
```yaml
|
||||
model_path: "FastVideo/FastHunyuan-diffusers"
|
||||
prompt: "A capybara lounging in a hammock"
|
||||
output_path: "outputs/"
|
||||
num_gpus: 2
|
||||
sp_size: 2
|
||||
tp_size: 1
|
||||
num_frames: 45
|
||||
height: 720
|
||||
width: 1280
|
||||
num_inference_steps: 6
|
||||
seed: 1024
|
||||
dit_precision: "bf16"
|
||||
vae_precision: "fp16"
|
||||
vae_tiling: true
|
||||
vae_sp: true
|
||||
enable_torch_compile: false
|
||||
generator:
|
||||
model_path: FastVideo/FastHunyuan-diffusers
|
||||
engine:
|
||||
num_gpus: 2
|
||||
parallelism:
|
||||
sp_size: 2
|
||||
tp_size: 1
|
||||
request:
|
||||
prompt: A capybara lounging in a hammock
|
||||
sampling:
|
||||
num_frames: 45
|
||||
height: 720
|
||||
width: 1280
|
||||
num_inference_steps: 6
|
||||
seed: 1024
|
||||
output:
|
||||
output_path: outputs/
|
||||
```
|
||||
|
||||
Notes:
|
||||
|
||||
- Use `dit_precision` / `vae_precision` (not `precision`).
|
||||
- Nested config objects are supported, for example `vae_config` and
|
||||
`dit_config`.
|
||||
- `generator` and `request` are the top-level keys for generation configs.
|
||||
- `serve` configs use `generator`, `server`, and optional `default_request`.
|
||||
- Prompt text files belong under `request.inputs.prompt_path`.
|
||||
|
||||
## Examples
|
||||
|
||||
Simple generation:
|
||||
|
||||
```bash
|
||||
fastvideo generate \
|
||||
--model-path FastVideo/FastHunyuan-diffusers \
|
||||
--prompt "A cat playing with a ball of yarn" \
|
||||
--num-frames 45 --height 720 --width 1280 \
|
||||
--num-inference-steps 6 --seed 1024 \
|
||||
--output-path outputs/
|
||||
fastvideo generate --config config.yaml
|
||||
```
|
||||
|
||||
Config + CLI override:
|
||||
Config + dotted override:
|
||||
|
||||
```bash
|
||||
fastvideo generate --config config.yaml --prompt "A panda skiing at sunset"
|
||||
fastvideo generate --config config.yaml --request.prompt "A panda skiing at sunset"
|
||||
```
|
||||
|
||||
Helper wrapper with positional config path:
|
||||
|
||||
```bash
|
||||
bash scripts/inference/run.sh scripts/inference/inference_wan.yaml
|
||||
```
|
||||
|
||||
@@ -73,32 +73,40 @@ if __name__ == '__main__':
|
||||
|
||||
## JSON/YAML Config Files (CLI)
|
||||
|
||||
The CLI supports `--config` with JSON or YAML. Command-line arguments override
|
||||
config file values.
|
||||
By default, `fastvideo generate` uses `return_frames=false` unless you set
|
||||
`--return-frames` (or `return_frames: true` in config).
|
||||
The inference CLI is config-first. Use an explicit subcommand with `--config`,
|
||||
then apply optional dotted overrides on top, matching the training CLI style.
|
||||
By default, CLI generation uses `return_frames=false` unless you set
|
||||
`request.output.return_frames: true` in config or via a dotted override.
|
||||
|
||||
```bash
|
||||
fastvideo generate --config config.yaml
|
||||
```
|
||||
|
||||
Use CLI argument names as keys (underscore or hyphen is accepted). Example:
|
||||
Example nested config:
|
||||
|
||||
```yaml
|
||||
model_path: "FastVideo/FastHunyuan-diffusers"
|
||||
prompt: "A capybara relaxing in a hammock"
|
||||
num_gpus: 2
|
||||
sp_size: 2
|
||||
num_frames: 45
|
||||
height: 720
|
||||
width: 1280
|
||||
num_inference_steps: 6
|
||||
seed: 1024
|
||||
dit_precision: "bf16"
|
||||
vae_precision: "fp16"
|
||||
vae_tiling: true
|
||||
vae_sp: true
|
||||
enable_torch_compile: false
|
||||
generator:
|
||||
model_path: FastVideo/FastHunyuan-diffusers
|
||||
engine:
|
||||
num_gpus: 2
|
||||
parallelism:
|
||||
sp_size: 2
|
||||
request:
|
||||
prompt: A capybara relaxing in a hammock
|
||||
sampling:
|
||||
num_frames: 45
|
||||
height: 720
|
||||
width: 1280
|
||||
num_inference_steps: 6
|
||||
seed: 1024
|
||||
output:
|
||||
output_path: outputs/
|
||||
```
|
||||
|
||||
Override individual values from the CLI with dotted paths:
|
||||
|
||||
```bash
|
||||
fastvideo generate --config config.yaml --request.sampling.seed 42
|
||||
```
|
||||
|
||||
## Performance Optimization
|
||||
|
||||
@@ -89,7 +89,7 @@ GEN3C defaults in FastVideo:
|
||||
|
||||
These values are defined in:
|
||||
|
||||
- `fastvideo/configs/sample/gen3c.py`
|
||||
- `fastvideo/pipelines/basic/gen3c/profiles.py`
|
||||
- `fastvideo/configs/pipelines/gen3c.py`
|
||||
|
||||
and align with the official GEN3C inference defaults in:
|
||||
|
||||
@@ -107,8 +107,6 @@ If you encounter CUDA out of memory errors:
|
||||
(single GPU) or `use_fsdp_inference=True` (multi-GPU)
|
||||
- Try a smaller model or use distilled versions
|
||||
- Use `num_gpus` > 1 if multiple GPUs are available
|
||||
- Try enabling FSDP inference with `use_fsdp_inference=True` (may slow down generation)
|
||||
- Try enabling DiT layerwise offload with `dit_layerwise_offload=True` (now only a few models support this, but may introduce less overhead than FSDP)
|
||||
|
||||
### Slow Generation
|
||||
|
||||
|
||||
@@ -21,8 +21,6 @@ This page describes the various options for speeding up generation times in Fast
|
||||
- Video Sparse Attention: `FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN`
|
||||
- Sage Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN`
|
||||
- Sage Attention 3: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN_THREE`
|
||||
- Attn QAT Infer: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER`
|
||||
- Attn QAT Train: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN`
|
||||
- Video MoBA Attention: `FASTVIDEO_ATTENTION_BACKEND=VMOBA_ATTN`
|
||||
- Sparse Linear Attention: `FASTVIDEO_ATTENTION_BACKEND=SLA_ATTN`
|
||||
- SageSLA Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_SLA_ATTN`
|
||||
@@ -105,14 +103,6 @@ python setup.py install # or pip install -e .
|
||||
|
||||
### Sage Attention 3
|
||||
|
||||
FastVideo now exposes two SageAttention3-compatible backends with distinct
|
||||
environment variable values:
|
||||
|
||||
- `SAGE_ATTN_THREE`: the regular upstream SageAttention3 backend imported from
|
||||
the `sageattn3` package.
|
||||
- `ATTN_QAT_INFER`: the inference CUDA-kernel backend imported from the
|
||||
in-repo `attn_qat_infer` package.
|
||||
|
||||
**`SAGE_ATTN_THREE`**
|
||||
|
||||
[SageAttention 3](https://github.com/thu-ml/SageAttention/tree/main/sageattention3_blackwell) is an advanced attention mechanism that leverages FP4 quantization and Blackwell GPU Tensor Cores for significant performance improvements.
|
||||
@@ -127,53 +117,6 @@ Note that Sage Attention 3 requires `python>=3.13`, `torch>=2.8.0`, `CUDA >=12.8
|
||||
|
||||
To use Sage Attention 3 in FastVideo, follow the `README.md` in the linked repository to install the package from source.
|
||||
|
||||
### Attn QAT Infer
|
||||
|
||||
**`ATTN_QAT_INFER`**
|
||||
|
||||
This backend uses the `attn_qat_infer` implementation that lives in the
|
||||
`fastvideo-kernel` repository alongside the `fastvideo_kernel` Triton kernels.
|
||||
Use this backend when you want to run the dedicated FP4 inference CUDA kernel
|
||||
directly during inference.
|
||||
|
||||
For the full Attention QAT guide, including Wan 2.1 14B checkpoint download,
|
||||
example editing steps, training launchers, and troubleshooting, see
|
||||
[Attention QAT](../attention/attn_qat/index.md).
|
||||
|
||||
This backend currently assumes access to the in-repo `fastvideo-kernel`
|
||||
checkout or an equivalent editable/source install that exposes:
|
||||
|
||||
- `attn_qat_infer`
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
```
|
||||
|
||||
### QAT Attention
|
||||
|
||||
**`ATTN_QAT_TRAIN`**
|
||||
|
||||
This backend uses the FastVideoKernel Triton attention implementation from
|
||||
`fastvideo_kernel.triton_kernels.attn_qat_train`. Use it when you specifically
|
||||
want the training-oriented Triton attention path rather than the
|
||||
`attn_qat_infer` CUDA kernel path.
|
||||
|
||||
The dedicated [Attention QAT](../attention/attn_qat/index.md) page covers when
|
||||
to use `ATTN_QAT_TRAIN` versus `ATTN_QAT_INFER`, the ready-made training
|
||||
launchers, and the end-to-end Wan 2.1 14B inference workflow.
|
||||
|
||||
This backend currently assumes access to an install that exposes:
|
||||
|
||||
- `fastvideo_kernel`
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_TRAIN"
|
||||
```
|
||||
|
||||
### V-MoBA / SLA / SageSLA
|
||||
|
||||
These backends are model-specific and require the corresponding kernels and
|
||||
|
||||
@@ -27,8 +27,7 @@ Useful variables:
|
||||
- `FASTVIDEO_LOGGING_LEVEL`: `DEBUG`, `INFO`, `WARNING`, `ERROR`
|
||||
- `FASTVIDEO_STAGE_LOGGING`: print per-stage timings during pipeline execution
|
||||
- `FASTVIDEO_ATTENTION_BACKEND`: force an attention backend (for example
|
||||
`TORCH_SDPA`, `FLASH_ATTN`, `SAGE_ATTN_THREE`, or
|
||||
`ATTN_QAT_INFER`, or `ATTN_QAT_TRAIN`)
|
||||
`TORCH_SDPA` or `FLASH_ATTN`)
|
||||
|
||||
## Common Failure Modes
|
||||
|
||||
@@ -53,11 +52,7 @@ If forcing a backend fails, verify optional dependencies are installed:
|
||||
- `VIDEO_SPARSE_ATTN`: `fastvideo-kernel`
|
||||
- `SLIDING_TILE_ATTN`: STA legacy workflow in
|
||||
`sta_do_not_delete` + `fastvideo-kernel`
|
||||
- `SAGE_ATTN`: SageAttention package
|
||||
- `SAGE_ATTN_THREE`: upstream `sageattn3` package
|
||||
- `ATTN_QAT_INFER`: `fastvideo-kernel` checkout/source install that exposes
|
||||
`attn_qat_infer`
|
||||
- `ATTN_QAT_TRAIN`: `fastvideo-kernel` install exposing `fastvideo_kernel`
|
||||
- `SAGE_ATTN` / `SAGE_ATTN_THREE`: SageAttention packages
|
||||
|
||||
As a fallback, use:
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@ export NODE_RANK=$SLURM_PROCID
|
||||
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
|
||||
export MASTER_ADDR=${nodes[0]}
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
export WANDB_API_KEY="2f25ad37933894dbf0966c838c0b8494987f9f2f"
|
||||
# export WANDB_API_KEY='your_wandb_api_key_here'
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
|
||||
@@ -9,7 +9,7 @@ pip install vsa
|
||||
|
||||
### 1. Download dataset:
|
||||
```bash
|
||||
bash examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/download_dataset.sh
|
||||
bash examples/distill/Wan-Syn-480P/download_dataset.sh
|
||||
```
|
||||
|
||||
### 2. Configure and run distillation:
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
#!/bin/bash
|
||||
mkdir -p data
|
||||
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "data/Wan-Syn_77x448x832_600k" --repo_type "dataset"
|
||||
|
||||
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "FastVideo/Wan-Syn_77x448x832_600k" --repo_type "dataset"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
def main():
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
|
||||
def main():
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
|
||||
def main():
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
|
||||
def main():
|
||||
|
||||
@@ -2,7 +2,7 @@ import os
|
||||
import time
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_dmd2"
|
||||
def main():
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
import json
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_hy15"
|
||||
def main():
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
import json
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_hy15_1080p"
|
||||
def main():
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.models.dits.lingbotworld.cam_utils import prepare_camera_embedding
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
OUTPUT_PATH = "video_samples_lingbotworld"
|
||||
def main():
|
||||
# FastVideo will automatically use the optimal default arguments for the
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
from fastvideo import VideoGenerator, PipelineConfig
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
def main():
|
||||
config = PipelineConfig.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
def main():
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
from fastvideo import VideoGenerator, SamplingParam
|
||||
import json
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_self_forcing_causal_wan2_2_14B_i2v"
|
||||
def main():
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_self_forcing_causal_wan2_2_14B_t2v"
|
||||
def main():
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_wan2_2_14B_t2v"
|
||||
def main():
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_wan2_1_Fun"
|
||||
OUTPUT_NAME = "wan2.1_test"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
# from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples_wan2_2_14B_i2v"
|
||||
def main():
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
"""Generate one LTX2 video and score it with VBench metrics.
|
||||
|
||||
The generation block is the same as
|
||||
``examples/inference/basic/basic_ltx2.py`` — same prompt, same model,
|
||||
same shape, same num_frames. After ``shutdown()`` the script loads the
|
||||
mp4 back, builds a single :class:`fastvideo.eval.Evaluator`, and runs
|
||||
the prompt-aware VBench subset that's meaningful for an arbitrary
|
||||
text→video sample.
|
||||
|
||||
The first run downloads CLIP / DINO / RAFT / AMT / ViCLIP / MUSIQ
|
||||
weights to ``~/.cache/fastvideo/eval/`` (~few GB total).
|
||||
|
||||
GPU memory caveat
|
||||
-----------------
|
||||
Scoring 1088×1920×121 with all 8 metrics needs a dedicated GPU (~80 GB).
|
||||
On a shared GPU, ``vbench.motion_smoothness`` (AMT correlation volume)
|
||||
will OOM — its memory autoscale reads ``total_memory`` rather than
|
||||
``mem_get_info()`` free memory and therefore underestimates the
|
||||
required scale-down. Drop ``motion_smoothness`` from ``METRICS`` if
|
||||
sharing, or run on a smaller-resolution generation.
|
||||
"""
|
||||
import torch
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.eval import Evaluator
|
||||
from fastvideo.eval.io import build_eval_kwargs
|
||||
|
||||
PROMPT = (
|
||||
"A warm sunny backyard. The camera starts in a tight cinematic close-up "
|
||||
"of a woman and a man in their 30s, facing each other with serious "
|
||||
"expressions. The woman, emotional and dramatic, says softly, \"That's "
|
||||
"it... Dad's lost it. And we've lost Dad.\" The man exhales, slightly "
|
||||
"annoyed: \"Stop being so dramatic, Jess.\" A beat. He glances aside, "
|
||||
"then mutters defensively, \"He's just having fun.\" The camera slowly "
|
||||
"pans right, revealing the grandfather in the garden wearing enormous "
|
||||
"butterfly wings, waving his arms in the air like he's trying to take "
|
||||
"off. He shouts, \"Wheeeew!\" as he flaps his wings with full commitment. "
|
||||
"The woman covers her face, on the verge of tears. The tone is deadpan, "
|
||||
"absurd, and quietly tragic."
|
||||
)
|
||||
|
||||
# VBench sub-metrics meaningful for an arbitrary text→video sample
|
||||
# (just the generated frames, optionally fps + the source prompt).
|
||||
# Structured-prompt metrics (vbench.color, vbench.multiple_objects,
|
||||
# vbench.scene, ...) are excluded — they need prompts built to a
|
||||
# specific schema.
|
||||
METRICS = [
|
||||
"vbench.aesthetic_quality", # CLIP + LAION aesthetic head
|
||||
"vbench.subject_consistency", # DINO frame-to-first cosine
|
||||
"vbench.background_consistency", # DINO on background patches
|
||||
"vbench.imaging_quality", # pyiqa MUSIQ
|
||||
"vbench.temporal_flickering", # pixel-wise frame deltas
|
||||
"vbench.motion_smoothness", # AMT frame interpolator residual
|
||||
"vbench.dynamic_degree", # RAFT optical-flow magnitude (needs fps)
|
||||
"vbench.overall_consistency", # ViCLIP video↔prompt similarity
|
||||
]
|
||||
|
||||
|
||||
def main() -> None:
|
||||
# ----- generation (matches examples/inference/basic/basic_ltx2.py) -----
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Davids048/LTX2-Base-Diffusers",
|
||||
num_gpus=1,
|
||||
)
|
||||
|
||||
output_path = "outputs_video/ltx2_basic/output_ltx2_base_t2v_1088_1920_1.1.mp4"
|
||||
generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
output_path=output_path,
|
||||
save_video=True,
|
||||
num_frames=121,
|
||||
height=1088,
|
||||
width=1920,
|
||||
)
|
||||
generator.shutdown()
|
||||
# Free residual CUDA memory the generator left behind so the
|
||||
# evaluator can grab the largest possible workspace for AMT/RAFT.
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
# ----- scoring -----
|
||||
print(f"\n[eval] building evaluator: {METRICS}")
|
||||
evaluator = Evaluator(metrics=METRICS)
|
||||
|
||||
# LTX2 outputs at 24 fps by default.
|
||||
sample = build_eval_kwargs({"prompt": PROMPT}, output_path, fps=24.0)
|
||||
print(f"[eval] running ({sample['video'].shape[1]} frames @ 24 fps)...")
|
||||
results = evaluator.evaluate(**sample)
|
||||
|
||||
print("\n=== VBench scores ===")
|
||||
for name in METRICS:
|
||||
r = results[name]
|
||||
if r.score is None:
|
||||
reason = r.details.get("skipped", "no score")
|
||||
print(f" {name}: SKIPPED ({reason})")
|
||||
else:
|
||||
print(f" {name}: {r.score:.4f}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,157 @@
|
||||
"""End-to-end Physics-IQ: dataset → generate → score → aggregate.
|
||||
|
||||
Generates one video per take-1 scenario with LTX2 (using the scenario
|
||||
caption as the prompt), scores each generated video against the take-1
|
||||
reference and the take-2 "physical-variance" reference, and prints
|
||||
aggregate scores using :meth:`PhysicsIQMetric.aggregate_components` —
|
||||
the official scoring recipe from the upstream benchmark.
|
||||
|
||||
Reference videos / masks / switch-frames auto-fetch on first miss into
|
||||
``${FASTVIDEO_EVAL_CACHE}/datasets/physics_iq/``; pass ``--dataset-root``
|
||||
to point at a pre-downloaded mirror instead.
|
||||
|
||||
Quick smoke run on 4 scenarios across 2 GPUs::
|
||||
|
||||
python examples/inference/eval/bench_physics_iq.py \\
|
||||
--limit 4 --num-gpus 2 \\
|
||||
--videos-dir outputs_video/physics_iq_smoke
|
||||
|
||||
Re-score existing generations without regenerating::
|
||||
|
||||
python examples/inference/eval/bench_physics_iq.py \\
|
||||
--videos-dir outputs_video/physics_iq_smoke \\
|
||||
--skip-generation
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from fastvideo.eval import create_evaluator, get_metric
|
||||
from fastvideo.eval.datasets import get_dataset
|
||||
|
||||
|
||||
def _expected_filename(row: dict) -> str:
|
||||
"""Filename Physics-IQ expects for the generated video for *row*.
|
||||
|
||||
Uses the dataset's own ``expected_gen_filename`` annotation so the
|
||||
output filenames match the benchmark's manifest convention.
|
||||
"""
|
||||
return row["auxiliary_info"]["expected_gen_filename"]
|
||||
|
||||
|
||||
def _generate_videos(rows: list[dict], videos_dir: Path,
|
||||
model: str, num_gpus: int,
|
||||
num_frames: int, height: int, width: int) -> None:
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
videos_dir.mkdir(parents=True, exist_ok=True)
|
||||
todo = [(row, videos_dir / _expected_filename(row)) for row in rows]
|
||||
todo = [(row, out) for (row, out) in todo if not out.is_file()]
|
||||
if not todo:
|
||||
print(f"[gen] all {len(rows)} videos already present; skipping.")
|
||||
return
|
||||
|
||||
print(f"[gen] {len(todo)}/{len(rows)} scenarios to render with {model} "
|
||||
f"({num_frames}x{height}x{width})...")
|
||||
gen = VideoGenerator.from_pretrained(model, num_gpus=num_gpus)
|
||||
try:
|
||||
for row, out_path in todo:
|
||||
gen.generate_video(
|
||||
prompt=row["prompt"], output_path=str(out_path), save_video=True,
|
||||
num_frames=num_frames, height=height, width=width,
|
||||
)
|
||||
finally:
|
||||
gen.shutdown()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
p = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
p.add_argument("--dataset-root", type=Path, default=None,
|
||||
help="Path to a pre-downloaded Physics-IQ release. "
|
||||
"Defaults to ${FASTVIDEO_EVAL_CACHE}/datasets/physics_iq, "
|
||||
"auto-fetching missing assets from the public bucket.")
|
||||
p.add_argument("--videos-dir", type=Path,
|
||||
default=Path("outputs_video/bench_physics_iq"),
|
||||
help="Where to read/write generated videos.")
|
||||
p.add_argument("--limit", type=int, default=None,
|
||||
help="Truncate to first N scenarios for smoke runs.")
|
||||
p.add_argument("--num-gpus", type=int, default=1)
|
||||
p.add_argument("--model", default="Davids048/LTX2-Base-Diffusers",
|
||||
help="HF repo id of the text→video generator to use.")
|
||||
p.add_argument("--num-frames", type=int, default=121)
|
||||
p.add_argument("--height", type=int, default=1088)
|
||||
p.add_argument("--width", type=int, default=1920)
|
||||
p.add_argument("--skip-generation", action="store_true",
|
||||
help="Re-score existing videos under --videos-dir.")
|
||||
p.add_argument("--scores-out", type=Path, default=None,
|
||||
help="Where to write per-scenario scores (JSON). "
|
||||
"Defaults to <videos-dir>/scores.json.")
|
||||
args = p.parse_args()
|
||||
|
||||
# 1. Walk the Physics-IQ corpus. Pass --limit to the dataset
|
||||
# constructor so auto-download only fetches the assets we'll use.
|
||||
ds = get_dataset("physics_iq", dataset_root=args.dataset_root, limit=args.limit)
|
||||
rows = list(ds)
|
||||
print(f"[load] Physics-IQ: {len(rows)} scenarios from {ds.dataset_dir}")
|
||||
|
||||
# 2. Generate (or reuse) one mp4 per scenario.
|
||||
if not args.skip_generation:
|
||||
_generate_videos(
|
||||
rows, args.videos_dir, args.model, args.num_gpus,
|
||||
args.num_frames, args.height, args.width,
|
||||
)
|
||||
|
||||
# 3. Score each scenario. The metric reads file paths directly out
|
||||
# of the row dict (reference, reference_take2, masks), so we
|
||||
# just attach the generated video path and forward.
|
||||
evaluator = create_evaluator(metrics=["physics_iq"], num_gpus=args.num_gpus)
|
||||
|
||||
samples: list[dict] = []
|
||||
matched: list[dict] = []
|
||||
for row in rows:
|
||||
video_path = args.videos_dir / _expected_filename(row)
|
||||
if not video_path.is_file():
|
||||
print(f"[eval] missing {video_path}; skipping.")
|
||||
continue
|
||||
# The physics_iq metric accepts file paths via its polymorphic
|
||||
# input handling — no need to load the tensors here.
|
||||
samples.append({"video": str(video_path), **row})
|
||||
matched.append(row)
|
||||
|
||||
all_results = evaluator.evaluate(samples=samples)
|
||||
evaluator.shutdown()
|
||||
|
||||
# 4. Aggregate per the upstream scoring recipe.
|
||||
metric = get_metric("physics_iq")
|
||||
components = metric.aggregate_components(
|
||||
[r["physics_iq"] for r in all_results]
|
||||
)
|
||||
|
||||
print()
|
||||
print("=== Physics-IQ aggregate ===")
|
||||
for name, value in components.items():
|
||||
print(f" {name:24s} {value:.4f}")
|
||||
|
||||
detailed = [
|
||||
{
|
||||
"scenario": row["auxiliary_info"]["scenario_id"],
|
||||
"view": row["view"],
|
||||
"scenario_name": row["auxiliary_info"]["scenario_name"],
|
||||
"score": results["physics_iq"].score,
|
||||
}
|
||||
for row, results in zip(matched, all_results)
|
||||
]
|
||||
out = args.scores_out or (args.videos_dir / "scores.json")
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
out.write_text(json.dumps(
|
||||
{"aggregate": components, "per_scenario": detailed},
|
||||
indent=2,
|
||||
))
|
||||
print(f"\n[done] per-scenario scores → {out}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,155 @@
|
||||
"""End-to-end VBench: dataset → generate → score → aggregate.
|
||||
|
||||
Iterates the VBench prompt corpus, generates one video per prompt with
|
||||
LTX2, scores each generated video against the requested ``vbench.*``
|
||||
sub-metrics, and prints per-metric averages over the run.
|
||||
|
||||
Re-running with ``--skip-generation`` reuses any mp4 already on disk
|
||||
under ``--videos-dir``, so you can iterate on metric selection without
|
||||
re-paying the generation cost.
|
||||
|
||||
Example — quick smoke run on 4 prompts from the ``aesthetic_quality``
|
||||
dimension across 2 GPUs::
|
||||
|
||||
python examples/inference/eval/bench_vbench.py \\
|
||||
--dimensions aesthetic_quality \\
|
||||
--limit 4 --num-gpus 2 \\
|
||||
--videos-dir outputs_video/vbench_smoke
|
||||
|
||||
Full benchmark on a single dimension::
|
||||
|
||||
python examples/inference/eval/bench_vbench.py \\
|
||||
--dimensions subject_consistency --num-gpus 8
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
from fastvideo.eval import create_evaluator
|
||||
from fastvideo.eval.datasets import get_dataset
|
||||
|
||||
|
||||
def _slugify(prompt: str, max_len: int = 100) -> str:
|
||||
"""Filesystem-safe filename stem; mirrors VBench's official convention."""
|
||||
s = re.sub(r'[\\/:*?"<>|]', "", prompt[:max_len]).strip().strip(".")
|
||||
return re.sub(r"\s+", " ", s) or "output"
|
||||
|
||||
|
||||
def _generate_videos(prompts: list[str], videos_dir: Path,
|
||||
model: str, num_gpus: int,
|
||||
num_frames: int, height: int, width: int) -> None:
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
videos_dir.mkdir(parents=True, exist_ok=True)
|
||||
todo = [(p, videos_dir / f"{_slugify(p)}.mp4") for p in prompts]
|
||||
todo = [(p, out) for (p, out) in todo if not out.is_file()]
|
||||
if not todo:
|
||||
print(f"[gen] all {len(prompts)} videos already present; skipping.")
|
||||
return
|
||||
|
||||
print(f"[gen] {len(todo)}/{len(prompts)} prompts to render with {model} "
|
||||
f"({num_frames}x{height}x{width})...")
|
||||
gen = VideoGenerator.from_pretrained(model, num_gpus=num_gpus)
|
||||
try:
|
||||
for prompt, out_path in todo:
|
||||
gen.generate_video(
|
||||
prompt=prompt, output_path=str(out_path), save_video=True,
|
||||
num_frames=num_frames, height=height, width=width,
|
||||
)
|
||||
finally:
|
||||
gen.shutdown()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
p = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
p.add_argument("--dimensions", default="aesthetic_quality,subject_consistency",
|
||||
help="Comma-separated VBench dimensions (or 'all').")
|
||||
p.add_argument("--limit", type=int, default=None,
|
||||
help="Truncate to first N prompts for smoke runs.")
|
||||
p.add_argument("--videos-dir", type=Path,
|
||||
default=Path("outputs_video/bench_vbench"))
|
||||
p.add_argument("--num-gpus", type=int, default=1)
|
||||
p.add_argument("--model", default="Davids048/LTX2-Base-Diffusers",
|
||||
help="HF repo id of the text→video generator to use.")
|
||||
p.add_argument("--num-frames", type=int, default=121)
|
||||
p.add_argument("--height", type=int, default=1088)
|
||||
p.add_argument("--width", type=int, default=1920)
|
||||
p.add_argument("--fps", type=float, default=24.0,
|
||||
help="Frame-rate annotation passed to fps-aware metrics.")
|
||||
p.add_argument("--skip-generation", action="store_true",
|
||||
help="Re-score existing videos under --videos-dir without "
|
||||
"regenerating.")
|
||||
p.add_argument("--scores-out", type=Path, default=None,
|
||||
help="Where to dump per-prompt scores as JSON. "
|
||||
"Defaults to <videos-dir>/scores.json.")
|
||||
args = p.parse_args()
|
||||
|
||||
# 1. Pull prompts from VBench.
|
||||
dims_arg: list[str] | str = (
|
||||
args.dimensions if args.dimensions == "all"
|
||||
else [d.strip() for d in args.dimensions.split(",") if d.strip()]
|
||||
)
|
||||
ds = get_dataset("vbench", dimensions=dims_arg)
|
||||
rows = list(ds)[: args.limit]
|
||||
print(f"[load] VBench: {len(rows)} prompts across {ds.dimensions}")
|
||||
|
||||
# 2. Generate (or reuse) one mp4 per prompt.
|
||||
if not args.skip_generation:
|
||||
_generate_videos(
|
||||
[row["prompt"] for row in rows],
|
||||
args.videos_dir, args.model, args.num_gpus,
|
||||
args.num_frames, args.height, args.width,
|
||||
)
|
||||
|
||||
# 3. Score each video against the requested vbench sub-metrics.
|
||||
metric_names = sorted(set(f"vbench.{d}" for d in ds.dimensions))
|
||||
print(f"[eval] metrics: {metric_names}")
|
||||
evaluator = create_evaluator(metrics=metric_names, num_gpus=args.num_gpus)
|
||||
|
||||
samples: list[dict] = []
|
||||
matched_rows: list[dict] = []
|
||||
for row in rows:
|
||||
video_path = args.videos_dir / f"{_slugify(row['prompt'])}.mp4"
|
||||
if not video_path.is_file():
|
||||
print(f"[eval] missing {video_path}; skipping this row.")
|
||||
continue
|
||||
# Pass the path; the worker decodes lazily so memory stays bounded.
|
||||
samples.append({
|
||||
"video": str(video_path),
|
||||
"fps": args.fps,
|
||||
**row, # prompt / aux / dims
|
||||
})
|
||||
matched_rows.append(row)
|
||||
|
||||
all_results = evaluator.evaluate(samples=samples)
|
||||
evaluator.shutdown()
|
||||
|
||||
# 4. Aggregate per-metric.
|
||||
by_metric: dict[str, list[float]] = defaultdict(list)
|
||||
detailed: list[dict] = []
|
||||
for row, results in zip(matched_rows, all_results):
|
||||
scores = {name: r.score for name, r in results.items()}
|
||||
detailed.append({"prompt": row["prompt"], "scores": scores})
|
||||
for name, score in scores.items():
|
||||
if score is not None:
|
||||
by_metric[name].append(score)
|
||||
|
||||
print()
|
||||
print("=== per-metric averages ===")
|
||||
for name in sorted(by_metric):
|
||||
avg = sum(by_metric[name]) / len(by_metric[name])
|
||||
print(f" {name:42s} {avg:.4f} (n={len(by_metric[name])})")
|
||||
|
||||
out = args.scores_out or (args.videos_dir / "scores.json")
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
out.write_text(json.dumps(detailed, indent=2))
|
||||
print(f"\n[done] per-prompt scores → {out}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,167 @@
|
||||
"""End-to-end: generate a video with LTX2 and score it with VBench.
|
||||
|
||||
Pipeline:
|
||||
prompt → LTX2-Base → mp4 → fastvideo.eval → vbench scores
|
||||
|
||||
Run::
|
||||
|
||||
pip install -e .[eval]
|
||||
git submodule update --init fastvideo/third_party/eval/vbench
|
||||
|
||||
python examples/inference/eval/eval_ltx2_vbench.py
|
||||
# or with 4 GPUs and the distilled checkpoint:
|
||||
python examples/inference/eval/eval_ltx2_vbench.py \
|
||||
--model FastVideo/LTX2-Distilled-Diffusers --num-gpus 4
|
||||
|
||||
The default metric set covers the vbench sub-metrics that are
|
||||
meaningful for an arbitrary text→video sample — i.e. those that need
|
||||
only the generated video (and optionally fps + the source prompt).
|
||||
Structured-prompt metrics like ``vbench.color``, ``vbench.scene``,
|
||||
``vbench.multiple_objects`` etc. are *not* on by default — they only
|
||||
make sense when the prompt is built to a specific schema, and they
|
||||
require GRiT/detectron2 setup. Pass them via ``--metrics`` if you have
|
||||
a matching prompt.
|
||||
|
||||
First-time runs download CLIP, DINO, RAFT, AMT, ViCLIP, and MUSIQ
|
||||
weights to ``~/.cache/fastvideo/eval/models/`` and
|
||||
``~/.cache/torch/hub/`` (~few GB total). Subsequent runs are fast.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.eval import create_evaluator
|
||||
from fastvideo.eval.io import load_video
|
||||
|
||||
|
||||
PROMPT = (
|
||||
"A warm sunny backyard. The camera starts in a tight cinematic close-up "
|
||||
"of a woman and a man in their 30s, facing each other with serious "
|
||||
"expressions. The woman, emotional and dramatic, says softly, \"That's "
|
||||
"it... Dad's lost it. And we've lost Dad.\" The man exhales, slightly "
|
||||
"annoyed: \"Stop being so dramatic, Jess.\" A beat. He glances aside, "
|
||||
"then mutters defensively, \"He's just having fun.\" The camera slowly "
|
||||
"pans right, revealing the grandfather in the garden wearing enormous "
|
||||
"butterfly wings, waving his arms in the air like he's trying to take "
|
||||
"off. He shouts, \"Wheeeew!\" as he flaps his wings with full commitment. "
|
||||
"The woman covers her face, on the verge of tears. The tone is deadpan, "
|
||||
"absurd, and quietly tragic."
|
||||
)
|
||||
|
||||
DEFAULT_METRICS = [
|
||||
# No-input metrics: just need the generated frames.
|
||||
"vbench.aesthetic_quality", # CLIP + LAION aesthetic head
|
||||
"vbench.subject_consistency", # DINO frame-to-first cosine
|
||||
"vbench.background_consistency", # DINO on background patches
|
||||
"vbench.imaging_quality", # pyiqa MUSIQ
|
||||
"vbench.temporal_flickering", # pixel-wise frame deltas
|
||||
"vbench.motion_smoothness", # AMT frame interpolator residual
|
||||
# Need fps annotation:
|
||||
"vbench.dynamic_degree", # RAFT optical-flow magnitude
|
||||
# Need the source prompt:
|
||||
"vbench.overall_consistency", # ViCLIP video↔prompt similarity
|
||||
]
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
p = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
p.add_argument("--model", default="Davids048/LTX2-Base-Diffusers",
|
||||
help="HF repo id of the LTX2 checkpoint.")
|
||||
p.add_argument("--num-gpus", type=int, default=1)
|
||||
p.add_argument("--output", default="outputs_video/ltx2_eval/clip.mp4",
|
||||
help="Where to save the generated mp4.")
|
||||
p.add_argument("--num-frames", type=int, default=121)
|
||||
p.add_argument("--height", type=int, default=1088)
|
||||
p.add_argument("--width", type=int, default=1920)
|
||||
p.add_argument("--prompt", default=PROMPT)
|
||||
p.add_argument("--fps", type=float, default=24.0,
|
||||
help="Frame-rate annotation passed to fps-aware metrics "
|
||||
"(e.g. vbench.dynamic_degree). LTX2 outputs at 24 fps "
|
||||
"by default.")
|
||||
p.add_argument("--metrics", default=",".join(DEFAULT_METRICS),
|
||||
help="Comma-separated metric names. Pass 'all' for every "
|
||||
"registered metric, or e.g. 'vbench' for the whole group.")
|
||||
p.add_argument("--scores-out", default="outputs_video/ltx2_eval/scores.json")
|
||||
p.add_argument("--skip-generation", action="store_true",
|
||||
help="Reuse an existing --output video instead of regenerating.")
|
||||
return p.parse_args()
|
||||
|
||||
|
||||
def generate(args: argparse.Namespace) -> Path:
|
||||
out = Path(args.output)
|
||||
if args.skip_generation and out.is_file():
|
||||
print(f"[gen] reusing existing video at {out}")
|
||||
return out
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
print(f"[gen] loading {args.model} ({args.num_gpus} GPU)...")
|
||||
generator = VideoGenerator.from_pretrained(args.model, num_gpus=args.num_gpus)
|
||||
try:
|
||||
print(f"[gen] generating to {out}...")
|
||||
generator.generate_video(
|
||||
prompt=args.prompt,
|
||||
output_path=str(out),
|
||||
save_video=True,
|
||||
num_frames=args.num_frames,
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
)
|
||||
finally:
|
||||
generator.shutdown()
|
||||
return out
|
||||
|
||||
|
||||
def evaluate_video(video_path: Path, prompt: str, fps: float,
|
||||
metric_names) -> dict:
|
||||
print(f"[eval] loading video from {video_path}...")
|
||||
video = load_video(str(video_path)) # (T, C, H, W) in [0, 1]
|
||||
video = video.unsqueeze(0) # → (1, T, C, H, W)
|
||||
|
||||
print(f"[eval] building evaluator: {metric_names}")
|
||||
evaluator = create_evaluator(metrics=metric_names, device="cuda")
|
||||
|
||||
print(f"[eval] running ({video.shape[1]} frames @ {fps} fps)...")
|
||||
results = evaluator.evaluate(
|
||||
video=video,
|
||||
text_prompt=[prompt],
|
||||
fps=fps,
|
||||
)
|
||||
|
||||
if isinstance(results, list):
|
||||
results = results[0] # batch of 1
|
||||
|
||||
return {
|
||||
name: {"score": r.score, "details": r.details}
|
||||
for name, r in results.items()
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
if args.metrics.strip() == "all":
|
||||
metric_names = "all"
|
||||
else:
|
||||
metric_names = [m.strip() for m in args.metrics.split(",") if m.strip()]
|
||||
|
||||
video_path = generate(args)
|
||||
scores = evaluate_video(video_path, args.prompt, args.fps, metric_names)
|
||||
|
||||
print("\n=== VBench scores ===")
|
||||
for name, payload in scores.items():
|
||||
print(f" {name}: {payload['score']}")
|
||||
|
||||
out = Path(args.scores_out)
|
||||
out.parent.mkdir(parents=True, exist_ok=True)
|
||||
out.write_text(json.dumps(
|
||||
{"video": str(video_path), "prompt": args.prompt, "scores": scores},
|
||||
indent=2,
|
||||
))
|
||||
print(f"[done] scores written to {out}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,95 @@
|
||||
"""Score a folder of videos in parallel across multiple GPUs.
|
||||
|
||||
Uses :meth:`Evaluator.evaluate(samples=[...])`, which round-robins each
|
||||
sample dict across the GPU replicas the evaluator was built with.
|
||||
|
||||
Example::
|
||||
|
||||
python examples/inference/eval/score_folder.py \\
|
||||
--videos generated/ \\
|
||||
--metrics vbench.aesthetic_quality,vbench.subject_consistency \\
|
||||
--num-gpus 4 \\
|
||||
--output scores.json
|
||||
|
||||
Pair each generated video with a same-name reference video (e.g.
|
||||
``ref/<stem>.mp4``) by passing ``--reference-dir``::
|
||||
|
||||
python examples/inference/eval/score_folder.py \\
|
||||
--videos generated/ --reference-dir ref/ \\
|
||||
--metrics common.psnr,common.ssim,common.lpips \\
|
||||
--num-gpus 4
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from fastvideo.eval import create_evaluator
|
||||
|
||||
|
||||
def _list_videos(directory: Path) -> list[Path]:
|
||||
exts = {".mp4", ".avi", ".mov", ".mkv", ".gif"}
|
||||
return sorted(p for p in directory.iterdir() if p.suffix.lower() in exts)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
p = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
p.add_argument("--videos", type=Path, required=True,
|
||||
help="Directory of generated videos.")
|
||||
p.add_argument("--reference-dir", type=Path, default=None,
|
||||
help="Directory of reference videos with matching stems.")
|
||||
p.add_argument("--metrics", default="vbench.aesthetic_quality")
|
||||
p.add_argument("--num-gpus", type=int, default=1)
|
||||
p.add_argument("--fps", type=float, default=None,
|
||||
help="Frame-rate annotation for fps-aware metrics.")
|
||||
p.add_argument("--output", type=Path, default=Path("scores.json"))
|
||||
args = p.parse_args()
|
||||
|
||||
video_paths = _list_videos(args.videos)
|
||||
if not video_paths:
|
||||
raise SystemExit(f"No videos under {args.videos}")
|
||||
print(f"Found {len(video_paths)} videos in {args.videos}")
|
||||
|
||||
metrics: list[str] | str = (
|
||||
args.metrics if args.metrics == "all"
|
||||
else [m.strip() for m in args.metrics.split(",") if m.strip()]
|
||||
)
|
||||
evaluator = create_evaluator(metrics=metrics, num_gpus=args.num_gpus)
|
||||
|
||||
# Build per-video sample dicts holding *paths*, not pre-loaded
|
||||
# tensors. Each path is decoded inside the worker thread that picks
|
||||
# up its sample, so peak resident memory is bounded by num_gpus
|
||||
# rather than scaling with the size of the folder.
|
||||
samples: list[dict] = []
|
||||
for vp in video_paths:
|
||||
sample: dict = {"video": str(vp)}
|
||||
if args.reference_dir is not None:
|
||||
ref_path = args.reference_dir / vp.name
|
||||
if not ref_path.is_file():
|
||||
raise FileNotFoundError(f"Missing reference for {vp.name} at {ref_path}")
|
||||
sample["reference"] = str(ref_path)
|
||||
if args.fps is not None:
|
||||
sample["fps"] = args.fps
|
||||
samples.append(sample)
|
||||
|
||||
print(f"Scoring with {len(evaluator.metric_names)} metric(s) "
|
||||
f"on {evaluator.num_gpus} GPU(s)...")
|
||||
all_results = evaluator.evaluate(samples=samples)
|
||||
evaluator.shutdown()
|
||||
|
||||
payload = [
|
||||
{
|
||||
"video": str(vp),
|
||||
"scores": {name: r.score for name, r in results.items()},
|
||||
}
|
||||
for vp, results in zip(video_paths, all_results)
|
||||
]
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(payload, indent=2))
|
||||
print(f"Wrote {args.output}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,68 @@
|
||||
"""Score one video on one GPU.
|
||||
|
||||
Smallest possible use of ``fastvideo.eval``: load an mp4, build an
|
||||
:class:`Evaluator` for the requested metric set, run it.
|
||||
|
||||
Examples::
|
||||
|
||||
# Reference-free (just the generated video):
|
||||
python examples/inference/eval/score_video.py \\
|
||||
--video clip.mp4 \\
|
||||
--metrics vbench.aesthetic_quality,vbench.imaging_quality
|
||||
|
||||
# Reference-paired (compare against ground truth):
|
||||
python examples/inference/eval/score_video.py \\
|
||||
--video gen.mp4 --reference ref.mp4 \\
|
||||
--metrics common.psnr,common.ssim,common.lpips
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
|
||||
from fastvideo.eval import create_evaluator
|
||||
from fastvideo.eval.io import load_video
|
||||
|
||||
|
||||
def main() -> None:
|
||||
p = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
p.add_argument("--video", required=True, help="Path to the generated mp4.")
|
||||
p.add_argument("--reference", default=None,
|
||||
help="Optional path to a reference mp4 (for paired metrics).")
|
||||
p.add_argument("--metrics", default="common.psnr,common.ssim",
|
||||
help="Comma-separated metric names, or a group name like 'vbench'.")
|
||||
p.add_argument("--device", default="cuda:0")
|
||||
p.add_argument("--text-prompt", default=None,
|
||||
help="Text prompt for prompt-aware metrics "
|
||||
"(vbench.overall_consistency, etc.).")
|
||||
p.add_argument("--fps", type=float, default=None,
|
||||
help="Frame-rate annotation for fps-aware metrics "
|
||||
"(vbench.dynamic_degree, etc.).")
|
||||
args = p.parse_args()
|
||||
|
||||
metrics: list[str] | str = (
|
||||
args.metrics if args.metrics in ("all",)
|
||||
else [m.strip() for m in args.metrics.split(",") if m.strip()]
|
||||
)
|
||||
evaluator = create_evaluator(metrics=metrics, device=args.device)
|
||||
|
||||
sample: dict = {"video": load_video(args.video)}
|
||||
if args.reference is not None:
|
||||
sample["reference"] = load_video(args.reference)
|
||||
if args.text_prompt is not None:
|
||||
sample["text_prompt"] = args.text_prompt
|
||||
if args.fps is not None:
|
||||
sample["fps"] = args.fps
|
||||
|
||||
results = evaluator.evaluate(**sample)
|
||||
evaluator.shutdown()
|
||||
|
||||
print(json.dumps(
|
||||
{name: r.score for name, r in results.items()},
|
||||
indent=2,
|
||||
))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -5,7 +5,7 @@ import time
|
||||
|
||||
import gradio as gr
|
||||
from fastvideo.entrypoints.video_generator import VideoGenerator
|
||||
from fastvideo.configs.sample.base import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
from copy import deepcopy
|
||||
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ import tempfile
|
||||
|
||||
import gradio as gr
|
||||
|
||||
from fastvideo.configs.sample.base import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
|
||||
MODEL_PATH_MAPPING = {
|
||||
|
||||
@@ -185,7 +185,7 @@ class BaseModelDeployment:
|
||||
|
||||
def _initialize_generator(self, config: Dict[str, Any]) -> None:
|
||||
from fastvideo.entrypoints.video_generator import VideoGenerator
|
||||
from fastvideo.configs.sample.base import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
print(f"Initializing model: {self.model_path}")
|
||||
self.generator = VideoGenerator.from_pretrained(
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "./lora_out"
|
||||
def main():
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
Inference using a LoRA checkpoint from FastVideo trainer.
|
||||
"""
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "./lora_out"
|
||||
def main():
|
||||
|
||||
@@ -1,79 +1,5 @@
|
||||
# Optimization Examples
|
||||
|
||||
## Wan 2.1 QAT Attention 14B Inference
|
||||
|
||||
Use these files for Wan 2.1 14B inference with the `ATTN_QAT_INFER` backend:
|
||||
|
||||
- `examples/inference/optimizations/download_14B_qat.sh`
|
||||
- `examples/inference/optimizations/attn_qat_inference_example.py`
|
||||
|
||||
### 1. Download the 14B QAT checkpoint
|
||||
|
||||
The helper script downloads the QAT safetensors from
|
||||
`FastVideo/14B_qat_400` into `checkpoints/14B_qat_400` by default.
|
||||
|
||||
Prerequisites:
|
||||
|
||||
- `huggingface_hub` installed, for example: `uv pip install huggingface_hub`
|
||||
- access to the model repo if it is private or gated: `huggingface-cli login`
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh
|
||||
python examples/inference/optimizations/attention_example.py
|
||||
```
|
||||
|
||||
To download into a custom directory, pass it as the first argument:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
|
||||
```
|
||||
|
||||
### 2. Edit the inference example for Wan 2.1 14B
|
||||
|
||||
Open `examples/inference/optimizations/attn_qat_inference_example.py` and
|
||||
update these two values:
|
||||
|
||||
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
|
||||
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
|
||||
2. Replace the placeholder
|
||||
`init_weights_from_safetensors="safetensors_path"` with the directory that
|
||||
contains the downloaded `.safetensors` files.
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
init_weights_from_safetensors="checkpoints/14B_qat_400",
|
||||
)
|
||||
```
|
||||
|
||||
The script already sets:
|
||||
|
||||
```python
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
```
|
||||
|
||||
### 3. Run the example
|
||||
|
||||
```bash
|
||||
python examples/inference/optimizations/attn_qat_inference_example.py
|
||||
```
|
||||
|
||||
The generated videos are written to `video_samples/` by default.
|
||||
|
||||
### Notes
|
||||
|
||||
- `ATTN_QAT_INFER` requires the in-repo `fastvideo-kernel` build to expose the
|
||||
`attn_qat_infer` package.
|
||||
- If you have not built the kernel yet, run `cd fastvideo-kernel && ./build.sh`
|
||||
first.
|
||||
- If you keep the example on the `1.3B` base model while loading the 14B QAT
|
||||
weights, the model/config will not match.
|
||||
|
||||
@@ -1,54 +0,0 @@
|
||||
from fastvideo import VideoGenerator
|
||||
import os
|
||||
from pathlib import Path
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
CHECKPOINT_PATH = Path(__file__).parent.parent.parent
|
||||
|
||||
def main():
|
||||
# FastVideo will automatically use the optimal default arguments for the
|
||||
# model.
|
||||
# If a local path is provided, FastVideo will make a best effort
|
||||
# attempt to identify the optimal arguments.
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
# image_encoder_cpu_offload=False,
|
||||
# Load custom weights from checkpoint
|
||||
init_weights_from_safetensors="safetensors_path"
|
||||
)
|
||||
|
||||
# sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
# sampling_param.num_frames = 45
|
||||
# sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
|
||||
# Generate videos with the same simple API, regardless of GPU count
|
||||
prompt = (
|
||||
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
)
|
||||
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True)
|
||||
# video = generator.generate_video(prompt, sampling_param=sampling_param, output_path="wan_t2v_videos/")
|
||||
|
||||
# Generate another video with a different prompt, without reloading the
|
||||
# model!
|
||||
prompt2 = (
|
||||
"A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently in "
|
||||
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic.")
|
||||
video2 = generator.generate_video(prompt2, output_path=OUTPUT_PATH, save_video=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,58 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
|
||||
REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." && pwd)"
|
||||
|
||||
HF_REPO_ID="${HF_REPO_ID:-FastVideo/14B_qat_400}"
|
||||
HF_REVISION="${HF_REVISION:-main}"
|
||||
LOCAL_DIR="${1:-${REPO_ROOT}/checkpoints/14B_qat_400}"
|
||||
PYTHON_BIN="${PYTHON:-python}"
|
||||
|
||||
if ! command -v "${PYTHON_BIN}" >/dev/null 2>&1; then
|
||||
echo "Python executable not found: ${PYTHON_BIN}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! "${PYTHON_BIN}" -c "import huggingface_hub" >/dev/null 2>&1; then
|
||||
echo "Missing dependency: huggingface_hub" >&2
|
||||
echo "Install it with: uv pip install huggingface_hub" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p "${LOCAL_DIR}"
|
||||
|
||||
echo "Downloading ${HF_REPO_ID}@${HF_REVISION}"
|
||||
echo "Local directory: ${LOCAL_DIR}"
|
||||
|
||||
"${PYTHON_BIN}" -c '
|
||||
import argparse
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--repo-id", required=True)
|
||||
parser.add_argument("--revision", required=True)
|
||||
parser.add_argument("--local-dir", required=True)
|
||||
args = parser.parse_args()
|
||||
|
||||
snapshot_download(
|
||||
repo_id=args.repo_id,
|
||||
revision=args.revision,
|
||||
repo_type="model",
|
||||
local_dir=args.local_dir,
|
||||
local_dir_use_symlinks=False,
|
||||
resume_download=True,
|
||||
)
|
||||
' \
|
||||
--repo-id "${HF_REPO_ID}" \
|
||||
--revision "${HF_REVISION}" \
|
||||
--local-dir "${LOCAL_DIR}"
|
||||
|
||||
echo
|
||||
echo "Download complete."
|
||||
echo "Use this in your inference script:"
|
||||
echo "init_weights_from_safetensors=\"${LOCAL_DIR}\""
|
||||
echo
|
||||
echo "If the repo is private or gated, make sure you are logged in with:"
|
||||
echo "huggingface-cli login"
|
||||
@@ -1,88 +0,0 @@
|
||||
import torch
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.pipelines.base import PipelineConfig
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
|
||||
|
||||
def main():
|
||||
print("=== FP4 Quantization Video Generation Example ===")
|
||||
|
||||
if not torch.cuda.is_available():
|
||||
print("Warning: CUDA not available. FP4 quantization requires GPU.")
|
||||
return
|
||||
|
||||
gpu_capability = torch.cuda.get_device_capability()
|
||||
if gpu_capability[0] < 9: # H100 and newer
|
||||
print(f"Warning: GPU capability {gpu_capability} may not support FP4. Recommended: 9.0+")
|
||||
|
||||
print(f"GPU: {torch.cuda.get_device_name()}")
|
||||
print(f"GPU Capability: {gpu_capability}")
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
# model_id = "Wan-AI/Wan2.1-T2V-14B-Diffusers"
|
||||
pipeline_config = PipelineConfig.from_pretrained(model_id)
|
||||
pipeline_config.dit_precision = "bf16"
|
||||
|
||||
print("\nLoading model with FP4 quantization...")
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_id,
|
||||
pipeline_config=pipeline_config,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
transformer_quant="fp4",
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
)
|
||||
|
||||
print("FP4 configuration applied. Generating videos...")
|
||||
|
||||
print("\n=== Generating Video with FP4 Quantization ===")
|
||||
|
||||
prompt1 = (
|
||||
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
)
|
||||
|
||||
print(f"Prompt: {prompt1}")
|
||||
print("Generating video...")
|
||||
|
||||
try:
|
||||
video1 = generator.generate_video(
|
||||
prompt1,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
)
|
||||
print("✓ First video generated successfully with FP4 quantization!")
|
||||
|
||||
# # Generate a second video to show the model can be reused
|
||||
prompt2 = (
|
||||
"A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently in "
|
||||
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic."
|
||||
)
|
||||
|
||||
print(f"\nGenerating second video...")
|
||||
print(f"Prompt: {prompt2}")
|
||||
|
||||
video2 = generator.generate_video(
|
||||
prompt2,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
)
|
||||
print("✓ Second video generated successfully with FP4 quantization!")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during video generation: {e}")
|
||||
return
|
||||
|
||||
print(f"Videos saved to: {OUTPUT_PATH}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,47 +1,27 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=wan_t2v_1.3B_finetune
|
||||
#SBATCH --partition=all
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --gres=gpu:4
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --output=logs/wan_t2v_1.3B_finetune.out
|
||||
#SBATCH --error=logs/wan_t2v_1.3B_finetune.err
|
||||
|
||||
source .venv/bin/activate
|
||||
|
||||
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
|
||||
|
||||
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATA_DIR=data/Wan-Syn_77x448x832_600k
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
|
||||
NUM_GPUS=1
|
||||
DATA_DIR="data/crush-smol_processed_t2v/combined_parquet_dataset/"
|
||||
VALIDATION_DATASET_FILE="$(dirname "$0")/validation.json"
|
||||
NUM_GPUS=4
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ---- torchrun rendezvous (multi-node) ----
|
||||
# Launch ONE torchrun per node (via srun) and let torchrun spawn 4 workers per node.
|
||||
MASTER_ADDR="$(scontrol show hostnames "$SLURM_JOB_NODELIST" | head -n 1)"
|
||||
MASTER_PORT="${MASTER_PORT:-29500}"
|
||||
export MASTER_ADDR MASTER_PORT
|
||||
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name "wan_t2v_finetune_qat"
|
||||
--output_dir "checkpoints/wan_t2v_finetune_1.3B_77"
|
||||
--max_train_steps 4000
|
||||
--tracker_project_name "wan_t2v_finetune"
|
||||
--output_dir "checkpoints/wan_t2v_finetune"
|
||||
--max_train_steps 5000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--gradient_accumulation_steps 8
|
||||
--num_latent_t 20
|
||||
--num_height 448
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--num_frames 77
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
@@ -50,7 +30,7 @@ training_args=(
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus $NUM_GPUS
|
||||
--sp_size 1
|
||||
--sp_size $NUM_GPUS
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1
|
||||
--hsdp_shard_dim $NUM_GPUS
|
||||
@@ -65,7 +45,7 @@ model_args=(
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path $DATA_DIR
|
||||
--dataloader_num_workers 4
|
||||
--dataloader_num_workers 1
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
@@ -74,16 +54,16 @@ validation_args=(
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "5.0"
|
||||
--validation_guidance_scale "3.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-6
|
||||
--learning_rate 5e-5
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 1000
|
||||
--training_state_checkpointing_steps 1000
|
||||
--weight_decay 0.01
|
||||
--weight_decay 1e-4
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
@@ -92,24 +72,23 @@ miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
--not_apply_cfg_solver
|
||||
--dit_precision "fp32"
|
||||
--num_euler_timesteps 50
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
# --resume_from_checkpoint "checkpoints/wan_t2v_finetune/checkpoint-2500"
|
||||
)
|
||||
|
||||
srun --nodes="$SLURM_NNODES" --ntasks="$SLURM_NNODES" --ntasks-per-node=1 \
|
||||
torchrun \
|
||||
--nnodes "$SLURM_NNODES" \
|
||||
--nproc_per_node 4 \
|
||||
--rdzv_backend c10d \
|
||||
--rdzv_endpoint "${MASTER_ADDR}:${MASTER_PORT}" \
|
||||
--rdzv_id "$SLURM_JOB_ID" \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
torchrun \
|
||||
--nnodes 1 \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
|
||||
@@ -1,124 +0,0 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
|
||||
#SBATCH --partition=all
|
||||
#SBATCH --nodes=4
|
||||
#SBATCH --gres=gpu:4
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
|
||||
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
|
||||
|
||||
source .venv/bin/activate
|
||||
|
||||
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
|
||||
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
# Use node-local Triton cache to avoid stale file handle errors on shared filesystems
|
||||
export TRITON_CACHE_DIR="/tmp/triton_cache_${SLURM_JOB_ID}_${SLURM_NODEID}"
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATA_DIR=YOUR_DATA_DIR
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
|
||||
NUM_GPUS=16
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ---- torchrun rendezvous (multi-node) ----
|
||||
# 1. Get the hostname of the first node (Master)
|
||||
nodes=( $( scontrol show hostnames $SLURM_JOB_NODELIST ) )
|
||||
nodes_array=($nodes)
|
||||
head_node=${nodes_array[0]}
|
||||
MASTER_ADDR=$(srun --nodes=1 --ntasks=1 -w "$head_node" hostname --ip-address)
|
||||
MASTER_PORT=29500
|
||||
|
||||
# 2. Get the node count automatically
|
||||
NNODES=$SLURM_NNODES
|
||||
GPUS_PER_NODE=$SLURM_GPUS_ON_NODE
|
||||
NUM_GPUS=$((NNODES * GPUS_PER_NODE))
|
||||
|
||||
echo "MASTER_ADDR=$MASTER_ADDR MASTER_PORT=$MASTER_PORT NNODES=$NNODES"
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name "wan_t2v_finetune_qat"
|
||||
--output_dir "checkpoints/wan_1.3B_t2v_finetune_qat"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 20
|
||||
--num_height 448
|
||||
--num_width 832
|
||||
--num_frames 77
|
||||
--enable_gradient_checkpointing_type "full" # if OOM enable this
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus $NUM_GPUS
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim $NUM_GPUS
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "5.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-6
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 200
|
||||
--training_state_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 1
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $NNODES \
|
||||
--nproc_per_node $GPUS_PER_NODE \
|
||||
--node_rank $SLURM_PROCID \
|
||||
--rdzv_backend c10d \
|
||||
--rdzv_endpoint $MASTER_ADDR:$MASTER_PORT \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
@@ -1,131 +1,31 @@
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
|
||||
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"caption": "A large metal cylinder is seen pressing down on a pile of Oreo cookies, flattening them as if they were under a hydraulic press.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
|
||||
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"caption": "A large metal cylinder is seen compressing colorful clay into a compact shape, demonstrating the power of a hydraulic press.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
|
||||
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"caption": "A large metal cylinder is seen pressing down on a pile of colorful candies, flattening them as if they were under a hydraulic press. The candies are crushed and broken into small pieces, creating a mess on the table.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
|
||||
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
|
||||
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
|
||||
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
|
||||
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
|
||||
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
|
||||
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
|
||||
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
|
||||
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
|
||||
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
|
||||
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
|
||||
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
|
||||
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
|
||||
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
} ]
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,123 +0,0 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
|
||||
#SBATCH --partition=main
|
||||
#SBATCH --nodes=4
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=128
|
||||
#SBATCH --mem=1440G
|
||||
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
|
||||
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
|
||||
#SBATCH --exclusive
|
||||
|
||||
source ~/conda/miniconda/bin/activate
|
||||
conda activate matthew-fv
|
||||
|
||||
# Basic Info
|
||||
export WANDB_MODE="online"
|
||||
export NCCL_P2P_DISABLE=1
|
||||
export TORCH_NCCL_ENABLE_MONITORING=0
|
||||
# different cache dir for different processes
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
|
||||
export MASTER_PORT=29500
|
||||
export NODE_RANK=$SLURM_PROCID
|
||||
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
|
||||
export MASTER_ADDR=${nodes[0]}
|
||||
export CUDA_VISIBLE_DEVICES=$SLURM_LOCALID
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
|
||||
echo "MASTER_ADDR: $MASTER_ADDR"
|
||||
echo "NODE_RANK: $NODE_RANK"
|
||||
|
||||
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-14B-Diffusers"
|
||||
DATA_DIR=YOUR_DATA_DIR
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
|
||||
NUM_GPUS_PER_NODE=8
|
||||
TOTAL_GPUS=$((NUM_GPUS_PER_NODE * SLURM_JOB_NUM_NODES))
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name "wan_t2v_finetune_qat"
|
||||
--output_dir "checkpoints/wan_14B_t2v_finetune_qat"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 20
|
||||
--num_height 768
|
||||
--num_width 1280
|
||||
--num_frames 77
|
||||
--enable_gradient_checkpointing_type "full" # if OOM enable this
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus $TOTAL_GPUS
|
||||
--sp_size 4
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 4
|
||||
--hsdp_shard_dim 8
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
# --log_validation
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "5.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-6
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 200
|
||||
--training_state_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS_PER_NODE \
|
||||
--node_rank $SLURM_PROCID \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
@@ -1,131 +0,0 @@
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
|
||||
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
|
||||
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
|
||||
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
|
||||
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
|
||||
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
|
||||
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
|
||||
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
|
||||
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
|
||||
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
|
||||
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
|
||||
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
|
||||
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
|
||||
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
|
||||
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
|
||||
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
|
||||
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
} ]
|
||||
}
|
||||
@@ -12,18 +12,6 @@ else()
|
||||
enable_language(CUDA)
|
||||
# Ensure CUDA toolkit targets (CUDA::cudart, CUDA::cuda_driver, etc.) are available.
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
if(NOT DEFINED CUDA_TOOLKIT_ROOT_DIR)
|
||||
if(DEFINED CUDAToolkit_ROOT)
|
||||
set(CUDA_TOOLKIT_ROOT_DIR "${CUDAToolkit_ROOT}" CACHE PATH
|
||||
"CUDA toolkit root directory" FORCE)
|
||||
elseif(DEFINED ENV{CUDAToolkit_ROOT})
|
||||
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDAToolkit_ROOT}" CACHE PATH
|
||||
"CUDA toolkit root directory" FORCE)
|
||||
elseif(DEFINED ENV{CUDA_HOME})
|
||||
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDA_HOME}" CACHE PATH
|
||||
"CUDA toolkit root directory" FORCE)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Import common utils if needed, but we keep it simple for now
|
||||
@@ -31,46 +19,13 @@ endif()
|
||||
# Find Python and Torch
|
||||
find_package(Python COMPONENTS Interpreter Development.Module REQUIRED)
|
||||
|
||||
# Locate the installed torch package without importing it. This keeps CMake
|
||||
# configure working even on nodes where CUDA runtime libraries are not yet on
|
||||
# the dynamic loader path.
|
||||
# Robustly find Torch include paths using Python
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('platlib'))"
|
||||
OUTPUT_VARIABLE PYTHON_PLATLIB
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import torch; from torch.utils.cpp_extension import include_paths; print(';'.join(include_paths()))"
|
||||
OUTPUT_VARIABLE TORCH_INCLUDE_PATHS
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('purelib'))"
|
||||
OUTPUT_VARIABLE PYTHON_PURELIB
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
|
||||
set(TORCH_PYTHON_PACKAGE_DIR "")
|
||||
foreach(_candidate
|
||||
"${PYTHON_PLATLIB}/torch"
|
||||
"${PYTHON_PURELIB}/torch"
|
||||
)
|
||||
if(EXISTS "${_candidate}")
|
||||
set(TORCH_PYTHON_PACKAGE_DIR "${_candidate}")
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if(NOT TORCH_PYTHON_PACKAGE_DIR)
|
||||
message(FATAL_ERROR "Could not locate the installed torch Python package.")
|
||||
endif()
|
||||
|
||||
list(APPEND TORCH_INCLUDE_DIRS
|
||||
"${TORCH_PYTHON_PACKAGE_DIR}/include"
|
||||
"${TORCH_PYTHON_PACKAGE_DIR}/include/torch/csrc/api/include"
|
||||
)
|
||||
|
||||
if(NOT Torch_DIR)
|
||||
set(_TORCH_CONFIG_DIR "${TORCH_PYTHON_PACKAGE_DIR}/share/cmake/Torch")
|
||||
if(EXISTS "${_TORCH_CONFIG_DIR}/TorchConfig.cmake")
|
||||
set(Torch_DIR "${_TORCH_CONFIG_DIR}" CACHE PATH "Path to Torch CMake config" FORCE)
|
||||
endif()
|
||||
endif()
|
||||
list(APPEND TORCH_INCLUDE_DIRS ${TORCH_INCLUDE_PATHS})
|
||||
|
||||
# Find Torch package (still useful for libraries)
|
||||
find_package(Torch REQUIRED)
|
||||
@@ -95,21 +50,6 @@ include_directories(
|
||||
set(FASTVIDEO_KERNEL_BUILD_TK "AUTO" CACHE STRING "Build ThunderKittens kernels: AUTO/ON/OFF")
|
||||
set_property(CACHE FASTVIDEO_KERNEL_BUILD_TK PROPERTY STRINGS AUTO ON OFF)
|
||||
|
||||
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "AUTO")
|
||||
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 AND NOT DEFINED CACHE{FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER})
|
||||
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "${FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3}")
|
||||
endif()
|
||||
|
||||
set(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER "${_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT}" CACHE STRING
|
||||
"Build attn_qat_infer Blackwell inference kernels: AUTO/ON/OFF")
|
||||
set_property(CACHE FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER PROPERTY STRINGS AUTO ON OFF)
|
||||
|
||||
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3)
|
||||
message(DEPRECATION
|
||||
"FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 is deprecated. "
|
||||
"Use FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER instead.")
|
||||
endif()
|
||||
|
||||
# Prefer environment variable (used by CI) if CMake var is not explicitly set.
|
||||
if(NOT DEFINED TORCH_CUDA_ARCH_LIST AND DEFINED ENV{TORCH_CUDA_ARCH_LIST})
|
||||
set(TORCH_CUDA_ARCH_LIST "$ENV{TORCH_CUDA_ARCH_LIST}")
|
||||
@@ -117,7 +57,6 @@ endif()
|
||||
|
||||
message(STATUS "TORCH_CUDA_ARCH_LIST (cmake/env): ${TORCH_CUDA_ARCH_LIST}")
|
||||
message(STATUS "FASTVIDEO_KERNEL_BUILD_TK: ${FASTVIDEO_KERNEL_BUILD_TK}")
|
||||
message(STATUS "FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER: ${FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER}")
|
||||
|
||||
set(ENABLE_TK_KERNELS OFF)
|
||||
if(FASTVIDEO_KERNEL_BUILD_TK STREQUAL "ON")
|
||||
@@ -152,54 +91,6 @@ else()
|
||||
message(STATUS "ThunderKittens kernels: DISABLED (will use Triton fallbacks at runtime)")
|
||||
endif()
|
||||
|
||||
set(ENABLE_ATTN_QAT_INFER OFF)
|
||||
if(GPU_BACKEND STREQUAL "ROCM")
|
||||
message(STATUS "attn_qat_infer kernels: DISABLED (ROCm build)")
|
||||
else()
|
||||
set(_WANTS_ATTN_QAT_INFER OFF)
|
||||
if(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "ON")
|
||||
set(_WANTS_ATTN_QAT_INFER ON)
|
||||
elseif(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "AUTO")
|
||||
if(TORCH_CUDA_ARCH_LIST)
|
||||
string(REGEX MATCH
|
||||
"(^|[; ,])((12\\.0a)|(120a)|(sm_120a))([; ,]|$)"
|
||||
_HAS_120A "${TORCH_CUDA_ARCH_LIST}")
|
||||
if(_HAS_120A)
|
||||
set(_WANTS_ATTN_QAT_INFER ON)
|
||||
endif()
|
||||
else()
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c
|
||||
"import torch; print('1' if (torch.cuda.is_available() and torch.version.cuda and torch.cuda.get_device_capability()[0] >= 12) else '0')"
|
||||
OUTPUT_VARIABLE _LOCAL_HAS_BLACKWELL
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_QUIET
|
||||
)
|
||||
if(_LOCAL_HAS_BLACKWELL STREQUAL "1")
|
||||
set(_WANTS_ATTN_QAT_INFER ON)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(_WANTS_ATTN_QAT_INFER)
|
||||
if(CUDAToolkit_VERSION VERSION_LESS 12.8)
|
||||
message(WARNING
|
||||
"attn_qat_infer kernels require CUDA Toolkit 12.8+. "
|
||||
"Skipping because CUDAToolkit_VERSION=${CUDAToolkit_VERSION}.")
|
||||
else()
|
||||
set(ENABLE_ATTN_QAT_INFER ON)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(ENABLE_ATTN_QAT_INFER)
|
||||
message(STATUS "attn_qat_infer kernels: ENABLED")
|
||||
else()
|
||||
message(STATUS
|
||||
"attn_qat_infer kernels: DISABLED "
|
||||
"(requires CUDA 12.8+ and Blackwell sm_120a)")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Always try to build the extension if CUDA is available, but conditionally add sources/flags
|
||||
set(BUILD_CXX_KERNELS ON)
|
||||
|
||||
@@ -270,15 +161,12 @@ if(BUILD_CXX_KERNELS)
|
||||
|
||||
# Also link against libtorch_python to satisfy Python-binding symbols
|
||||
# (e.g., torch::PyWarningHandler) required by torch/extension.h.
|
||||
file(GLOB TORCH_PYTHON_LIBRARY_CANDIDATES
|
||||
"${TORCH_PYTHON_PACKAGE_DIR}/lib/libtorch_python*"
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import torch; from pathlib import Path; p=Path(torch.__file__).parent/'lib'; m=sorted(p.glob('libtorch_python*')); print(str(m[0]) if m else '')"
|
||||
OUTPUT_VARIABLE TORCH_PYTHON_LIBRARY_PATH
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_QUIET
|
||||
)
|
||||
list(LENGTH TORCH_PYTHON_LIBRARY_CANDIDATES _TORCH_PYTHON_LIBRARY_COUNT)
|
||||
if(_TORCH_PYTHON_LIBRARY_COUNT GREATER 0)
|
||||
list(GET TORCH_PYTHON_LIBRARY_CANDIDATES 0 TORCH_PYTHON_LIBRARY_PATH)
|
||||
else()
|
||||
set(TORCH_PYTHON_LIBRARY_PATH "")
|
||||
endif()
|
||||
if(TORCH_PYTHON_LIBRARY_PATH)
|
||||
message(STATUS "TORCH_PYTHON_LIBRARY_PATH: ${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
target_link_libraries(fastvideo_kernel_ops PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
@@ -295,73 +183,3 @@ if(BUILD_CXX_KERNELS)
|
||||
install(TARGETS fastvideo_kernel_ops LIBRARY DESTINATION fastvideo_kernel/_C)
|
||||
endif()
|
||||
|
||||
if(ENABLE_ATTN_QAT_INFER)
|
||||
set(ATTN_QAT_INFER_DIR ${CMAKE_SOURCE_DIR}/attn_qat_infer)
|
||||
set(ATTN_QAT_INFER_INCLUDE_DIRS
|
||||
${ATTN_QAT_INFER_DIR}
|
||||
${CMAKE_SOURCE_DIR}/include/cutlass/include
|
||||
${CMAKE_SOURCE_DIR}/include/cutlass/tools/util/include
|
||||
${TORCH_INCLUDE_DIRS}
|
||||
)
|
||||
set(ATTN_QAT_INFER_CUDA_FLAGS
|
||||
"-O3"
|
||||
"-std=c++17"
|
||||
"-U__CUDA_NO_HALF_OPERATORS__"
|
||||
"-U__CUDA_NO_HALF_CONVERSIONS__"
|
||||
"-U__CUDA_NO_BFLOAT16_OPERATORS__"
|
||||
"-U__CUDA_NO_BFLOAT16_CONVERSIONS__"
|
||||
"-U__CUDA_NO_BFLOAT162_OPERATORS__"
|
||||
"-U__CUDA_NO_BFLOAT162_CONVERSIONS__"
|
||||
"--expt-relaxed-constexpr"
|
||||
"--expt-extended-lambda"
|
||||
"--use_fast_math"
|
||||
"--ptxas-options=--verbose,--warn-on-local-memory-usage"
|
||||
"-lineinfo"
|
||||
"-DCUTLASS_DEBUG_TRACE_LEVEL=0"
|
||||
"-DNDEBUG"
|
||||
"-DQBLKSIZE=128"
|
||||
"-DKBLKSIZE=128"
|
||||
"-DCTA256"
|
||||
"-DDQINRMEM"
|
||||
)
|
||||
|
||||
Python_add_library(fp4attn_cuda MODULE WITH_SOABI
|
||||
attn_qat_infer/blackwell/api.cu
|
||||
)
|
||||
target_include_directories(fp4attn_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
|
||||
target_compile_definitions(fp4attn_cuda PRIVATE TORCH_EXTENSION_NAME=fp4attn_cuda)
|
||||
target_compile_options(fp4attn_cuda PRIVATE
|
||||
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
|
||||
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
|
||||
)
|
||||
set_target_properties(fp4attn_cuda PROPERTIES
|
||||
CUDA_ARCHITECTURES "120a"
|
||||
CXX_STANDARD 17
|
||||
CUDA_STANDARD 17
|
||||
)
|
||||
target_link_libraries(fp4attn_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
|
||||
|
||||
Python_add_library(fp4quant_cuda MODULE WITH_SOABI
|
||||
attn_qat_infer/quantization/fp4_quantization_4d.cu
|
||||
)
|
||||
target_include_directories(fp4quant_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
|
||||
target_compile_definitions(fp4quant_cuda PRIVATE TORCH_EXTENSION_NAME=fp4quant_cuda)
|
||||
target_compile_options(fp4quant_cuda PRIVATE
|
||||
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
|
||||
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
|
||||
)
|
||||
set_target_properties(fp4quant_cuda PROPERTIES
|
||||
CUDA_ARCHITECTURES "120a"
|
||||
CXX_STANDARD 17
|
||||
CUDA_STANDARD 17
|
||||
)
|
||||
target_link_libraries(fp4quant_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
|
||||
|
||||
if(TORCH_PYTHON_LIBRARY_PATH)
|
||||
target_link_libraries(fp4attn_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
target_link_libraries(fp4quant_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
endif()
|
||||
|
||||
install(TARGETS fp4attn_cuda LIBRARY DESTINATION .)
|
||||
install(TARGETS fp4quant_cuda LIBRARY DESTINATION .)
|
||||
endif()
|
||||
|
||||
@@ -2,6 +2,5 @@ include LICENSE
|
||||
include README.md
|
||||
include pyproject.toml
|
||||
recursive-include python/fastvideo_kernel *.py
|
||||
recursive-include attn_qat_infer *.py *.cu *.cuh *.cpp *.h
|
||||
recursive-include csrc *.cu *.cuh *.cpp *.h
|
||||
recursive-include include/tk *.cu *.cuh *.cpp *.h *.src
|
||||
|
||||
@@ -20,11 +20,6 @@ cd fastvideo-kernel
|
||||
./build.sh
|
||||
```
|
||||
|
||||
On supported Blackwell environments, the same install also packages
|
||||
`attn_qat_infer` and builds its `fp4attn_cuda` / `fp4quant_cuda`
|
||||
extensions directly from `fastvideo-kernel/attn_qat_infer/`. This path
|
||||
requires CUDA Toolkit 12.8+ and targets `sm_120a`.
|
||||
|
||||
### Rocm Build
|
||||
If you are in a rocm environment without the compilation toolchaine of CUDA.
|
||||
|
||||
@@ -63,18 +58,6 @@ cd fastvideo-kernel
|
||||
python benchmarks/bench_vsa.py --batch_size 1 --num_heads 16 --head_dim 128 --q_seq_lens 49152 --topk 64
|
||||
```
|
||||
|
||||
### Attn QAT Attention Benchmarks
|
||||
|
||||
The Attn QAT microbenchmarks now live alongside the kernel package:
|
||||
|
||||
```bash
|
||||
cd fastvideo-kernel
|
||||
python benchmarks/benchmark_flashattn2.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
|
||||
python benchmarks/benchmark_sageattn3.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
|
||||
python benchmarks/benchmark_blockscaled_fp4_attn.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
|
||||
python benchmarks/benchmark_combined.py --output benchmark_attention.png
|
||||
```
|
||||
|
||||
### TurboDiffusion Kernels
|
||||
|
||||
This package also includes kernels from [TurboDiffusion](https://github.com/thu-ml/TurboDiffusion), including INT8 GEMM, Quantization, RMSNorm and LayerNorm.
|
||||
|
||||
@@ -1,16 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
from .api import sageattn_blackwell
|
||||
@@ -1,185 +0,0 @@
|
||||
# Modified from the original SageATtention3 code
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import triton
|
||||
import triton.language as tl
|
||||
import torch.nn.functional as F
|
||||
from typing import Tuple
|
||||
from torch.nn.functional import scaled_dot_product_attention as sdpa
|
||||
import fp4attn_cuda
|
||||
import fp4quant_cuda
|
||||
|
||||
# Centralized block size configuration for sageattn_blackwell kernels
|
||||
# These should match the values in fastvideo/attention/backends/sageattn/blackwell/block_config.h
|
||||
BLOCK_M = 128 # Block size for M dimension (query sequence length)
|
||||
BLOCK_N = 128 # Block size for N dimension (key/value sequence length)
|
||||
|
||||
|
||||
@triton.jit
|
||||
def group_mean_kernel(
|
||||
q_ptr,
|
||||
q_out_ptr,
|
||||
qm_out_ptr,
|
||||
B, H, L, D: tl.constexpr,
|
||||
stride_qb, stride_qh, stride_ql, stride_qd,
|
||||
stride_qmb, stride_qmh, stride_qml, stride_qmd,
|
||||
GROUP_SIZE: tl.constexpr
|
||||
):
|
||||
pid_b = tl.program_id(0)
|
||||
pid_h = tl.program_id(1)
|
||||
pid_group = tl.program_id(2)
|
||||
|
||||
group_start = pid_group * GROUP_SIZE
|
||||
offsets = group_start + tl.arange(0, GROUP_SIZE)
|
||||
|
||||
q_offsets = pid_b * stride_qb + pid_h * stride_qh + offsets[:, None] * stride_ql + tl.arange(0, D)[None, :] * stride_qd
|
||||
q_group = tl.load(q_ptr + q_offsets)
|
||||
|
||||
qm_group = tl.sum(q_group, axis=0) / GROUP_SIZE
|
||||
|
||||
q_group = q_group - qm_group
|
||||
tl.store(q_out_ptr + q_offsets, q_group)
|
||||
|
||||
qm_offset = pid_b * stride_qmb + pid_h * stride_qmh + pid_group * stride_qml + tl.arange(0, D) * stride_qmd
|
||||
tl.store(qm_out_ptr + qm_offset, qm_group)
|
||||
|
||||
|
||||
def triton_group_mean(q: torch.Tensor):
|
||||
B, H, L, D = q.shape
|
||||
GROUP_SIZE = BLOCK_M
|
||||
num_groups = L // GROUP_SIZE
|
||||
|
||||
q_out = torch.empty_like(q) # [B, H, L, D]
|
||||
qm = torch.empty(B, H, num_groups, D, device=q.device, dtype=q.dtype)
|
||||
|
||||
grid = (B, H, num_groups)
|
||||
|
||||
group_mean_kernel[grid](
|
||||
q, q_out, qm,
|
||||
B, H, L, D,
|
||||
q.stride(0), q.stride(1), q.stride(2), q.stride(3),
|
||||
qm.stride(0), qm.stride(1), qm.stride(2), qm.stride(3),
|
||||
GROUP_SIZE=GROUP_SIZE
|
||||
)
|
||||
return q_out, qm
|
||||
|
||||
|
||||
def preprocess_qkv(q: torch.Tensor, k: torch.Tensor, v: torch.Tensor, per_block_mean: bool = True, enable_smoothing_q: bool = False, enable_smoothing_k: bool = False):
|
||||
|
||||
def pad_to_block_size(x):
|
||||
L = x.size(2)
|
||||
pad_len = (BLOCK_M - L % BLOCK_M) % BLOCK_M
|
||||
if pad_len == 0:
|
||||
return x.contiguous()
|
||||
return F.pad(x, (0, 0, 0, pad_len), value=0).contiguous()
|
||||
|
||||
if enable_smoothing_k:
|
||||
k -= k.mean(dim=-2, keepdim=True)
|
||||
q, k, v = map(lambda x: pad_to_block_size(x), [q, k, v])
|
||||
if per_block_mean and enable_smoothing_q:
|
||||
q, qm = triton_group_mean(q)
|
||||
elif enable_smoothing_q:
|
||||
qm = q.mean(dim=-2, keepdim=True)
|
||||
q = q - qm
|
||||
if enable_smoothing_q:
|
||||
delta_s = torch.matmul(qm, k.transpose(-2, -1)).to(torch.float32).contiguous()
|
||||
else: # used to disable q smoothing
|
||||
B, H, L, D = q.shape
|
||||
delta_s = torch.zeros((B, H, L // BLOCK_M, k.shape[2]), device=q.device, dtype=torch.float32)
|
||||
|
||||
return q, k, v, delta_s
|
||||
|
||||
def scale_and_quant_fp4(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
assert x.ndim == 4
|
||||
B, H, N, D = x.shape
|
||||
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
|
||||
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
|
||||
fp4quant_cuda.scaled_fp4_quant(x, packed_fp4, fp8_scale, 1)
|
||||
return packed_fp4, fp8_scale
|
||||
|
||||
def scale_and_quant_fp4_permute(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
assert x.ndim == 4
|
||||
B, H, N, D = x.shape
|
||||
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
|
||||
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
|
||||
fp4quant_cuda.scaled_fp4_quant_permute(x, packed_fp4, fp8_scale, 1)
|
||||
return packed_fp4, fp8_scale
|
||||
|
||||
def scale_and_quant_fp4_transpose(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
assert x.ndim == 4
|
||||
B, H, N, D = x.shape
|
||||
packed_fp4 = torch.empty((B, H, D, N // 2), device=x.device, dtype=torch.uint8)
|
||||
fp8_scale = torch.empty((B, H, D, N // 16), device=x.device, dtype=torch.float8_e4m3fn)
|
||||
fp4quant_cuda.scaled_fp4_quant_trans(x, packed_fp4, fp8_scale, 1)
|
||||
return packed_fp4, fp8_scale
|
||||
|
||||
def blockscaled_fp4_attn(qlist: Tuple,
|
||||
klist: Tuple,
|
||||
vlist: Tuple,
|
||||
delta_s: torch.Tensor,
|
||||
KL: int,
|
||||
is_causal: bool = False,
|
||||
per_block_mean: bool = True,
|
||||
is_bf16: bool = True,
|
||||
single_level_p_quant: bool = False
|
||||
):
|
||||
softmax_scale = (qlist[0].shape[-1] * 2) ** (-0.5)
|
||||
return fp4attn_cuda.fwd(qlist[0], klist[0], vlist[0], qlist[1], klist[1], vlist[1], delta_s, KL, None, softmax_scale, is_causal, per_block_mean, is_bf16, single_level_p_quant)
|
||||
|
||||
|
||||
def sageattn_blackwell(q, k, v, attn_mask = None, is_causal = False, per_block_mean = True, single_level_p_quant = True, **kwargs):
|
||||
"""
|
||||
SageAttention3 Blackwell kernel for FP4 attention.
|
||||
|
||||
Args:
|
||||
q: Query tensor [B, H, L, D]
|
||||
k: Key tensor [B, H, L, D]
|
||||
v: Value tensor [B, H, L, D]
|
||||
attn_mask: Attention mask (not used)
|
||||
is_causal: Whether to use causal masking
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly
|
||||
(standard per-block FP4 quantization like V, no s_P1).
|
||||
If False (default), use two-level quantization:
|
||||
s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1).
|
||||
**kwargs: Additional arguments (ignored)
|
||||
|
||||
Returns:
|
||||
Output tensor [B, H, L, D]
|
||||
"""
|
||||
if q.size(-1) >= 256:
|
||||
print(f"Unsupported Headdim {q.size(-1)}")
|
||||
return sdpa(q, k, v, is_causal = is_causal)
|
||||
QL = q.size(2)
|
||||
KL = k.size(2)
|
||||
is_bf16 = q.dtype == torch.bfloat16
|
||||
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean)
|
||||
qlist_from_cuda = scale_and_quant_fp4(q)
|
||||
klist_from_cuda = scale_and_quant_fp4_permute(k)
|
||||
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
|
||||
o_fp4 = blockscaled_fp4_attn(
|
||||
qlist_from_cuda,
|
||||
klist_from_cuda,
|
||||
vlist_from_cuda,
|
||||
delta_s,
|
||||
KL,
|
||||
is_causal,
|
||||
per_block_mean,
|
||||
is_bf16,
|
||||
single_level_p_quant
|
||||
)[0][:, :, :QL, :].contiguous()
|
||||
return o_fp4
|
||||
@@ -1 +0,0 @@
|
||||
__version__ = "3.0.0.b1"
|
||||
@@ -1,346 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
// Include these 2 headers instead of torch/extension.h since we don't need all of the torch headers.
|
||||
#include <torch/python.h>
|
||||
#include <torch/nn/functional.h>
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <c10/cuda/CUDAGuard.h>
|
||||
|
||||
#include <cutlass/numeric_types.h>
|
||||
|
||||
#include "params.h"
|
||||
#include "launch.h"
|
||||
#include "static_switch.h"
|
||||
#include "block_config.h"
|
||||
|
||||
#define CHECK_DEVICE(x) TORCH_CHECK(x.is_cuda(), #x " must be on CUDA")
|
||||
#define CHECK_SHAPE(x, ...) TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), #x " must have shape (" #__VA_ARGS__ ")")
|
||||
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
|
||||
|
||||
|
||||
void set_params_fprop(Flash_fwd_params ¶ms,
|
||||
// sizes
|
||||
const size_t b,
|
||||
const size_t seqlen_q,
|
||||
const size_t seqlen_k,
|
||||
const size_t unpadded_seqlen_k,
|
||||
const size_t seqlen_q_rounded,
|
||||
const size_t seqlen_k_rounded,
|
||||
const size_t h,
|
||||
const size_t h_k,
|
||||
const size_t d,
|
||||
const size_t d_rounded,
|
||||
// device pointers
|
||||
const at::Tensor q,
|
||||
const at::Tensor k,
|
||||
const at::Tensor v,
|
||||
const at::Tensor delta_s,
|
||||
at::Tensor out,
|
||||
const at::Tensor sfq,
|
||||
const at::Tensor sfk,
|
||||
const at::Tensor sfv,
|
||||
void *cu_seqlens_q_d,
|
||||
void *cu_seqlens_k_d,
|
||||
void *seqused_k,
|
||||
void *p_d,
|
||||
void *softmax_lse_d,
|
||||
float p_dropout,
|
||||
float softmax_scale,
|
||||
int window_size_left,
|
||||
int window_size_right,
|
||||
bool per_block_mean,
|
||||
bool is_bf16,
|
||||
bool single_level_p_quant=false,
|
||||
bool seqlenq_ngroups_swapped=false) {
|
||||
|
||||
// Reset the parameters
|
||||
params = {};
|
||||
// Set the pointers and strides.
|
||||
params.q_ptr = q.data_ptr();
|
||||
params.k_ptr = k.data_ptr();
|
||||
params.v_ptr = v.data_ptr();
|
||||
params.delta_s_ptr = delta_s.data_ptr();
|
||||
params.sfq_ptr = sfq.data_ptr();
|
||||
params.sfk_ptr = sfk.data_ptr();
|
||||
params.sfv_ptr = sfv.data_ptr();
|
||||
|
||||
// All stride are in elements, not bytes.
|
||||
params.q_row_stride = q.stride(-2) * 2;
|
||||
params.k_row_stride = k.stride(-2) * 2;
|
||||
params.v_row_stride = v.stride(-2) * 2;;
|
||||
params.q_head_stride = q.stride(-3) * 2;
|
||||
params.k_head_stride = k.stride(-3) * 2;
|
||||
params.v_head_stride = v.stride(-3) * 2; // for packed q k v
|
||||
|
||||
params.ds_row_stride = delta_s.stride(-2);
|
||||
params.ds_head_stride = delta_s.stride(-3);
|
||||
|
||||
params.sfq_row_stride = sfq.stride(-2);
|
||||
params.sfk_row_stride = sfk.stride(-2);
|
||||
params.sfv_row_stride = sfv.stride(-2);
|
||||
params.sfq_head_stride = sfq.stride(-3);
|
||||
params.sfk_head_stride = sfk.stride(-3);
|
||||
params.sfv_head_stride = sfv.stride(-3);
|
||||
params.o_ptr = out.data_ptr();
|
||||
params.o_row_stride = out.stride(-2);
|
||||
params.o_head_stride = out.stride(-3);
|
||||
|
||||
if (cu_seqlens_q_d == nullptr) {
|
||||
params.q_batch_stride = q.stride(0) * 2;
|
||||
params.k_batch_stride = k.stride(0) * 2;
|
||||
params.v_batch_stride = v.stride(0) * 2;
|
||||
params.ds_batch_stride = delta_s.stride(0);
|
||||
params.sfq_batch_stride = sfq.stride(0);
|
||||
params.sfk_batch_stride = sfk.stride(0);
|
||||
params.sfv_batch_stride = sfv.stride(0);
|
||||
params.o_batch_stride = out.stride(0);
|
||||
if (seqlenq_ngroups_swapped) {
|
||||
params.q_batch_stride *= seqlen_q;
|
||||
params.o_batch_stride *= seqlen_q;
|
||||
}
|
||||
}
|
||||
|
||||
params.cu_seqlens_q = static_cast<int *>(cu_seqlens_q_d);
|
||||
params.cu_seqlens_k = static_cast<int *>(cu_seqlens_k_d);
|
||||
params.seqused_k = static_cast<int *>(seqused_k);
|
||||
|
||||
// P = softmax(QK^T)
|
||||
params.p_ptr = p_d;
|
||||
|
||||
// Softmax sum
|
||||
params.softmax_lse_ptr = softmax_lse_d;
|
||||
|
||||
// Set the dimensions.
|
||||
params.b = b;
|
||||
params.h = h;
|
||||
params.h_k = h_k;
|
||||
params.h_h_k_ratio = h / h_k;
|
||||
params.seqlen_q = seqlen_q;
|
||||
params.seqlen_k = seqlen_k;
|
||||
params.unpadded_seqlen_k = unpadded_seqlen_k;
|
||||
params.seqlen_q_rounded = seqlen_q_rounded;
|
||||
params.seqlen_k_rounded = seqlen_k_rounded;
|
||||
params.d = d;
|
||||
params.d_rounded = d_rounded;
|
||||
|
||||
params.head_divmod = cutlass::FastDivmod(int(h));
|
||||
|
||||
// Set the different scale values.
|
||||
params.scale_softmax = softmax_scale;
|
||||
params.scale_softmax_log2 = softmax_scale * M_LOG2E;
|
||||
__half scale_softmax_log2_half = __float2half(params.scale_softmax_log2);
|
||||
__half2 scale_softmax_log2_half2 = __half2(scale_softmax_log2_half, scale_softmax_log2_half);
|
||||
params.scale_softmax_log2_half2 = reinterpret_cast<uint32_t&>(scale_softmax_log2_half2);
|
||||
|
||||
// Set this to probability of keeping an element to simplify things.
|
||||
params.p_dropout = 1.f - p_dropout;
|
||||
// Convert p from float to int so we don't have to convert the random uint to float to compare.
|
||||
// [Minor] We want to round down since when we do the comparison we use <= instead of <
|
||||
// params.p_dropout_in_uint = uint32_t(std::floor(params.p_dropout * 4294967295.0));
|
||||
// params.p_dropout_in_uint16_t = uint16_t(std::floor(params.p_dropout * 65535.0));
|
||||
params.p_dropout_in_uint8_t = uint8_t(std::floor(params.p_dropout * 255.0));
|
||||
params.rp_dropout = 1.f / params.p_dropout;
|
||||
params.scale_softmax_rp_dropout = params.rp_dropout * params.scale_softmax;
|
||||
TORCH_CHECK(p_dropout < 1.f);
|
||||
#ifdef FLASHATTENTION_DISABLE_DROPOUT
|
||||
TORCH_CHECK(p_dropout == 0.0f, "This flash attention build does not support dropout.");
|
||||
#endif
|
||||
|
||||
// Causal is the special case where window_size_right == 0 and window_size_left < 0.
|
||||
// Local is the more general case where window_size_right >= 0 or window_size_left >= 0.
|
||||
params.is_causal = window_size_left < 0 && window_size_right == 0;
|
||||
params.per_block_mean = per_block_mean;
|
||||
if (per_block_mean) {
|
||||
params.seqlen_s = seqlen_q;
|
||||
} else {
|
||||
params.seqlen_s = flash::BLOCK_M; // size of BLOCK_M
|
||||
}
|
||||
if (window_size_left < 0 && window_size_right >= 0) { window_size_left = seqlen_k; }
|
||||
if (window_size_left >= 0 && window_size_right < 0) { window_size_right = seqlen_k; }
|
||||
params.window_size_left = window_size_left;
|
||||
params.window_size_right = window_size_right;
|
||||
|
||||
#ifdef FLASHATTENTION_DISABLE_LOCAL
|
||||
TORCH_CHECK(params.is_causal || (window_size_left < 0 && window_size_right < 0),
|
||||
"This flash attention build does not support local attention.");
|
||||
#endif
|
||||
|
||||
params.is_seqlens_k_cumulative = true;
|
||||
params.is_bf16 = is_bf16;
|
||||
params.single_level_p_quant = single_level_p_quant;
|
||||
#ifdef FLASHATTENTION_DISABLE_UNEVEN_K
|
||||
TORCH_CHECK(d == d_rounded, "This flash attention build does not support headdim not being a multiple of 32.");
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool IsBF16>
|
||||
void run_mha_fwd_dispatch_dtype(Flash_fwd_params ¶ms, cudaStream_t stream) {
|
||||
using OType = std::conditional_t<IsBF16, cutlass::bfloat16_t, cutlass::half_t>;
|
||||
if (params.d == 64) {
|
||||
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 64, OType>(params, stream);
|
||||
} else if (params.d == 128) {
|
||||
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 128, OType>(params, stream);
|
||||
}
|
||||
}
|
||||
|
||||
void run_mha_fwd(Flash_fwd_params ¶ms, cudaStream_t stream, bool force_split_kernel = false) {
|
||||
BOOL_SWITCH(params.is_bf16, IsBF16, ([&] {
|
||||
run_mha_fwd_dispatch_dtype<IsBF16>(params, stream);
|
||||
}));
|
||||
}
|
||||
|
||||
std::vector<at::Tensor>
|
||||
mha_fwd(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head_size // 2)
|
||||
const at::Tensor &k, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
|
||||
const at::Tensor &v, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
|
||||
const at::Tensor &sfq,
|
||||
const at::Tensor &sfk,
|
||||
const at::Tensor &sfv,
|
||||
const at::Tensor &delta_s,
|
||||
int unpadded_k,
|
||||
c10::optional<at::Tensor> &out_, // batch_size x seqlen_q x num_heads x head_size
|
||||
const float softmax_scale,
|
||||
bool is_causal,
|
||||
bool per_block_mean,
|
||||
bool is_bf16,
|
||||
bool single_level_p_quant=false // If true, use only per-row scale s_P2 (no per-block s_P1)
|
||||
) {
|
||||
|
||||
auto dprops = at::cuda::getCurrentDeviceProperties();
|
||||
bool is_sm120 = dprops->major == 12 && dprops->minor == 0;
|
||||
TORCH_CHECK(is_sm120, "only supports Blackwell GPUs or newer.");
|
||||
|
||||
auto q_dtype = q.dtype();
|
||||
auto sfq_dtype = sfq.dtype();
|
||||
TORCH_CHECK(q_dtype == torch::kUInt8, "q dtype must be uint8");
|
||||
TORCH_CHECK(k.dtype() == q_dtype, "query and key must have the same dtype");
|
||||
TORCH_CHECK(v.dtype() == q_dtype, "query and value must have the same dtype");
|
||||
CHECK_DEVICE(q); CHECK_DEVICE(k); CHECK_DEVICE(v);
|
||||
|
||||
TORCH_CHECK(sfq_dtype == torch::kFloat8_e4m3fn, "q dtype must be uint8");
|
||||
TORCH_CHECK(sfk.dtype() == sfq_dtype, "query and key must have the same dtype");
|
||||
TORCH_CHECK(sfv.dtype() == sfq_dtype, "query and value must have the same dtype");
|
||||
CHECK_DEVICE(sfq); CHECK_DEVICE(sfk); CHECK_DEVICE(sfv);
|
||||
|
||||
TORCH_CHECK(q.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
TORCH_CHECK(k.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
TORCH_CHECK(v.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
TORCH_CHECK(delta_s.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
|
||||
TORCH_CHECK(q.is_contiguous(), "Input tensor must be contiguous");
|
||||
TORCH_CHECK(k.is_contiguous(), "Input tensor must be contiguous");
|
||||
TORCH_CHECK(v.is_contiguous(), "Input tensor must be contiguous");
|
||||
|
||||
const auto sizes = q.sizes();
|
||||
auto opts = q.options();
|
||||
const int batch_size = sizes[0];
|
||||
int seqlen_q = sizes[2];
|
||||
int num_heads = sizes[1];
|
||||
const int head_size_og = sizes[3];
|
||||
const int unpacked_head_size = head_size_og * 2;
|
||||
const int seqlen_k = k.size(2);
|
||||
const int num_heads_k = k.size(1);
|
||||
|
||||
TORCH_CHECK(batch_size > 0, "batch size must be postive");
|
||||
TORCH_CHECK(unpacked_head_size <= 256, "FlashAttention forward only supports head dimension at most 256");
|
||||
TORCH_CHECK(num_heads % num_heads_k == 0, "Number of heads in key/value must divide number of heads in query");
|
||||
TORCH_CHECK(num_heads == num_heads_k, "We do not support MQA/GQA yet");
|
||||
|
||||
TORCH_CHECK(unpacked_head_size == 64 || unpacked_head_size == 128 || unpacked_head_size == 256, "Only support head size 64, 128, and 256 for now");
|
||||
|
||||
CHECK_SHAPE(q, batch_size, num_heads, seqlen_q, head_size_og);
|
||||
CHECK_SHAPE(k, batch_size, num_heads_k, seqlen_k, head_size_og);
|
||||
CHECK_SHAPE(v, batch_size, num_heads_k, unpacked_head_size, seqlen_k/2);
|
||||
// CHECK_SHAPE(delta_s, batch_size, num_heads, seqlen_q / 128, seqlen_k);
|
||||
// CHECK_SHAPE(sfq, batch_size, seqlen_q, num_heads, unpacked_head_size);
|
||||
// CHECK_SHAPE(sfk, batch_size, seqlen_k, num_heads_k, unpacked_head_size);
|
||||
// CHECK_SHAPE(sfv, batch_size, unpacked_head_size, num_heads_k, seqlen_k);
|
||||
TORCH_CHECK(unpacked_head_size % 8 == 0, "head_size must be a multiple of 8");
|
||||
|
||||
auto dtype = is_bf16 ? at::ScalarType::BFloat16 : at::ScalarType::Half;
|
||||
at::Tensor out = torch::empty({batch_size, num_heads, seqlen_q, unpacked_head_size}, opts.dtype(dtype));
|
||||
|
||||
auto round_multiple = [](int x, int m) { return (x + m - 1) / m * m; };
|
||||
// const int head_size = round_multiple(head_size_og, 8);
|
||||
// const int head_size_rounded = round_multiple(head_size, 32);
|
||||
const int seqlen_q_rounded = round_multiple(seqlen_q, flash::BLOCK_M);
|
||||
const int seqlen_k_rounded = round_multiple(seqlen_k, flash::BLOCK_N);
|
||||
|
||||
// Otherwise the kernel will be launched from cuda:0 device
|
||||
// Cast to char to avoid compiler warning about narrowing
|
||||
at::cuda::CUDAGuard device_guard{(char)q.get_device()};
|
||||
|
||||
|
||||
|
||||
auto softmax_lse = torch::empty({batch_size, num_heads, seqlen_q}, opts.dtype(at::kFloat));
|
||||
at::Tensor p;
|
||||
|
||||
Flash_fwd_params params;
|
||||
set_params_fprop(params,
|
||||
batch_size,
|
||||
seqlen_q, seqlen_k, unpadded_k,
|
||||
seqlen_q_rounded, seqlen_k_rounded,
|
||||
num_heads, num_heads_k,
|
||||
unpacked_head_size, unpacked_head_size,
|
||||
q, k, v, delta_s, out,
|
||||
sfq, sfk, sfv,
|
||||
/*cu_seqlens_q_d=*/nullptr,
|
||||
/*cu_seqlens_k_d=*/nullptr,
|
||||
/*seqused_k=*/nullptr,
|
||||
nullptr,
|
||||
softmax_lse.data_ptr(),
|
||||
/*p_dropout=*/0.f,
|
||||
softmax_scale,
|
||||
/*window_size_left=*/-1,
|
||||
/*window_size_right=*/is_causal ? 0 : -1,
|
||||
per_block_mean,
|
||||
is_bf16,
|
||||
single_level_p_quant
|
||||
);
|
||||
// TODO: 132 sm count?
|
||||
auto tile_count_semaphore = is_causal ? torch::full({1}, 132, opts.dtype(torch::kInt32)) : torch::empty({1}, opts.dtype(torch::kInt32));
|
||||
params.tile_count_semaphore = tile_count_semaphore.data_ptr<int>();
|
||||
|
||||
if (seqlen_k > 0) {
|
||||
auto stream = at::cuda::getCurrentCUDAStream().stream();
|
||||
run_mha_fwd(params, stream);
|
||||
} else {
|
||||
// If seqlen_k == 0, then we have an empty tensor. We need to set the output to 0.
|
||||
out.zero_();
|
||||
softmax_lse.fill_(std::numeric_limits<float>::infinity());
|
||||
}
|
||||
|
||||
// at::Tensor out_padded = out;
|
||||
// if (head_size_og % 8 != 0) {
|
||||
// out = out.index({"...", torch::indexing::Slice(torch::indexing::None, head_size_og)});
|
||||
// if (out_.has_value()) { out_.value().copy_(out); }
|
||||
// }
|
||||
|
||||
// return {out, q_padded, k_padded, v_padded, out_padded, softmax_lse, p};
|
||||
// cudaDeviceSynchronize();
|
||||
// auto err = cudaGetLastError();
|
||||
// printf("%s\n", cudaGetErrorString(err));
|
||||
return {out, softmax_lse};
|
||||
}
|
||||
|
||||
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
m.doc() = "FlashAttention";
|
||||
m.def("fwd", &mha_fwd, "Forward pass");
|
||||
}
|
||||
@@ -1,28 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
// Centralized block size configuration for sageattn_blackwell kernels
|
||||
// Block sizes for M and N dimensions
|
||||
namespace flash {
|
||||
// Block size for M dimension (query sequence length)
|
||||
static constexpr int BLOCK_M = 128;
|
||||
|
||||
// Block size for N dimension (key/value sequence length)
|
||||
static constexpr int BLOCK_N = 128;
|
||||
}
|
||||
|
||||
@@ -1,60 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
|
||||
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
namespace flash {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<bool Varlen=true>
|
||||
struct BlockInfo {
|
||||
|
||||
template<typename Params>
|
||||
__device__ BlockInfo(const Params ¶ms, const int bidb)
|
||||
: sum_s_q(!Varlen || params.cu_seqlens_q == nullptr ? -1 : params.cu_seqlens_q[bidb])
|
||||
, sum_s_k(!Varlen || params.cu_seqlens_k == nullptr || !params.is_seqlens_k_cumulative ? -1 : params.cu_seqlens_k[bidb])
|
||||
, actual_seqlen_q(!Varlen || params.cu_seqlens_q == nullptr ? params.seqlen_q : params.cu_seqlens_q[bidb + 1] - sum_s_q)
|
||||
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
|
||||
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
|
||||
, seqlen_k_cache(!Varlen || params.cu_seqlens_k == nullptr ? params.seqlen_k : (params.is_seqlens_k_cumulative ? params.cu_seqlens_k[bidb + 1] - sum_s_k : params.cu_seqlens_k[bidb]))
|
||||
, actual_seqlen_k(params.seqused_k ? params.seqused_k[bidb] : seqlen_k_cache + (params.knew_ptr == nullptr ? 0 : params.seqlen_knew))
|
||||
{
|
||||
}
|
||||
|
||||
template <typename index_t>
|
||||
__forceinline__ __device__ index_t q_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
|
||||
return sum_s_q == -1 ? bidb * batch_stride : uint32_t(sum_s_q) * row_stride;
|
||||
}
|
||||
|
||||
template <typename index_t>
|
||||
__forceinline__ __device__ index_t k_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
|
||||
return sum_s_k == -1 ? bidb * batch_stride : uint32_t(sum_s_k) * row_stride;
|
||||
}
|
||||
|
||||
const int sum_s_q;
|
||||
const int sum_s_k;
|
||||
const int actual_seqlen_q;
|
||||
// We have to have seqlen_k_cache declared before actual_seqlen_k, otherwise actual_seqlen_k is set to 0.
|
||||
const int seqlen_k_cache;
|
||||
const int actual_seqlen_k;
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,149 +0,0 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2023 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Blocked Scale configs specific for SM100 BlockScaled MMA
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/layout/matrix.h"
|
||||
|
||||
#include "cute/int_tuple.hpp"
|
||||
#include "cute/atom/mma_traits_sm100.hpp"
|
||||
|
||||
namespace flash {
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
using namespace cute;
|
||||
|
||||
template<int SFVecSize, UMMA::Major major = UMMA::Major::K>
|
||||
struct BlockScaledBasicChunk {
|
||||
|
||||
using Blk_MN = _64;
|
||||
using Blk_SF = _4;
|
||||
|
||||
using SfAtom = Layout< Shape< Shape<_16,_4>, Shape<Int<SFVecSize>, _4>>,
|
||||
Stride<Stride<_16,_4>, Stride< _0, _1>>>;
|
||||
};
|
||||
|
||||
template<int SFVecSize_>
|
||||
struct BlockScaledConfig {
|
||||
// We are creating the SFA and SFB tensors' layouts in the collective since they always have the same layout.
|
||||
// k-major order
|
||||
static constexpr int SFVecSize = SFVecSize_;
|
||||
static constexpr int MMA_NSF = 4; // SFVecSize, MMA_NSF
|
||||
using BlkScaledChunk = BlockScaledBasicChunk<SFVecSize>;
|
||||
using Blk_MN = _64;
|
||||
using Blk_SF = _4;
|
||||
using mnBasicBlockShape = Shape<_16,_4>;
|
||||
using mnBasicBlockStride = Stride<_16,_4>;
|
||||
using kBasicBlockShape = Shape<Int<SFVecSize>, Int<MMA_NSF>>; // SFVecSize, MMA_NSF
|
||||
using kBasicBlockStride = Stride<_0, _1>;
|
||||
using SfAtom = Layout< Shape< mnBasicBlockShape, kBasicBlockShape>,
|
||||
Stride<mnBasicBlockStride, kBasicBlockStride>>;
|
||||
|
||||
using LayoutSF = decltype(blocked_product(SfAtom{},
|
||||
make_layout(
|
||||
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
|
||||
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))));
|
||||
// A single indivisible block will hold 4 scale factors of 64 rows/columns (A/B matrix).
|
||||
// 4 is chosen to make consecutive 32bits of data to have scale factors for only a single row (col). 32bits corresponds to the TMEM word size
|
||||
using Blk_Elems = decltype(Blk_MN{} * Blk_SF{});
|
||||
using sSF_strideMN = decltype(prepend(Blk_Elems{}, mnBasicBlockStride{}));
|
||||
|
||||
|
||||
// The following function is provided for user fill dynamic problem size to the layout_SFA.
|
||||
template < class ProblemShape>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
tile_atom_to_shape_SFQKV(ProblemShape problem_shape) {
|
||||
auto [Seqlen, Dim, HeadNum, Batch] = problem_shape;
|
||||
return tile_to_shape(SfAtom{}, make_shape(Seqlen, Dim, HeadNum, Batch), Step<_2,_1,_3,_4>{});
|
||||
}
|
||||
|
||||
// The following function is provided for user fill dynamic problem size to the layout_SFB.
|
||||
template <class ProblemShape>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
tile_atom_to_shape_SFVt(ProblemShape problem_shape) {
|
||||
auto [Dim, Seqlen, HeadNum, Batch] = problem_shape;
|
||||
return tile_to_shape(SfAtom{}, make_shape(Dim, Seqlen, HeadNum, Batch), Step<_2,_1,_3,_4>{});
|
||||
}
|
||||
|
||||
template<class TiledMma, class TileShape_MNK>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
deduce_smem_layoutSFQ(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
|
||||
|
||||
using sSFQ_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
|
||||
using sSFQ_shapeM = decltype(prepend(size<0>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
|
||||
using sSFQ_strideM = sSF_strideMN;
|
||||
using sSFQ_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<0>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
|
||||
using sSFQ_shape = decltype(make_shape(sSFQ_shapeM{}, sSFQ_shapeK{}));
|
||||
using sSFQ_stride = decltype(make_stride(sSFQ_strideM{}, sSFQ_strideK{}));
|
||||
using SmemLayoutAtomSFQ = decltype(make_layout(sSFQ_shape{}, sSFQ_stride{}));
|
||||
return SmemLayoutAtomSFQ{};
|
||||
}
|
||||
|
||||
template<class TiledMma, class TileShape_MNK>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
deduce_smem_layoutSFKV(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
|
||||
|
||||
using sSFK_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
|
||||
using sSFK_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
|
||||
using sSFK_strideN = sSF_strideMN;
|
||||
using sSFK_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
|
||||
using sSFK_shape = decltype(make_shape(sSFK_shapeN{}, sSFK_shapeK{}));
|
||||
using sSFK_stride = decltype(make_stride(sSFK_strideN{}, sSFK_strideK{}));
|
||||
using SmemLayoutAtomSFK = decltype(make_layout(sSFK_shape{}, sSFK_stride{}));
|
||||
return SmemLayoutAtomSFK{};
|
||||
}
|
||||
|
||||
template<class TiledMma, class TileShape_MNK>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
deduce_smem_layoutSFVt(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
|
||||
|
||||
using sSFVt_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
|
||||
using sSFVt_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
|
||||
using sSFVt_strideN = sSF_strideMN;
|
||||
using sSFVt_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
|
||||
using sSFVt_shape = decltype(make_shape(sSFVt_shapeN{}, sSFVt_shapeK{}));
|
||||
using sSFVt_stride = decltype(make_stride(sSFVt_strideN{}, sSFVt_strideK{}));
|
||||
using SmemLayoutAtomSFVt = decltype(make_layout(sSFVt_shape{}, sSFVt_stride{}));
|
||||
return SmemLayoutAtomSFVt{};
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,327 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#include "cute/arch/mma_sm120.hpp"
|
||||
#include "cute/atom/mma_traits_sm120.hpp"
|
||||
#include "cute/atom/mma_atom.hpp"
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/float8.h"
|
||||
#include "cutlass/float_subbyte.h"
|
||||
|
||||
namespace cute::SM120::BLOCKSCALED {
|
||||
|
||||
using cutlass::float_e2m1_t;
|
||||
using cutlass::float_ue4m3_t;
|
||||
|
||||
// MMA.SF 16x32x64 TN E2M1 x E2M1 with SF E4M3
|
||||
struct SM120_16x32x64_TN_VS_NVFP4 {
|
||||
using DRegisters = float[16];
|
||||
using ARegisters = uint32_t[4];
|
||||
using BRegisters = uint32_t[8];
|
||||
using CRegisters = float[16];
|
||||
|
||||
static constexpr int SFBits = 32;
|
||||
using RegTypeSF = cute::uint_bit_t<SFBits>;
|
||||
|
||||
using SFARegisters = RegTypeSF[1];
|
||||
using SFBRegisters = RegTypeSF[1];
|
||||
|
||||
CUTE_HOST_DEVICE static void
|
||||
fma(float & d0 , float & d1 , float & d2 , float & d3 ,
|
||||
float & d4 , float & d5 , float & d6 , float & d7 ,
|
||||
float & d8 , float & d9 , float & d10, float & d11,
|
||||
float & d12, float & d13, float & d14, float & d15,
|
||||
uint32_t const& a0 , uint32_t const& a1 , uint32_t const& a2 , uint32_t const& a3 ,
|
||||
uint32_t const& b0 , uint32_t const& b1 , uint32_t const& b2 , uint32_t const& b3 ,
|
||||
uint32_t const& b4 , uint32_t const& b5 , uint32_t const& b6 , uint32_t const& b7 ,
|
||||
float const & c0 , float const & c1 , float const & c2 , float const & c3 ,
|
||||
float const & c4 , float const & c5 , float const & c6 , float const & c7 ,
|
||||
float const & c8 , float const & c9 , float const & c10 , float const & c11,
|
||||
float const & c12, float const & c13, float const & c14, float const & c15,
|
||||
RegTypeSF const& sfa0,
|
||||
RegTypeSF const& sfb0)
|
||||
{
|
||||
static constexpr uint16_t tidA = 0;
|
||||
static constexpr uint16_t bidA = 0;
|
||||
static constexpr uint16_t bidB = 0;
|
||||
static constexpr uint16_t tidB0 = 0;
|
||||
static constexpr uint16_t tidB1 = 1;
|
||||
static constexpr uint16_t tidB2 = 2;
|
||||
static constexpr uint16_t tidB3 = 3;
|
||||
|
||||
#if defined(CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED)
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d0), "=f"(d1), "=f"(d8), "=f"(d9)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b0), "r"(b1),
|
||||
"f"(c0), "f"(c1), "f"(c8), "f"(c9),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB0));
|
||||
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d2), "=f"(d3), "=f"(d10), "=f"(d11)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b2), "r"(b3),
|
||||
"f"(c2), "f"(c3), "f"(c10), "f"(c11),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB1));
|
||||
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d4), "=f"(d5), "=f"(d12), "=f"(d13)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b4), "r"(b5),
|
||||
"f"(c4), "f"(c5), "f"(c12), "f"(c13),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB2));
|
||||
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d6), "=f"(d7), "=f"(d14), "=f"(d15)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b6), "r"(b7),
|
||||
"f"(c6), "f"(c7), "f"(c14), "f"(c15),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB3));
|
||||
#else
|
||||
CUTE_INVALID_CONTROL_PATH("Attempting to use SM120::BLOCKSCALED::SM120_16x8x64_TN_VS without CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED");
|
||||
#endif
|
||||
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace cute::SM120::BLOCKSCALED
|
||||
|
||||
namespace cute {
|
||||
|
||||
// MMA NVFP4 16x32x64 TN
|
||||
template <>
|
||||
struct MMA_Traits<SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4>
|
||||
{
|
||||
// The MMA accepts 4-bit inputs regardless of the types for A and B
|
||||
using ValTypeA = uint4_t;
|
||||
using ValTypeB = uint4_t;
|
||||
|
||||
using ValTypeD = float;
|
||||
using ValTypeC = float;
|
||||
|
||||
using ValTypeSF = cutlass::float_ue4m3_t;
|
||||
constexpr static int SFVecSize = 16;
|
||||
|
||||
using Shape_MNK = Shape<_16,_32,_64>;
|
||||
using ThrID = Layout<_32>;
|
||||
|
||||
// (T32,V32) -> (M16,K64)
|
||||
using ALayout = Layout<Shape <Shape < _4,_8>,Shape < _8,_2, _2>>,
|
||||
Stride<Stride<_128,_1>,Stride<_16,_8,_512>>>;
|
||||
// (T32,V64) -> (N32,K64)
|
||||
using BLayout = Layout<Shape <Shape < _4,_8>,Shape <_8, _2, _4>>,
|
||||
Stride<Stride<_256,_1>,Stride<_32,_1024, _8>>>;
|
||||
// (T32,V64) -> (M16,K64)
|
||||
using SFALayout = Layout<Shape <Shape <_2,_2,_8>,_64>,
|
||||
Stride<Stride<_8,_0,_1>,_16>>;
|
||||
// (T32,V64) -> (N32,K64)
|
||||
using SFBLayout = Layout<Shape <Shape <_4,_8>,_64>,
|
||||
Stride<Stride<_8,_1>, _32>>;
|
||||
// (T32,V16) -> (M16,N32)
|
||||
using CLayout = Layout<Shape <Shape < _4,_8>,Shape < Shape<_2, _4>,_2>>,
|
||||
Stride<Stride<_32,_1>,Stride<Stride<_16, _128>,_8>>>;
|
||||
};
|
||||
|
||||
|
||||
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<0>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<1>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
|
||||
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<1>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<2>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFATensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
|
||||
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
return thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
}
|
||||
|
||||
template <class SFATensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
return make_fragment_like<ValTypeSF>(partition_SFA(sfatensor, thread_mma));
|
||||
}
|
||||
|
||||
template <class SFBTensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
|
||||
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
return thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
}
|
||||
|
||||
template <class SFBTensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
return make_fragment_like<ValTypeSF>(partition_SFB(sfbtensor, thread_mma));
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFA_TV(TiledMma& mma)
|
||||
{
|
||||
// (M,K) -> (M,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto atile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<1>{} , Int<0>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFB_TV(TiledMma& mma)
|
||||
{
|
||||
// (N,K) -> (N,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto btile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<0>{} , Int<1>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
} // namespace cute
|
||||
@@ -1,222 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cutlass/cutlass.h>
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include "cutlass/gemm/collective/collective_builder.hpp"
|
||||
#include "named_barrier.h"
|
||||
#include "utils.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <typename Ktraits>
|
||||
struct CollectiveEpilogueFwd{
|
||||
|
||||
using Element = typename Ktraits::ElementOut;
|
||||
static constexpr int kBlockM = Ktraits::kBlockM;
|
||||
static constexpr int kBlockN = Ktraits::kBlockN;
|
||||
static constexpr int kHeadDim = Ktraits::kHeadDim;
|
||||
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
|
||||
static constexpr int kNWarps = Ktraits::kNWarps;
|
||||
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
|
||||
static constexpr int NumMmaThreads = kNThreads - cutlass::NumThreadsPerWarpGroup;
|
||||
|
||||
using GmemTiledCopyOTMA = cute::SM90_TMA_STORE;
|
||||
|
||||
// These are for storing the output tensor without TMA (e.g., for setting output to zero)
|
||||
static constexpr int kGmemElemsPerLoad = sizeof(cute::uint128_t) / sizeof(Element);
|
||||
static_assert(kHeadDim % kGmemElemsPerLoad == 0, "kHeadDim must be a multiple of kGmemElemsPerLoad");
|
||||
static constexpr int kGmemThreadsPerRow = kHeadDim / kGmemElemsPerLoad;
|
||||
static_assert(NumMmaThreads % kGmemThreadsPerRow == 0, "NumMmaThreads must be a multiple of kGmemThreadsPerRow");
|
||||
using GmemLayoutAtom = Layout<Shape <Int<NumMmaThreads / kGmemThreadsPerRow>, Int<kGmemThreadsPerRow>>,
|
||||
Stride<Int<kGmemThreadsPerRow>, _1>>;
|
||||
using GmemTiledCopyO = decltype(
|
||||
make_tiled_copy(Copy_Atom<DefaultCopy, Element>{},
|
||||
GmemLayoutAtom{},
|
||||
Layout<Shape<_1, Int<kGmemElemsPerLoad>>>{})); // Val layout, 8 or 16 vals per store
|
||||
|
||||
using SmemLayoutO = typename Ktraits::SmemLayoutO;
|
||||
|
||||
using SmemCopyAtomO = Copy_Atom<SM90_U32x2_STSM_N, Element>;
|
||||
using SharedStorage = cute::array_aligned<Element, cute::cosize_v<SmemLayoutO>>;
|
||||
|
||||
using ShapeO = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen_q, d, head, batch)
|
||||
using StrideO = cute::Stride<int64_t, _1, int64_t, int64_t>;
|
||||
using StrideLSE = cute::Stride<_1, int64_t, int64_t>; // (seqlen_q, head, batch)
|
||||
|
||||
using TMA_O = decltype(make_tma_copy(
|
||||
GmemTiledCopyOTMA{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element*>(nullptr)), repeat_like(StrideO{}, int32_t(0)), StrideO{}),
|
||||
SmemLayoutO{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{})); // no mcast for O
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
Element* ptr_O;
|
||||
ShapeO const shape_O;
|
||||
StrideO const stride_O;
|
||||
float* ptr_LSE;
|
||||
StrideLSE const stride_LSE;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
Element* ptr_O;
|
||||
ShapeO const shape_O;
|
||||
StrideO const stride_O;
|
||||
float* ptr_LSE;
|
||||
StrideLSE const stride_LSE;
|
||||
TMA_O tma_store_O;
|
||||
};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
Tensor mO = make_tensor(make_gmem_ptr(args.ptr_O), args.shape_O, args.stride_O);
|
||||
TMA_O tma_store_O = make_tma_copy(
|
||||
GmemTiledCopyOTMA{},
|
||||
mO,
|
||||
SmemLayoutO{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{}); // no mcast for O
|
||||
return {args.ptr_O, args.shape_O, args.stride_O, args.ptr_LSE, args.stride_LSE, tma_store_O};
|
||||
}
|
||||
|
||||
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
|
||||
CUTLASS_DEVICE
|
||||
static void prefetch_tma_descriptors(Params const& epilogue_params) {
|
||||
cute::prefetch_tma_descriptor(epilogue_params.tma_store_O.get_tma_descriptor());
|
||||
}
|
||||
|
||||
template <typename SharedStorage, typename FrgTensorO, typename TiledMma>
|
||||
CUTLASS_DEVICE void
|
||||
mma_store(
|
||||
SharedStorage& shared_storage,
|
||||
TiledMma tiled_mma,
|
||||
FrgTensorO const& tOrO,
|
||||
int thread_idx
|
||||
){
|
||||
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
|
||||
auto smem_tiled_copy_O = make_tiled_copy_C(SmemCopyAtomO{}, tiled_mma);
|
||||
auto smem_thr_copy_O = smem_tiled_copy_O.get_thread_slice(thread_idx);
|
||||
constexpr int numel = decltype(size(tOrO))::value;
|
||||
cutlass::NumericArrayConverter<Element, float, numel> convert_op;
|
||||
// HACK: this requires tensor to be "contiguous"
|
||||
auto frag = convert_op(*reinterpret_cast<const cutlass::Array<float, numel> *>(tOrO.data()));
|
||||
auto tOrO_out = make_tensor(make_rmem_ptr<Element>(&frag), tOrO.layout());
|
||||
Tensor taccOrO = smem_thr_copy_O.retile_S(tOrO_out); // ((Atom,AtomNum), MMA_M, MMA_N)
|
||||
Tensor taccOsO = smem_thr_copy_O.partition_D(sO); // ((Atom,AtomNum),PIPE_M,PIPE_N)
|
||||
cute::copy(smem_tiled_copy_O, taccOrO, taccOsO);
|
||||
cutlass::arch::fence_view_async_shared(); // ensure smem writes are visible to TMA
|
||||
}
|
||||
|
||||
template<typename SharedStorage, typename Params, typename WorkTileInfo, typename SchedulerParams>
|
||||
CUTLASS_DEVICE void
|
||||
tma_store(
|
||||
SharedStorage& shared_storage,
|
||||
Params const& epilogue_params,
|
||||
WorkTileInfo work_tile_info,
|
||||
SchedulerParams const& scheduler_params,
|
||||
int thread_idx
|
||||
) {
|
||||
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
|
||||
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
|
||||
Tensor mO = epilogue_params.tma_store_O.get_tma_tensor(epilogue_params.shape_O);
|
||||
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
|
||||
auto block_tma_O = epilogue_params.tma_store_O.get_slice(_0{});
|
||||
Tensor tOgO = block_tma_O.partition_D(gO); // (TMA, TMA_M, TMA_K)
|
||||
Tensor tOsO = block_tma_O.partition_S(sO); // (TMA, TMA_M, TMA_K)
|
||||
|
||||
// auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
|
||||
// Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
|
||||
// Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
|
||||
|
||||
// Tensor caccO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{}));
|
||||
// auto thread_mma = tiled_mma.get_thread_slice(thread_idx);
|
||||
// Tensor taccOcO = thread_mma.partition_C(caccO); // (MMA,MMA_M,MMA_K)
|
||||
// static_assert(decltype(size<0, 0>(taccOcO))::value == 2);
|
||||
// static_assert(decltype(size<0, 1>(taccOcO))::value == 2);
|
||||
// // // // taccOcO has shape ((2, 2, V), MMA_M, MMA_K), we only take only the row indices.
|
||||
// Tensor taccOcO_row = taccOcO(make_coord(_0{}, _), _, _0{});
|
||||
// CUTE_STATIC_ASSERT_V(size(lse) == size(taccOcO_row)); // MMA_M
|
||||
// if (get<1>(taccOcO_row(_0{})) == 0) {
|
||||
// #pragma unroll
|
||||
// for (int mi = 0; mi < size(lse); ++mi) {
|
||||
// const int row = get<0>(taccOcO_row(mi));
|
||||
// if (row < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(row) = lse(mi); }
|
||||
// }
|
||||
// }
|
||||
|
||||
// if (cutlass::canonical_warp_idx_sync() == kNWarps - 1) {
|
||||
// cutlass::arch::NamedBarrier::sync(NumMmaThreads + cutlass::NumThreadsPerWarp,
|
||||
// static_cast<uint32_t>(FP4NamedBarriers::EpilogueBarrier));
|
||||
// int const lane_predicate = cute::elect_one_sync();
|
||||
// if (lane_predicate) {
|
||||
// cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
|
||||
// tma_store_arrive();
|
||||
// }
|
||||
// }
|
||||
cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
|
||||
tma_store_arrive();
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
store_tail() {
|
||||
tma_store_wait<0>();
|
||||
}
|
||||
|
||||
// Write 0 to output and -inf to LSE
|
||||
CUTLASS_DEVICE void
|
||||
store_zero(
|
||||
Params const& epilogue_params,
|
||||
int thread_idx,
|
||||
cute::tuple<int32_t, int32_t, int32_t> const& block_coord
|
||||
) {
|
||||
auto [m_block, bidh, bidb] = block_coord;
|
||||
Tensor mO = make_tensor(make_gmem_ptr(epilogue_params.ptr_O), epilogue_params.shape_O, epilogue_params.stride_O);
|
||||
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
|
||||
auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
|
||||
Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
|
||||
Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
|
||||
|
||||
GmemTiledCopyO gmem_tiled_copy_O;
|
||||
auto gmem_thr_copy_O = gmem_tiled_copy_O.get_thread_slice(thread_idx);
|
||||
Tensor tOgO = gmem_thr_copy_O.partition_D(gO);
|
||||
Tensor tOrO = make_fragment_like(tOgO);
|
||||
clear(tOrO);
|
||||
// Construct identity layout for sO
|
||||
Tensor cO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{})); // (BLK_M,BLK_K) -> (blk_m,blk_k)
|
||||
// Repeat the partitioning with identity layouts
|
||||
Tensor tOcO = gmem_thr_copy_O.partition_D(cO);
|
||||
Tensor tOpO = make_tensor<bool>(make_shape(size<2>(tOgO)));
|
||||
#pragma unroll
|
||||
for (int k = 0; k < size(tOpO); ++k) { tOpO(k) = get<1>(tOcO(_0{}, _0{}, k)) < get<1>(epilogue_params.shape_O); }
|
||||
// Clear_OOB_K must be false since we don't want to write zeros to gmem
|
||||
flash::copy</*Is_even_MN=*/false, /*Is_even_K=*/false, /*Clear_OOB_MN=*/false, /*Clear_OOB_K=*/false>(
|
||||
gmem_tiled_copy_O, tOrO, tOgO, tOcO, tOpO, get<0>(epilogue_params.shape_O) - m_block * kBlockM
|
||||
);
|
||||
static_assert(kBlockM <= NumMmaThreads);
|
||||
if (thread_idx < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(thread_idx) = INFINITY; }
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,202 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cute/algorithm/copy.hpp"
|
||||
#include "cute/atom/mma_atom.hpp"
|
||||
#include "cutlass/gemm/collective/collective_builder.hpp"
|
||||
#include "cute/tensor.hpp"
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/layout/layout.h"
|
||||
#include "cutlass/numeric_types.h"
|
||||
#include "cutlass/pipeline/pipeline.hpp"
|
||||
|
||||
#include "blockscaled_layout.h"
|
||||
#include "cute_extension.h"
|
||||
#include "named_barrier.h"
|
||||
using namespace cute;
|
||||
|
||||
template <
|
||||
int kStages,
|
||||
int EpiStages,
|
||||
typename Element,
|
||||
typename ElementSF,
|
||||
typename OutputType,
|
||||
typename SmemLayoutQ,
|
||||
typename SmemLayoutK,
|
||||
typename SmemLayoutV,
|
||||
typename SmemLayoutDS,
|
||||
typename SmemLayoutO,
|
||||
typename SmemLayoutSFQ,
|
||||
typename SmemLayoutSFK,
|
||||
typename SmemLayoutSFV
|
||||
>
|
||||
struct SharedStorageQKVOwithSF : cute::aligned_struct<128, _0>{
|
||||
|
||||
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutQ>> smem_q;
|
||||
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutK>> smem_k;
|
||||
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFQ>> smem_SFQ;
|
||||
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFK>> smem_SFK;
|
||||
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFV>> smem_SFV;
|
||||
alignas(1024) cute::ArrayEngine<float, cute::cosize_v<SmemLayoutDS>> smem_ds;
|
||||
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutV>> smem_v;
|
||||
alignas(1024) cute::ArrayEngine<OutputType, cute::cosize_v<SmemLayoutO>> smem_o;
|
||||
|
||||
struct {
|
||||
alignas(16) typename cutlass::PipelineTmaAsync<1>::SharedStorage pipeline_q;
|
||||
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_k;
|
||||
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_v;
|
||||
alignas(16) typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>::SharedStorage barrier_o;
|
||||
int tile_count_semaphore;
|
||||
};
|
||||
};
|
||||
|
||||
template <
|
||||
int kHeadDim_,
|
||||
int kBlockM_,
|
||||
int kBlockN_,
|
||||
int kStages_,
|
||||
int kClusterM_,
|
||||
bool BlockMean_,
|
||||
typename ElementPairType_ = cutlass::nv_float4_t<cutlass::float_e2m1_t>,
|
||||
typename ElementOut_ = cutlass::bfloat16_t
|
||||
>
|
||||
struct Flash_fwd_kernel_traits {
|
||||
static constexpr int kBlockM = kBlockM_;
|
||||
static constexpr int kBlockN = kBlockN_;
|
||||
static constexpr int kHeadDim = kHeadDim_;
|
||||
static constexpr bool BlockMean = BlockMean_;
|
||||
static constexpr bool SmoothQ = true;
|
||||
static_assert(kHeadDim % 32 == 0);
|
||||
static_assert(kBlockM == 64 || kBlockM == 128);
|
||||
static constexpr int kNWarps = kBlockM == 128 ? 12 : 8;
|
||||
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
|
||||
static constexpr int kClusterM = kClusterM_;
|
||||
static constexpr int kStages = kStages_;
|
||||
static constexpr int EpiStages = 1;
|
||||
static constexpr int NumSFQK = kHeadDim / 16;
|
||||
static constexpr int NumSFPV = kBlockN / 16;
|
||||
using ElementSF = cutlass::float_ue4m3_t;
|
||||
using Element = cutlass::float_e2m1_t;
|
||||
using ElementAccum = float;
|
||||
using ElementOut = ElementOut_;
|
||||
using index_t = int64_t;
|
||||
static constexpr auto SFVectorSize = 16;
|
||||
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
|
||||
using ClusterShape_MNK = Shape<_1, _1, _1>;
|
||||
using PermTileM = decltype(cute::min(size<0>(TileShape_MNK{}), _128{}));
|
||||
using PermTileN = _32;
|
||||
using PermTileK = Int<kHeadDim>;
|
||||
|
||||
using ElementQMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
|
||||
using ElementKMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
|
||||
|
||||
using AtomLayoutMNK = std::conditional_t<kBlockM == 128,
|
||||
Layout<Shape<_8, _1, _1>>,
|
||||
Layout<Shape<_4, _1, _1>>
|
||||
>;
|
||||
using TiledMmaQK = decltype(cute::make_tiled_mma(
|
||||
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
|
||||
AtomLayoutMNK{},
|
||||
Tile<PermTileM, PermTileN, PermTileK>{}
|
||||
));
|
||||
|
||||
using TiledMmaPV = decltype(cute::make_tiled_mma(
|
||||
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
|
||||
AtomLayoutMNK{},
|
||||
Tile<PermTileM, _32, PermTileK>{}
|
||||
));
|
||||
|
||||
static constexpr int MMA_NSF = size<2>(typename TiledMmaQK::AtomShape_MNK{}) / SFVectorSize;
|
||||
|
||||
using GmemTiledCopy = SM90_TMA_LOAD;
|
||||
using GmemTiledCopySF = SM90_TMA_LOAD;
|
||||
|
||||
using SmemLayoutAtomQ = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutAtomK = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutAtomV = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutAtomVt = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<1>(TileShape_MNK{}))>());
|
||||
using SmemLayoutQ = decltype(tile_to_shape(SmemLayoutAtomQ{}, select<0, 2>(TileShape_MNK{})));
|
||||
using SmemLayoutK =
|
||||
decltype(tile_to_shape(SmemLayoutAtomK{},
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
|
||||
using SmemLayoutV =
|
||||
decltype(tile_to_shape(SmemLayoutAtomV{},
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
|
||||
using SmemLayoutVt =
|
||||
decltype(tile_to_shape(SmemLayoutAtomVt{},
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
|
||||
using SmemLayoutAtomDS = Layout<Shape<Int<kBlockM>, Int<kBlockN>>, Stride<_0, _1>>;
|
||||
using SmemLayoutDS =
|
||||
decltype(tile_to_shape(SmemLayoutAtomDS{},
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
|
||||
|
||||
using SmemCopyAtomQ = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
|
||||
using SmemCopyAtomKV = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
|
||||
using SmemCopyAtomSF = Copy_Atom<UniversalCopy<ElementSF>, ElementSF>;
|
||||
using SmemCopyAtomDS = Copy_Atom<UniversalCopy<float>, float>;
|
||||
|
||||
using BlkScaledConfig = flash::BlockScaledConfig<SFVectorSize>;
|
||||
using LayoutSF = typename BlkScaledConfig::LayoutSF;
|
||||
using SfAtom = typename BlkScaledConfig::SfAtom;
|
||||
using SmemLayoutAtomSFQ = decltype(BlkScaledConfig::deduce_smem_layoutSFQ(TiledMmaQK{}, TileShape_MNK{}));
|
||||
using SmemLayoutAtomSFK = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaQK{}, TileShape_MNK{}));
|
||||
using SmemLayoutAtomSFV = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaPV{}, TileShape_MNK{}));
|
||||
using SmemLayoutAtomSFVt = decltype(BlkScaledConfig::deduce_smem_layoutSFVt(TiledMmaPV{}, Shape<Int<kBlockM>, Int<kHeadDim>, Int<kBlockN>>{}));
|
||||
using LayoutSFP = decltype(
|
||||
make_layout(
|
||||
make_shape(make_shape(_16{}, _4{}), _1{}, Int<kBlockN / 64>{}),
|
||||
make_stride(make_stride(_0{}, _1{}), _0{}, _4{})
|
||||
)
|
||||
);
|
||||
using LayoutP = decltype(
|
||||
make_layout(
|
||||
make_shape(make_shape(_8{}, _2{}, _2{}), _1{}, Int<kBlockN / 64>{}),
|
||||
make_stride(make_stride(_1{}, _8{}, _16{}), _0{}, _32{})
|
||||
)
|
||||
);
|
||||
using SmemLayoutSFQ = decltype(make_layout(
|
||||
shape(SmemLayoutAtomSFQ{}),
|
||||
stride(SmemLayoutAtomSFQ{})
|
||||
));
|
||||
using SmemLayoutSFK = decltype(make_layout(
|
||||
append(shape(SmemLayoutAtomSFK{}), Int<kStages>{}),
|
||||
append(stride(SmemLayoutAtomSFK{}), size(filter_zeros(SmemLayoutAtomSFK{})))
|
||||
));
|
||||
using SmemLayoutSFV = decltype(make_layout(
|
||||
append(shape(SmemLayoutAtomSFV{}), Int<kStages>{}),
|
||||
append(stride(SmemLayoutAtomSFV{}), size(filter_zeros(SmemLayoutAtomSFV{})))
|
||||
));
|
||||
using SmemLayoutSFVt = decltype(make_layout(
|
||||
append(shape(SmemLayoutAtomSFVt{}), Int<kStages>{}),
|
||||
append(stride(SmemLayoutAtomSFVt{}), size(filter_zeros(SmemLayoutAtomSFVt{})))
|
||||
));
|
||||
|
||||
using SmemLayoutAtomO = decltype(cutlass::gemm::collective::detail::ss_smem_selector<GMMA::Major::K, ElementOut,
|
||||
decltype(cute::get<0>(TileShape_MNK{})), decltype(cute::get<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutO = decltype(tile_to_shape(SmemLayoutAtomO{}, select<0, 2>(TileShape_MNK{}), Step<_1, _2>{}));
|
||||
using SharedStorage = SharedStorageQKVOwithSF<kStages, EpiStages, Element, ElementSF, ElementOut,
|
||||
SmemLayoutQ, SmemLayoutK, SmemLayoutV, SmemLayoutDS,
|
||||
SmemLayoutO, SmemLayoutSFQ, SmemLayoutSFK, SmemLayoutSFVt>;
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<kStages>;
|
||||
using PipelineState = typename cutlass::PipelineState<kStages>;
|
||||
using MainloopPipelineQ = cutlass::PipelineTmaAsync<1>;
|
||||
using PipelineParamsQ = typename MainloopPipelineQ::Params;
|
||||
using PipelineStateQ = typename cutlass::PipelineState<1>;
|
||||
using EpilogueBarrier = typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>;
|
||||
};
|
||||
|
||||
@@ -1,204 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include <cutlass/cutlass.h>
|
||||
#include <cutlass/arch/reg_reconfig.h>
|
||||
#include <cutlass/array.h>
|
||||
#include <cutlass/numeric_types.h>
|
||||
#include <cutlass/numeric_conversion.h>
|
||||
#include "cutlass/pipeline/pipeline.hpp"
|
||||
|
||||
#include "params.h"
|
||||
#include "utils.h"
|
||||
#include "tile_scheduler.h"
|
||||
#include "mainloop_tma_ws.h"
|
||||
#include "epilogue_tma_ws.h"
|
||||
#include "named_barrier.h"
|
||||
#include "softmax_fused.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <typename Ktraits, bool Is_causal, typename TileScheduler>
|
||||
__global__ void __launch_bounds__(Ktraits::kNWarps * cutlass::NumThreadsPerWarp, 1)
|
||||
compute_attn_ws(CUTE_GRID_CONSTANT Flash_fwd_params const params,
|
||||
CUTE_GRID_CONSTANT typename CollectiveMainloopFwd<Ktraits, Is_causal>::Params const mainloop_params,
|
||||
CUTE_GRID_CONSTANT typename CollectiveEpilogueFwd<Ktraits>::Params const epilogue_params,
|
||||
CUTE_GRID_CONSTANT typename TileScheduler::Params const scheduler_params
|
||||
) {
|
||||
|
||||
using Element = typename Ktraits::Element;
|
||||
using ElementAccum = typename Ktraits::ElementAccum;
|
||||
using SoftType = ElementAccum;
|
||||
using TileShape_MNK = typename Ktraits::TileShape_MNK;
|
||||
using ClusterShape = typename Ktraits::ClusterShape_MNK;
|
||||
|
||||
static constexpr int NumMmaThreads = size(typename Ktraits::TiledMmaQK{});
|
||||
static constexpr int NumCopyThreads = cutlass::NumThreadsPerWarpGroup;
|
||||
static constexpr int kBlockM = Ktraits::kBlockM;
|
||||
|
||||
using CollectiveMainloop = CollectiveMainloopFwd<Ktraits, Is_causal>;
|
||||
using CollectiveEpilogue = CollectiveEpilogueFwd<Ktraits>;
|
||||
|
||||
using MainloopPipeline = typename Ktraits::MainloopPipeline;
|
||||
using PipelineParams = typename MainloopPipeline::Params;
|
||||
using PipelineState = typename MainloopPipeline::PipelineState;
|
||||
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
|
||||
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
|
||||
using PipelineStateQ = typename Ktraits::PipelineStateQ;
|
||||
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
|
||||
|
||||
|
||||
enum class WarpGroupRole {
|
||||
Producer = 0,
|
||||
Consumer0 = 1,
|
||||
Consumer1 = 2
|
||||
};
|
||||
enum class ProducerWarpRole {
|
||||
Mainloop = 0,
|
||||
Epilogue = 1,
|
||||
Warp2 = 2,
|
||||
Warp3 = 3
|
||||
};
|
||||
|
||||
extern __shared__ char shared_memory[];
|
||||
auto &shared_storage = *reinterpret_cast<typename Ktraits::SharedStorage*>(shared_memory);
|
||||
|
||||
int const lane_predicate = cute::elect_one_sync();
|
||||
int const warp_idx = cutlass::canonical_warp_idx_sync();
|
||||
int warp_group_idx = cutlass::canonical_warp_group_idx();
|
||||
int const warp_group_thread_idx = threadIdx.x % cutlass::NumThreadsPerWarpGroup;
|
||||
int warp_idx_in_warp_group = warp_idx % cutlass::NumWarpsPerWarpGroup;
|
||||
auto warp_group_role = WarpGroupRole(warp_group_idx);
|
||||
auto producer_warp_role = ProducerWarpRole(warp_idx_in_warp_group);
|
||||
|
||||
// Issue Tma Descriptor Prefetch from a single thread
|
||||
if (warp_idx == 0 && lane_predicate) {
|
||||
CollectiveMainloop::prefetch_tma_descriptors(mainloop_params);
|
||||
CollectiveEpilogue::prefetch_tma_descriptors(epilogue_params);
|
||||
}
|
||||
|
||||
// Obtain warp index
|
||||
|
||||
PipelineParams pipeline_params_v;
|
||||
pipeline_params_v.transaction_bytes = CollectiveMainloop::TmaTransactionBytesV;
|
||||
pipeline_params_v.role = warp_group_role == WarpGroupRole::Producer
|
||||
? MainloopPipeline::ThreadCategory::Producer
|
||||
: MainloopPipeline::ThreadCategory::Consumer;
|
||||
pipeline_params_v.is_leader = warp_group_thread_idx == 0;
|
||||
pipeline_params_v.num_consumers = NumMmaThreads;
|
||||
|
||||
PipelineParams pipeline_params_k;
|
||||
pipeline_params_k.transaction_bytes = CollectiveMainloop::TmaTransactionBytesK;
|
||||
pipeline_params_k.role = warp_group_role == WarpGroupRole::Producer
|
||||
? MainloopPipeline::ThreadCategory::Producer
|
||||
: MainloopPipeline::ThreadCategory::Consumer;
|
||||
pipeline_params_k.is_leader = warp_group_thread_idx == 0;
|
||||
pipeline_params_k.num_consumers = NumMmaThreads;
|
||||
|
||||
PipelineParamsQ pipeline_params_q;
|
||||
pipeline_params_q.transaction_bytes = CollectiveMainloop::TmaTransactionBytesQ;
|
||||
pipeline_params_q.role = warp_group_role == WarpGroupRole::Producer
|
||||
? MainloopPipelineQ::ThreadCategory::Producer
|
||||
: MainloopPipelineQ::ThreadCategory::Consumer;
|
||||
pipeline_params_q.is_leader = warp_group_thread_idx == 0;
|
||||
pipeline_params_q.num_consumers = NumMmaThreads;
|
||||
|
||||
// We're counting on pipeline_k to call cutlass::arch::fence_barrier_init();
|
||||
MainloopPipelineQ pipeline_q(shared_storage.pipeline_q, pipeline_params_q, ClusterShape{});
|
||||
MainloopPipeline pipeline_k(shared_storage.pipeline_k, pipeline_params_k, ClusterShape{});
|
||||
MainloopPipeline pipeline_v(shared_storage.pipeline_v, pipeline_params_v, ClusterShape{});
|
||||
|
||||
uint32_t epilogue_barrier_group_size_list[2] = {cutlass::NumThreadsPerWarp, NumMmaThreads};
|
||||
typename EpilogueBarrier::Params params_epilogue_barrier;
|
||||
params_epilogue_barrier.group_id = (warp_group_role == WarpGroupRole::Producer);
|
||||
params_epilogue_barrier.group_size_list = epilogue_barrier_group_size_list;
|
||||
EpilogueBarrier barrier_o(shared_storage.barrier_o, params_epilogue_barrier);
|
||||
|
||||
CollectiveMainloop collective_mainloop;
|
||||
CollectiveEpilogue collective_epilogue;
|
||||
__syncthreads();
|
||||
|
||||
if (warp_group_role == WarpGroupRole::Producer) {
|
||||
cutlass::arch::warpgroup_reg_dealloc<24>();
|
||||
TileScheduler scheduler;
|
||||
|
||||
if (producer_warp_role == ProducerWarpRole::Mainloop) { // Load Q, K, V
|
||||
PipelineStateQ smem_pipe_write_q = cutlass::make_producer_start_state<MainloopPipelineQ>();
|
||||
PipelineState smem_pipe_write_k = cutlass::make_producer_start_state<MainloopPipeline>();
|
||||
PipelineState smem_pipe_write_v = cutlass::make_producer_start_state<MainloopPipeline>();
|
||||
|
||||
int work_idx = 0;
|
||||
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
|
||||
int tile_count_semaphore = 0;
|
||||
collective_mainloop.load(mainloop_params, scheduler_params,
|
||||
pipeline_q, pipeline_k, pipeline_v,
|
||||
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v,
|
||||
shared_storage, work_tile_info, work_idx, tile_count_semaphore);
|
||||
}
|
||||
collective_mainloop.load_tail(pipeline_q, pipeline_k, pipeline_v,
|
||||
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v);
|
||||
} else if (producer_warp_role == ProducerWarpRole::Epilogue) {
|
||||
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
|
||||
barrier_o.wait();
|
||||
collective_epilogue.tma_store(shared_storage, epilogue_params, work_tile_info, scheduler_params, threadIdx.x);
|
||||
collective_epilogue.store_tail();
|
||||
barrier_o.arrive();
|
||||
}
|
||||
|
||||
}
|
||||
} else if (warp_group_role == WarpGroupRole::Consumer0 || warp_group_role == WarpGroupRole::Consumer1) {
|
||||
cutlass::arch::warpgroup_reg_alloc<232>();
|
||||
typename Ktraits::TiledMmaPV tiled_mma_pv;
|
||||
TileScheduler scheduler{};
|
||||
PipelineState smem_pipe_read_k, smem_pipe_read_v;
|
||||
PipelineStateQ smem_pipe_read_q;
|
||||
|
||||
int work_idx = 0;
|
||||
|
||||
CUTLASS_PRAGMA_NO_UNROLL
|
||||
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
|
||||
// Attention output (GEMM-II) accumulator.
|
||||
Tensor tOrO = partition_fragment_C(tiled_mma_pv, select<0, 2>(TileShape_MNK{}));
|
||||
// flash::Softmax<2 * (2 * kBlockM / NumMmaThreads)> softmax;
|
||||
// Pass single_level_p_quant flag to control P quantization mode
|
||||
flash::SoftmaxFused<2 * (2 * kBlockM / NumMmaThreads)> softmax_fused(params.single_level_p_quant);
|
||||
auto block_coord = work_tile_info.get_block_coord(scheduler_params);
|
||||
auto [m_block, bidh, bidb] = block_coord;
|
||||
|
||||
int n_block_max = collective_mainloop.get_n_block_max(mainloop_params, m_block);
|
||||
if (Is_causal && n_block_max <= 0) { // We exit early and write 0 to gO and -inf to gLSE.
|
||||
collective_epilogue.store_zero(epilogue_params, threadIdx.x - NumCopyThreads, block_coord);
|
||||
continue;
|
||||
}
|
||||
|
||||
collective_mainloop.mma(mainloop_params, pipeline_q, pipeline_k, pipeline_v, smem_pipe_read_q, smem_pipe_read_k, smem_pipe_read_v,
|
||||
tOrO, softmax_fused, n_block_max, threadIdx.x - NumCopyThreads, work_idx, m_block, shared_storage);
|
||||
barrier_o.wait();
|
||||
collective_epilogue.mma_store(shared_storage, tiled_mma_pv, tOrO, threadIdx.x - NumCopyThreads);
|
||||
barrier_o.arrive();
|
||||
++work_idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,114 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include "cutlass/cluster_launch.hpp"
|
||||
|
||||
#include "static_switch.h"
|
||||
#include "params.h"
|
||||
#include "tile_scheduler.h"
|
||||
#include "kernel_ws.h"
|
||||
#include "kernel_traits.h"
|
||||
#include "block_config.h"
|
||||
|
||||
|
||||
template<typename Kernel_traits, bool Is_causal>
|
||||
void run_flash_fwd(Flash_fwd_params ¶ms, cudaStream_t stream) {
|
||||
using Element = typename Kernel_traits::Element;
|
||||
using ElementSF = typename Kernel_traits::ElementSF;
|
||||
using ElementOut = typename Kernel_traits::ElementOut;
|
||||
using TileShape_MNK = typename Kernel_traits::TileShape_MNK;
|
||||
using ClusterShape = typename Kernel_traits::ClusterShape_MNK;
|
||||
using CollectiveMainloop = flash::CollectiveMainloopFwd<Kernel_traits, Is_causal>;
|
||||
using CollectiveEpilogue = flash::CollectiveEpilogueFwd<Kernel_traits>;
|
||||
// using Scheduler = flash::SingleTileScheduler;
|
||||
using Scheduler = flash::StaticPersistentTileScheduler;
|
||||
typename CollectiveMainloop::Params mainloop_params =
|
||||
CollectiveMainloop::to_underlying_arguments({
|
||||
static_cast<Element const*>(params.q_ptr),
|
||||
{params.seqlen_q, params.d, params.h, params.b}, // shape_Q
|
||||
{params.q_row_stride, _1{}, params.q_head_stride, params.q_batch_stride}, // stride_Q
|
||||
static_cast<Element const*>(params.k_ptr),
|
||||
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_K
|
||||
{params.k_row_stride, _1{}, params.k_head_stride, params.k_batch_stride}, // stride_K
|
||||
{params.unpadded_seqlen_k, params.d, params.h_k, params.b}, // shape_K
|
||||
static_cast<Element const*>(params.v_ptr),
|
||||
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_Vt
|
||||
{params.v_row_stride, _1{}, params.v_head_stride, params.v_batch_stride}, // stride_Vt
|
||||
static_cast<ElementSF const*>(params.sfq_ptr),
|
||||
{params.seqlen_q, params.d, params.h, params.b}, // shape_SFQ
|
||||
static_cast<ElementSF const*>(params.sfk_ptr),
|
||||
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_SFK
|
||||
static_cast<ElementSF const*>(params.sfv_ptr),
|
||||
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_SFVt
|
||||
static_cast<float const*>(params.delta_s_ptr),
|
||||
{params.seqlen_s, params.seqlen_k, params.h_k, params.b},
|
||||
{params.ds_row_stride, _1{}, params.ds_head_stride, params.ds_batch_stride},
|
||||
params.scale_softmax_log2
|
||||
});
|
||||
typename CollectiveEpilogue::Params epilogue_params =
|
||||
CollectiveEpilogue::to_underlying_arguments({
|
||||
static_cast<ElementOut*>(params.o_ptr),
|
||||
{params.seqlen_q, params.d, params.h, params.b}, // shape_O
|
||||
{params.o_row_stride, _1{}, params.o_head_stride, params.o_batch_stride}, // stride_O
|
||||
static_cast<float*>(params.softmax_lse_ptr),
|
||||
{_1{}, params.seqlen_q, params.h * params.seqlen_q}, // stride_LSE
|
||||
});
|
||||
|
||||
int num_blocks_m = cutlass::ceil_div(params.seqlen_q, Kernel_traits::kBlockM);
|
||||
num_blocks_m = cutlass::ceil_div(num_blocks_m, size<0>(ClusterShape{})) * size<0>(ClusterShape{});
|
||||
typename Scheduler::Arguments scheduler_args = {num_blocks_m, params.h, params.b};
|
||||
typename Scheduler::Params scheduler_params = Scheduler::to_underlying_arguments(scheduler_args);
|
||||
// Get the ptr to kernel function.
|
||||
void *kernel;
|
||||
kernel = (void *)flash::compute_attn_ws<Kernel_traits, Is_causal, Scheduler>;
|
||||
int smem_size = sizeof(typename Kernel_traits::SharedStorage);
|
||||
if (smem_size >= 48 * 1024) {
|
||||
C10_CUDA_CHECK(cudaFuncSetAttribute(kernel, cudaFuncAttributeMaxDynamicSharedMemorySize, smem_size));
|
||||
}
|
||||
static constexpr int ctaSize = Kernel_traits::kNWarps * 32;
|
||||
params.m_block_divmod = cutlass::FastDivmod(num_blocks_m);
|
||||
params.total_blocks = num_blocks_m * params.h * params.b;
|
||||
dim3 grid_dims = Scheduler::get_grid_dim(scheduler_args, 170);
|
||||
dim3 block_dims(ctaSize);
|
||||
dim3 cluster_dims(size<0>(ClusterShape{}), size<1>(ClusterShape{}), size<2>(ClusterShape{}));
|
||||
cutlass::ClusterLaunchParams launch_params{grid_dims, block_dims, cluster_dims, smem_size, stream};
|
||||
cutlass::launch_kernel_on_cluster(launch_params, kernel, params, mainloop_params, epilogue_params, scheduler_params);
|
||||
|
||||
C10_CUDA_KERNEL_LAUNCH_CHECK();
|
||||
}
|
||||
|
||||
|
||||
template<typename T, int Headdim, typename O = cutlass::bfloat16_t>
|
||||
void run_mha_fwd_(Flash_fwd_params ¶ms, cudaStream_t stream) {
|
||||
BOOL_SWITCH(params.is_causal, Is_causal, [&] {
|
||||
BOOL_SWITCH(params.per_block_mean, per_block, [&] {
|
||||
if constexpr (Headdim == 64 || Headdim == 128) {
|
||||
run_flash_fwd<
|
||||
Flash_fwd_kernel_traits<Headdim, flash::BLOCK_M, flash::BLOCK_N, 3, 1, per_block, T, O>,
|
||||
Is_causal
|
||||
>(params, stream);
|
||||
} else {
|
||||
static_assert(Headdim == 64 || Headdim == 128, "Unsupported Headdim");
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
@@ -1,908 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cutlass/cutlass.h>
|
||||
#include <cutlass/array.h>
|
||||
#include <cutlass/numeric_types.h>
|
||||
#include <cutlass/numeric_conversion.h>
|
||||
#include "cutlass/pipeline/pipeline.hpp"
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include "cutlass/gemm/collective/collective_builder.hpp"
|
||||
|
||||
#include "utils.h"
|
||||
#include "named_barrier.h"
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <typename Ktraits, bool Is_causal>
|
||||
struct CollectiveMainloopFwd {
|
||||
|
||||
using Element = typename Ktraits::Element;
|
||||
using ElementSF = typename Ktraits::ElementSF;
|
||||
// using TMAElement = Element;
|
||||
// using TMAElementSF = typename Ktraits::ElementSF;
|
||||
using TileShape_MNK = typename Ktraits::TileShape_MNK;
|
||||
using ClusterShape = typename Ktraits::ClusterShape_MNK;
|
||||
|
||||
static constexpr int kStages = Ktraits::kStages;
|
||||
static constexpr int kHeadDim = Ktraits::kHeadDim;
|
||||
static constexpr int BlockMean = Ktraits::BlockMean;
|
||||
using GmemTiledCopy = typename Ktraits::GmemTiledCopy;
|
||||
using SmemLayoutQ = typename Ktraits::SmemLayoutQ;
|
||||
using SmemLayoutK = typename Ktraits::SmemLayoutK;
|
||||
using SmemLayoutV = typename Ktraits::SmemLayoutV;
|
||||
using SmemLayoutVt = typename Ktraits::SmemLayoutVt;
|
||||
using SmemLayoutDS = typename Ktraits::SmemLayoutDS;
|
||||
using SmemLayoutAtomDS = typename Ktraits::SmemLayoutAtomDS;
|
||||
using LayoutDS = decltype(
|
||||
blocked_product(
|
||||
SmemLayoutAtomDS{},
|
||||
make_layout(
|
||||
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
|
||||
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))
|
||||
)
|
||||
);
|
||||
using ShapeQKV = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d, head, batch)
|
||||
using StrideQKV = cute::Stride<int64_t, _1, int64_t, int64_t>;
|
||||
using ShapeSF = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d // 16, head, batch)
|
||||
using LayoutSF = typename Ktraits::LayoutSF;
|
||||
using LayoutP = typename Ktraits::LayoutP;
|
||||
using LayoutSFP = typename Ktraits::LayoutSFP;
|
||||
using SfAtom = typename Ktraits::SfAtom;
|
||||
using TMA_Q = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
|
||||
SmemLayoutQ{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{}));
|
||||
|
||||
using TMA_KV = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
|
||||
take<0, 2>(SmemLayoutK{}),
|
||||
select<1, 2>(TileShape_MNK{}),
|
||||
_1{}));
|
||||
|
||||
using TMA_Vt = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
|
||||
take<0, 2>(SmemLayoutVt{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using TMA_DS = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<float const*>(nullptr)), LayoutDS{}),
|
||||
take<0, 2>(SmemLayoutDS{}),
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using BlkScaledConfig = typename Ktraits::BlkScaledConfig;
|
||||
using GmemTiledCopySF = typename Ktraits::GmemTiledCopySF;
|
||||
using SmemLayoutSFQ = typename Ktraits::SmemLayoutSFQ;
|
||||
using SmemLayoutSFK = typename Ktraits::SmemLayoutSFK;
|
||||
using SmemLayoutSFV = typename Ktraits::SmemLayoutSFV;
|
||||
using SmemLayoutSFVt = typename Ktraits::SmemLayoutSFVt;
|
||||
|
||||
using TMA_SFQ = decltype(make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
|
||||
SmemLayoutSFQ{},
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{})); // No programmatic multicast
|
||||
|
||||
|
||||
using TMA_SFKV = decltype(make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
|
||||
SmemLayoutSFK{}(_,_,cute::Int<0>{}),
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using TMA_SFVt = decltype(make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
|
||||
SmemLayoutSFVt{}(_,_,cute::Int<0>{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using SmemCopyAtomQ = typename Ktraits::SmemCopyAtomQ;
|
||||
using SmemCopyAtomKV = typename Ktraits::SmemCopyAtomKV;
|
||||
using SmemCopyAtomSF = typename Ktraits::SmemCopyAtomSF;
|
||||
using TiledMmaQK = typename Ktraits::TiledMmaQK;
|
||||
using TiledMmaPV = typename Ktraits::TiledMmaPV;
|
||||
static constexpr int NumMmaThreads = size(TiledMmaQK{});
|
||||
using MainloopPipeline = typename Ktraits::MainloopPipeline;
|
||||
using PipelineParams = typename MainloopPipeline::Params;
|
||||
using PipelineState = typename MainloopPipeline::PipelineState;
|
||||
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
|
||||
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
|
||||
using PipelineStateQ = typename Ktraits::PipelineStateQ;
|
||||
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
|
||||
|
||||
// Set the bytes transferred in this TMA transaction (may involve multiple issues)
|
||||
static constexpr uint32_t TmaTransactionBytesQ = static_cast<uint32_t>(
|
||||
cutlass::bits_to_bytes(cosize((SmemLayoutSFQ{})) * cute::sizeof_bits_v<ElementSF>) +
|
||||
cutlass::bits_to_bytes(size((SmemLayoutQ{})) * sizeof_bits<Element>::value));
|
||||
|
||||
static constexpr uint32_t TmaTransactionBytesK = static_cast<uint32_t>(
|
||||
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFK{})) * cute::sizeof_bits_v<ElementSF>) +
|
||||
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutDS{})) * cute::sizeof_bits_v<float>) +
|
||||
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutK{})) * sizeof_bits<Element>::value));
|
||||
|
||||
static constexpr uint32_t TmaTransactionBytesV = static_cast<uint32_t>(
|
||||
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFVt{})) * cute::sizeof_bits_v<ElementSF>) +
|
||||
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutVt{})) * sizeof_bits<Element>::value));
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
Element const* ptr_Q;
|
||||
ShapeQKV const shape_Q;
|
||||
StrideQKV const stride_Q;
|
||||
Element const* ptr_K;
|
||||
ShapeQKV const shape_K;
|
||||
StrideQKV const stride_K;
|
||||
ShapeQKV const unpadded_shape_K;
|
||||
Element const* ptr_Vt;
|
||||
ShapeQKV const shape_Vt;
|
||||
StrideQKV const stride_Vt;
|
||||
ElementSF const* ptr_SFQ{nullptr};
|
||||
ShapeSF const shape_SFQ{};
|
||||
ElementSF const* ptr_SFK{nullptr};
|
||||
ShapeSF const shape_SFK{};
|
||||
ElementSF const* ptr_SFVt{nullptr};
|
||||
ShapeSF const shape_SFVt{};
|
||||
float const* ptr_ds;
|
||||
ShapeQKV const shape_ds;
|
||||
StrideQKV const stride_ds;
|
||||
float const softmax_scale_log2;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
ShapeQKV const shape_Q;
|
||||
LayoutSF const layout_SFQ;
|
||||
ShapeQKV const shape_K;
|
||||
ShapeQKV const unpadded_shape_K;
|
||||
LayoutSF const layout_SFK;
|
||||
ShapeQKV const shape_Vt;
|
||||
LayoutSF const layout_SFVt;
|
||||
LayoutDS const layout_DS;
|
||||
TMA_Q tma_load_Q;
|
||||
TMA_SFQ tma_load_SFQ;
|
||||
TMA_KV tma_load_K;
|
||||
TMA_SFKV tma_load_SFK;
|
||||
TMA_Vt tma_load_Vt;
|
||||
TMA_SFVt tma_load_SFVt;
|
||||
TMA_DS tma_load_DS;
|
||||
float const softmax_scale_log2;
|
||||
};
|
||||
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
Tensor mQ = make_tensor(make_gmem_ptr(args.ptr_Q), args.shape_Q, args.stride_Q);
|
||||
TMA_Q tma_load_Q = make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
mQ,
|
||||
SmemLayoutQ{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{}); // no mcast for Q
|
||||
Tensor mK = make_tensor(make_gmem_ptr(args.ptr_K), args.shape_K, args.stride_K);
|
||||
TMA_KV tma_load_K = make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
mK,
|
||||
SmemLayoutK{}(_, _, _0{}),
|
||||
select<1, 2>(TileShape_MNK{}),
|
||||
_1{}); // mcast along M mode for this N load, if any
|
||||
Tensor mVt = make_tensor(make_gmem_ptr(args.ptr_Vt), args.shape_Vt, args.stride_Vt);
|
||||
TMA_Vt tma_load_Vt = make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
mVt,
|
||||
SmemLayoutVt{}(_, _, _0{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}); // mcast along M mode for this N load, if any
|
||||
auto [Seqlen_Q, Seqlen_K, HeadNum, Batch] = args.shape_ds;
|
||||
LayoutDS layout_ds = tile_to_shape(SmemLayoutAtomDS{}, make_shape(Seqlen_Q, Seqlen_K, HeadNum, Batch), Step<_2,_1,_3,_4>{});
|
||||
Tensor mDS = make_tensor(make_gmem_ptr(args.ptr_ds), layout_ds);
|
||||
TMA_DS tma_load_ds = make_tma_copy (
|
||||
GmemTiledCopy{},
|
||||
mDS,
|
||||
SmemLayoutDS{}(_, _, _0{}),
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{});
|
||||
LayoutSF layout_sfq = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFQ);
|
||||
Tensor mSFQ = make_tensor(make_gmem_ptr(args.ptr_SFQ), layout_sfq);
|
||||
TMA_SFQ tma_load_sfq = make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
mSFQ,
|
||||
SmemLayoutSFQ{},
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{});
|
||||
LayoutSF layout_sfk = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFK);
|
||||
Tensor mSFK = make_tensor(make_gmem_ptr(args.ptr_SFK), layout_sfk);
|
||||
TMA_SFKV tma_load_sfk = make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
mSFK,
|
||||
SmemLayoutSFK{}(_, _, _0{}),
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{});
|
||||
LayoutSF layout_sfvt = BlkScaledConfig::tile_atom_to_shape_SFVt(args.shape_SFVt);
|
||||
Tensor mSFVt = make_tensor(make_gmem_ptr(args.ptr_SFVt), layout_sfvt);
|
||||
TMA_SFVt tma_load_sfvt = make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
mSFVt,
|
||||
SmemLayoutSFVt{}(_, _, _0{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{});
|
||||
return {args.shape_Q, layout_sfq,
|
||||
args.shape_K, args.unpadded_shape_K, layout_sfk,
|
||||
args.shape_Vt, layout_sfvt,
|
||||
layout_ds,
|
||||
tma_load_Q, tma_load_sfq,
|
||||
tma_load_K, tma_load_sfk,
|
||||
tma_load_Vt, tma_load_sfvt,
|
||||
tma_load_ds,
|
||||
args.softmax_scale_log2};
|
||||
}
|
||||
|
||||
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
|
||||
CUTLASS_DEVICE
|
||||
static void prefetch_tma_descriptors(Params const& mainloop_params) {
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Q.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_K.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Vt.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFQ.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFK.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFVt.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_DS.get_tma_descriptor());
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
int get_n_block_max(Params const& mainloop_params, int m_block) {
|
||||
static constexpr int kBlockM = get<0>(TileShape_MNK{});
|
||||
static constexpr int kBlockN = get<1>(TileShape_MNK{});
|
||||
int const seqlen_q = get<0>(mainloop_params.shape_Q);
|
||||
int const seqlen_k = get<0>(mainloop_params.shape_K);
|
||||
int n_block_max = cute::ceil_div(seqlen_k, kBlockN);
|
||||
if constexpr (Is_causal) {
|
||||
n_block_max = std::min(n_block_max,
|
||||
cute::ceil_div((m_block + 1) * kBlockM + seqlen_k - seqlen_q, kBlockN));
|
||||
}
|
||||
return n_block_max;
|
||||
}
|
||||
|
||||
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<0>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<1>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
|
||||
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<1>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<2>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFATensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma)
|
||||
{
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
auto partition_SFA = thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
return make_fragment_like<ValTypeSF>(partition_SFA);
|
||||
}
|
||||
|
||||
template <class SFBTensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma)
|
||||
{
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
auto partition_SFB = thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
return make_fragment_like<ValTypeSF>(partition_SFB);
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFA_TV(TiledMma& mma)
|
||||
{
|
||||
// (M,K) -> (M,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto atile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<1>{} , Int<0>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFB_TV(TiledMma& mma)
|
||||
{
|
||||
// (N,K) -> (N,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto btile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<0>{} , Int<1>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
template <typename SchedulerParams, typename SharedStorage, typename WorkTileInfo>
|
||||
CUTLASS_DEVICE void
|
||||
load(Params const& mainloop_params,
|
||||
SchedulerParams const& scheduler_params,
|
||||
MainloopPipelineQ pipeline_q,
|
||||
MainloopPipeline pipeline_k,
|
||||
MainloopPipeline pipeline_v,
|
||||
PipelineStateQ& smem_pipe_write_q,
|
||||
PipelineState& smem_pipe_write_k,
|
||||
PipelineState& smem_pipe_write_v,
|
||||
SharedStorage &shared_storage,
|
||||
WorkTileInfo work_tile_info,
|
||||
int& work_idx,
|
||||
int& tile_count_semaphore
|
||||
) {
|
||||
|
||||
static constexpr int kBlockM = get<0>(TileShape_MNK{});
|
||||
static constexpr int kBlockN = get<1>(TileShape_MNK{});
|
||||
|
||||
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
|
||||
|
||||
int n_block_max = get_n_block_max(mainloop_params, m_block);
|
||||
|
||||
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
|
||||
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
|
||||
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
|
||||
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
|
||||
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
|
||||
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
|
||||
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
|
||||
|
||||
Tensor mQ = mainloop_params.tma_load_Q.get_tma_tensor(mainloop_params.shape_Q);
|
||||
Tensor mK = mainloop_params.tma_load_K.get_tma_tensor(mainloop_params.shape_K);
|
||||
Tensor mVt = mainloop_params.tma_load_Vt.get_tma_tensor(mainloop_params.shape_Vt);
|
||||
Tensor mDS = mainloop_params.tma_load_DS.get_tma_tensor(shape(mainloop_params.layout_DS));
|
||||
Tensor mSFQ = mainloop_params.tma_load_SFQ.get_tma_tensor(shape(mainloop_params.layout_SFQ));
|
||||
Tensor mSFK = mainloop_params.tma_load_SFK.get_tma_tensor(shape(mainloop_params.layout_SFK));
|
||||
Tensor mSFVt = mainloop_params.tma_load_SFVt.get_tma_tensor(shape(mainloop_params.layout_SFVt));
|
||||
uint32_t block_rank_in_cluster = cute::block_rank_in_cluster();
|
||||
constexpr uint32_t cluster_shape_x = get<0>(ClusterShape());
|
||||
uint2 cluster_local_block_id = {block_rank_in_cluster % cluster_shape_x, block_rank_in_cluster / cluster_shape_x};
|
||||
Tensor gQ = local_tile(mQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
|
||||
Tensor gK = local_tile(mK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{})); // (N, K, _)
|
||||
Tensor gVt = local_tile(mVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _)); // (N, K, _)
|
||||
Tensor gDS = [&] {
|
||||
if constexpr (BlockMean) {
|
||||
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(m_block, _));
|
||||
} else {
|
||||
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(_0{}, _));
|
||||
}
|
||||
}();
|
||||
Tensor gSFQ = local_tile(mSFQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{}));
|
||||
Tensor gSFK = local_tile(mSFK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{}));
|
||||
Tensor gSFVt = local_tile(mSFVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _));
|
||||
auto block_tma_q = mainloop_params.tma_load_Q.get_slice(_0{});
|
||||
Tensor tQgQ = block_tma_q.partition_S(gQ);
|
||||
Tensor tQsQ = block_tma_q.partition_D(sQ);
|
||||
auto block_tma_sfq = mainloop_params.tma_load_SFQ.get_slice(_0{});
|
||||
Tensor tQgSFQ = block_tma_sfq.partition_S(gSFQ);
|
||||
Tensor tQsSFQ = block_tma_sfq.partition_D(sSFQ);
|
||||
auto block_tma_k = mainloop_params.tma_load_K.get_slice(cluster_local_block_id.x);
|
||||
Tensor tKgK = group_modes<0, 3>(block_tma_k.partition_S(gK));
|
||||
Tensor tKsK = group_modes<0, 3>(block_tma_k.partition_D(sK));
|
||||
auto block_tma_sfk = mainloop_params.tma_load_SFK.get_slice(cluster_local_block_id.x);
|
||||
Tensor tKgSFK = group_modes<0, 3>(block_tma_sfk.partition_S(gSFK));
|
||||
Tensor tKsSFK = group_modes<0, 3>(block_tma_sfk.partition_D(sSFK));
|
||||
auto block_tma_vt = mainloop_params.tma_load_Vt.get_slice(cluster_local_block_id.x);
|
||||
Tensor tVgVt = group_modes<0, 3>(block_tma_vt.partition_S(gVt));
|
||||
Tensor tVsVt = group_modes<0, 3>(block_tma_vt.partition_D(sVt));
|
||||
auto block_tma_sfvt = mainloop_params.tma_load_SFVt.get_slice(cluster_local_block_id.x);
|
||||
Tensor tVgSFVt = group_modes<0, 3>(block_tma_sfvt.partition_S(gSFVt));
|
||||
Tensor tVsSFVt = group_modes<0, 3>(block_tma_sfvt.partition_D(sSFVt));
|
||||
auto block_tma_ds = mainloop_params.tma_load_DS.get_slice(cluster_local_block_id.x);
|
||||
Tensor tDSgDS = group_modes<0, 3>(block_tma_ds.partition_S(gDS));
|
||||
Tensor tDSsDS = group_modes<0, 3>(block_tma_ds.partition_D(sDS));
|
||||
uint16_t mcast_mask_kv = 0;
|
||||
|
||||
int n_block = n_block_max - 1;
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
if (lane_predicate) {
|
||||
pipeline_q.producer_acquire(smem_pipe_write_q);
|
||||
copy(mainloop_params.tma_load_Q.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgQ, tQsQ);
|
||||
copy(mainloop_params.tma_load_SFQ.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgSFQ, tQsSFQ);
|
||||
++smem_pipe_write_q;
|
||||
pipeline_k.producer_acquire(smem_pipe_write_k);
|
||||
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
|
||||
++smem_pipe_write_k;
|
||||
pipeline_v.producer_acquire(smem_pipe_write_v);
|
||||
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
|
||||
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
|
||||
++smem_pipe_write_v;
|
||||
}
|
||||
|
||||
n_block--;
|
||||
if (lane_predicate) {
|
||||
// CUTLASS_PRAGMA_NO_UNROLL
|
||||
#pragma unroll 2
|
||||
for (; n_block >= 0; --n_block) {
|
||||
pipeline_k.producer_acquire(smem_pipe_write_k);
|
||||
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
|
||||
++smem_pipe_write_k;
|
||||
pipeline_v.producer_acquire(smem_pipe_write_v);
|
||||
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
|
||||
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
|
||||
++smem_pipe_write_v;
|
||||
}
|
||||
}
|
||||
++work_idx;
|
||||
}
|
||||
|
||||
/// Perform a Producer Epilogue to prevent early exit of blocks in a Cluster
|
||||
CUTLASS_DEVICE void
|
||||
load_tail(MainloopPipelineQ pipeline_q,
|
||||
MainloopPipeline pipeline_k,
|
||||
MainloopPipeline pipeline_v,
|
||||
PipelineStateQ& smem_pipe_write_q,
|
||||
PipelineState& smem_pipe_write_k,
|
||||
PipelineState& smem_pipe_write_v) {
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
// Issue the epilogue waits
|
||||
if (lane_predicate) {
|
||||
pipeline_q.producer_tail(smem_pipe_write_q);
|
||||
pipeline_k.producer_tail(smem_pipe_write_k);
|
||||
pipeline_v.producer_tail(smem_pipe_write_v);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename SharedStorage, typename FrgTensorO, typename SoftmaxFused>
|
||||
CUTLASS_DEVICE void
|
||||
mma(Params const& mainloop_params,
|
||||
MainloopPipelineQ pipeline_q,
|
||||
MainloopPipeline pipeline_k,
|
||||
MainloopPipeline pipeline_v,
|
||||
PipelineStateQ& smem_pipe_read_q,
|
||||
PipelineState& smem_pipe_read_k,
|
||||
PipelineState& smem_pipe_read_v,
|
||||
FrgTensorO& tOrO_store,
|
||||
SoftmaxFused& softmax_fused,
|
||||
int n_block_count,
|
||||
int thread_idx,
|
||||
int work_idx,
|
||||
int m_block,
|
||||
SharedStorage& shared_storage
|
||||
) {
|
||||
|
||||
static_assert(is_rmem<FrgTensorO>::value, "O tensor must be rmem resident.");
|
||||
|
||||
static constexpr int kBlockM = get<0>(TileShape_MNK{});
|
||||
static constexpr int kBlockN = get<1>(TileShape_MNK{});
|
||||
static constexpr int kBlockK = get<2>(TileShape_MNK{});
|
||||
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
|
||||
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
|
||||
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
|
||||
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
|
||||
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
|
||||
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
|
||||
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
|
||||
|
||||
Tensor cQ = make_identity_tensor(make_shape(size<0>(sQ), size<1>(sQ)));
|
||||
Tensor cKV = make_identity_tensor(make_shape(size<0>(sK), size<1>(sK)));
|
||||
TiledMmaQK tiled_mma_qk;
|
||||
TiledMmaPV tiled_mma_pv;
|
||||
auto thread_mma_qk = tiled_mma_qk.get_thread_slice(thread_idx);
|
||||
auto thread_mma_pv = tiled_mma_pv.get_thread_slice(thread_idx);
|
||||
|
||||
Tensor tSrQ = thread_mma_qk.partition_fragment_A(sQ);
|
||||
Tensor tSrK = thread_mma_qk.partition_fragment_B(sK(_,_,Int<0>{}));
|
||||
Tensor tOrVt = thread_mma_pv.partition_fragment_B(sVt(_,_,Int<0>{}));
|
||||
Tensor tOrP = make_tensor_like<Element>(LayoutP{});
|
||||
Tensor tSrSFQ = partition_fragment_SFA(sSFQ, thread_mma_qk);
|
||||
Tensor tSrSFK = partition_fragment_SFB(sSFK(_,_,Int<0>{}), thread_mma_qk);
|
||||
Tensor tOrSFVt = partition_fragment_SFB(sSFVt(_,_,Int<0>{}), thread_mma_pv);
|
||||
Tensor tOrSFP = make_tensor<ElementSF>(LayoutSFP{});
|
||||
Tensor tOrSFP_flt = filter_zeros(tOrSFP);
|
||||
Tensor tSrDS = make_tensor<float>(make_shape(_8{}, _4{}), make_stride(_1{}, _8{}));
|
||||
// copy qk and sf from smem to rmem
|
||||
auto smem_tiled_copy_Q = make_tiled_copy_A(SmemCopyAtomQ{}, tiled_mma_qk);
|
||||
auto smem_thr_copy_Q = smem_tiled_copy_Q.get_thread_slice(thread_idx);
|
||||
Tensor tSsQ = smem_thr_copy_Q.partition_S(as_position_independent_swizzle_tensor(sQ));
|
||||
Tensor tSrQ_copy_view = smem_thr_copy_Q.retile_D(tSrQ);
|
||||
|
||||
auto smem_tiled_copy_K = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_qk);
|
||||
auto smem_thr_copy_K = smem_tiled_copy_K.get_thread_slice(thread_idx);
|
||||
Tensor tSsK = smem_thr_copy_K.partition_S(as_position_independent_swizzle_tensor(sK));
|
||||
Tensor tSrK_copy_view = smem_thr_copy_K.retile_D(tSrK);
|
||||
|
||||
auto smem_tiled_copy_V = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_pv);
|
||||
auto smem_thr_copy_V = smem_tiled_copy_V.get_thread_slice(thread_idx);
|
||||
Tensor tOsVt = smem_thr_copy_V.partition_S(as_position_independent_swizzle_tensor(sVt));
|
||||
Tensor tOrVt_copy_view = smem_thr_copy_V.retile_D(tOrVt);
|
||||
|
||||
auto tile_shape_mnk = tile_shape(tiled_mma_qk);
|
||||
auto smem_tiled_copy_SFQ = make_tiled_copy_impl(SmemCopyAtomSF{},
|
||||
get_layoutSFA_TV(tiled_mma_qk),
|
||||
make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk))
|
||||
);
|
||||
auto smem_thr_copy_SFQ = smem_tiled_copy_SFQ.get_thread_slice(thread_idx);
|
||||
Tensor tSsSFQ = smem_thr_copy_SFQ.partition_S(as_position_independent_swizzle_tensor(sSFQ));
|
||||
Tensor tSrSFQ_copy_view = smem_thr_copy_SFQ.retile_D(tSrSFQ);
|
||||
|
||||
auto smem_tiled_copy_SFK = make_tiled_copy_impl(SmemCopyAtomSF{},
|
||||
get_layoutSFB_TV(tiled_mma_qk),
|
||||
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
|
||||
);
|
||||
auto smem_thr_copy_SFK = smem_tiled_copy_SFK.get_thread_slice(thread_idx);
|
||||
Tensor tSsSFK = smem_thr_copy_SFK.partition_S(as_position_independent_swizzle_tensor(sSFK));
|
||||
Tensor tSrSFK_copy_view = smem_thr_copy_SFK.retile_D(tSrSFK);
|
||||
|
||||
auto smem_tiled_copy_SFV = make_tiled_copy_impl(SmemCopyAtomSF{},
|
||||
get_layoutSFB_TV(tiled_mma_pv),
|
||||
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
|
||||
);
|
||||
auto smem_thr_copy_SFV = smem_tiled_copy_SFV.get_thread_slice(thread_idx);
|
||||
Tensor tOsSFVt = smem_thr_copy_SFV.partition_S(as_position_independent_swizzle_tensor(sSFVt));
|
||||
Tensor tOrSFVt_copy_view = smem_thr_copy_SFV.retile_D(tOrSFVt);
|
||||
|
||||
auto consumer_wait = [](auto& pipeline, auto& smem_pipe_read) {
|
||||
auto barrier_token = pipeline.consumer_try_wait(smem_pipe_read);
|
||||
pipeline.consumer_wait(smem_pipe_read, barrier_token);
|
||||
};
|
||||
|
||||
int const seqlen_q = get<0>(mainloop_params.shape_Q);
|
||||
int const seqlen_k = get<0>(mainloop_params.shape_K);
|
||||
int const unpadded_seqlen_k = get<0>(mainloop_params.unpadded_shape_K);
|
||||
int n_block = n_block_count - 1;
|
||||
|
||||
auto copy_k_block = [&](auto block_id) {
|
||||
auto tSsK_stage = tSsK(_, _, _, smem_pipe_read_k.index());
|
||||
auto tSsSFK_stage = tSsSFK(_, _, _, smem_pipe_read_k.index());
|
||||
copy(smem_tiled_copy_K, tSsK_stage(_, _, block_id), tSrK_copy_view(_, _, block_id));
|
||||
copy(smem_tiled_copy_SFK, tSsSFK_stage(_, _, block_id), tSrSFK_copy_view(_, _, block_id));
|
||||
};
|
||||
|
||||
auto copy_v_block = [&](auto block_id) {
|
||||
auto tOsVt_stage = tOsVt(_, _, _, smem_pipe_read_v.index());
|
||||
auto tOsSFVt_stage = tOsSFVt(_, _, _, smem_pipe_read_v.index());
|
||||
copy(smem_tiled_copy_V, tOsVt_stage(_, _, block_id), tOrVt_copy_view(_, _, block_id));
|
||||
copy(smem_tiled_copy_SFV, tOsSFVt_stage(_, _, block_id), tOrSFVt_copy_view(_, _, block_id));
|
||||
};
|
||||
// auto gemm_qk = [&](auto block_id) {
|
||||
// cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, block_id), tSrSFQ(_, _, block_id)), make_zip_tensor(tSrK(_, _, block_id), tSrSFK(_, _, block_id)), tSrS);
|
||||
// };
|
||||
// auto gemm_pv = [&](auto block_id) {
|
||||
// cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, block_id), tOrSFP(_, _, block_id)), make_zip_tensor(tOrVt(_, _, block_id), tOrSFVt(_, _, block_id)), tOrO);
|
||||
// };
|
||||
auto add_delta_s = [&](auto& acc) {
|
||||
auto tSsDS_stage = recast<float4>(sDS(_, _, smem_pipe_read_k.index()));
|
||||
auto acc_float4 = recast<float4>(acc);
|
||||
int quad_id = (threadIdx.x % 4) * 2;
|
||||
for (int i = 0; i < 4; i++) {
|
||||
auto num = quad_id + i * 8;
|
||||
float4 delta_s_0 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num, _0{}));
|
||||
float4 delta_s_1 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num + 1, _0{}));
|
||||
acc_float4(make_coord(make_coord(_0{}, _0{}), _0{}), _0{}, i) = delta_s_0;
|
||||
acc_float4(make_coord(make_coord(_0{}, _0{}), _1{}), _0{}, i) = delta_s_0;
|
||||
acc_float4(make_coord(make_coord(_0{}, _1{}), _0{}), _0{}, i) = delta_s_1;
|
||||
acc_float4(make_coord(make_coord(_0{}, _1{}), _1{}), _0{}, i) = delta_s_1;
|
||||
}
|
||||
};
|
||||
consumer_wait(pipeline_q, smem_pipe_read_q);
|
||||
copy(smem_tiled_copy_Q, tSsQ, tSrQ_copy_view);
|
||||
copy(smem_tiled_copy_SFQ, tSsSFQ, tSrSFQ_copy_view);
|
||||
pipeline_q.consumer_release(smem_pipe_read_q);
|
||||
++smem_pipe_read_q;
|
||||
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
Tensor AbsMaxP = make_tensor_like<float>(
|
||||
make_layout(shape(group<1, 4>(flatten(tSrS_converion_view.layout()(make_coord(_0{}, _), _, _)))))
|
||||
);
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
copy_k_block(_0{});
|
||||
add_delta_s(tSrS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
|
||||
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
|
||||
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
|
||||
if (k_block < size<2>(tSrQ) - 1) {
|
||||
copy_k_block(k_block + 1);
|
||||
} else {
|
||||
pipeline_k.consumer_release(smem_pipe_read_k);
|
||||
++smem_pipe_read_k;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
auto col_limit_causal = [&](int row, int n_block) {
|
||||
return row + 1 + seqlen_k - n_block * kBlockN - seqlen_q + m_block * kBlockM;
|
||||
};
|
||||
{
|
||||
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tScS = thread_mma_qk.partition_C(cS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size(tSrS); ++i) {
|
||||
if constexpr (!Is_causal) { // Just masking based on col
|
||||
if (int(get<1>(tScS(i))) >= int(unpadded_seqlen_k - n_block * kBlockN)) { tSrS(i) = -INFINITY; }
|
||||
} else {
|
||||
if (int(get<1>(tScS(i))) >= std::min(seqlen_k - n_block * kBlockN,
|
||||
col_limit_causal(int(get<0>(tScS(i))), n_block))) {
|
||||
tSrS(i) = -INFINITY;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
auto quantize = [&](auto mma_k, auto acc_conversion_view) {
|
||||
Tensor AbsMaxP_stagek = AbsMaxP(_, make_coord(_, _, mma_k));
|
||||
Tensor acc_conversion_stagek = acc_conversion_view(_, _, mma_k);
|
||||
Tensor SFP = make_tensor_like<cutlass::float_ue4m3_t>(AbsMaxP_stagek.layout());
|
||||
Tensor SFP_uint32_view = recast<uint32_t>(SFP);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size(AbsMaxP_stagek); i += 4) {
|
||||
uint32_t& tmp = SFP_uint32_view(i / 4);
|
||||
flash::packed_float_to_ue4m3(
|
||||
AbsMaxP_stagek(i),
|
||||
AbsMaxP_stagek(i + 1),
|
||||
AbsMaxP_stagek(i + 2),
|
||||
AbsMaxP_stagek(i + 3),
|
||||
tmp
|
||||
);
|
||||
}
|
||||
int const quad_id = threadIdx.x & 3;
|
||||
uint32_t MASK = (0xFF00FF) << ((quad_id & 1) * 8);
|
||||
Tensor tOrSFP_uint32_view = recast<uint32_t>(tOrSFP(_, _, mma_k));
|
||||
Tensor tOrP_uint32_view = recast<uint32_t>(tOrP(_, _, mma_k));
|
||||
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mma_m = 0; mma_m < size<1>(tOrP); ++mma_m) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
flash::packed_float_to_e2m1(
|
||||
acc_conversion_stagek(make_coord(_0{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_1{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_2{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_3{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_4{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_5{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_6{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_7{}, i), mma_m),
|
||||
tOrP_uint32_view(i, mma_m)
|
||||
);
|
||||
}
|
||||
uint32_t local_sfp = SFP_uint32_view(_0{}, _0{}, mma_m);
|
||||
uint32_t peer_sfp = __shfl_xor_sync(int32_t(-1), local_sfp, 2);
|
||||
if ((quad_id & 1) == 0) {
|
||||
uint32_t sfp = (local_sfp & MASK) | ((peer_sfp & MASK) << 8);
|
||||
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
|
||||
} else {
|
||||
uint32_t sfp = (peer_sfp & MASK) | ((local_sfp & MASK) >> 8);
|
||||
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
softmax_fused.template online_softmax_with_quant</*Is_first=*/true>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
|
||||
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
copy_v_block(_0{});
|
||||
quantize(_0{}, tSrS_converion_view);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO_store);
|
||||
if (v_block < size<2>(tOrP) - 1) {
|
||||
copy_v_block(v_block + 1);
|
||||
quantize(v_block + 1, tSrS_converion_view);
|
||||
} else {
|
||||
pipeline_v.consumer_release(smem_pipe_read_v);
|
||||
++smem_pipe_read_v;
|
||||
}
|
||||
}
|
||||
|
||||
n_block--;
|
||||
constexpr int n_masking_steps = !Is_causal ? 1 : cute::ceil_div(kBlockM, kBlockN) + 1;
|
||||
// // Only go through these if Is_causal, since n_masking_steps = 1 when !Is_causal
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int masking_step = 0; masking_step < n_masking_steps - 1 && n_block >= 0; ++masking_step, --n_block) {
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
copy_k_block(_0{});
|
||||
add_delta_s(tSrS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
|
||||
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
|
||||
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
|
||||
if (k_block < size<2>(tSrQ) - 1) {
|
||||
copy_k_block(k_block + 1);
|
||||
}
|
||||
}
|
||||
pipeline_k.consumer_release(smem_pipe_read_k); // release K
|
||||
++smem_pipe_read_k;
|
||||
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tScS = thread_mma_qk.partition_C(cS);
|
||||
#pragma unroll
|
||||
for (int i = 0; i < size(tSrS); ++i) {
|
||||
if (int(get<1>(tScS(i))) >= col_limit_causal(int(get<0>(tScS(i))), n_block - 1)) {
|
||||
tSrS(i) = -INFINITY;
|
||||
}
|
||||
}
|
||||
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
|
||||
Tensor tOrO = make_fragment_like(tOrO_store);
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
copy_v_block(_0{});
|
||||
quantize(_0{}, tSrS_converion_view);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
|
||||
if (v_block < size<2>(tOrP) - 1) {
|
||||
copy_v_block(v_block + 1);
|
||||
quantize(v_block + 1, tSrS_converion_view);
|
||||
}
|
||||
}
|
||||
pipeline_v.consumer_release(smem_pipe_read_v);
|
||||
++smem_pipe_read_v;
|
||||
if (masking_step > 0) { softmax_fused.rescale_o(tOrO_store, tOrO); }
|
||||
}
|
||||
|
||||
#pragma unroll 1
|
||||
for (; n_block >= 0; --n_block) {
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
copy_k_block(_0{});
|
||||
add_delta_s(tSrS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
|
||||
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
|
||||
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
|
||||
if (k_block < size<2>(tSrQ) - 1) {
|
||||
copy_k_block(k_block + 1);
|
||||
} else {
|
||||
pipeline_k.consumer_release(smem_pipe_read_k);
|
||||
++smem_pipe_read_k;
|
||||
}
|
||||
}
|
||||
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
|
||||
Tensor tOrO = make_fragment_like(tOrO_store);
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
copy_v_block(_0{});
|
||||
quantize(_0{}, tSrS_converion_view);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
|
||||
if (v_block < size<2>(tOrP) - 1) {
|
||||
copy_v_block(v_block + 1);
|
||||
quantize(v_block + 1, tSrS_converion_view);
|
||||
} else {
|
||||
pipeline_v.consumer_release(smem_pipe_read_v);
|
||||
++smem_pipe_read_v;
|
||||
}
|
||||
}
|
||||
softmax_fused.rescale_o(tOrO_store, tOrO);
|
||||
}
|
||||
softmax_fused.finalize(tOrO_store);
|
||||
return;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
} // namespace flash
|
||||
|
||||
@@ -1,119 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cutlass/pipeline/sm90_pipeline.hpp"
|
||||
|
||||
namespace flash {
|
||||
|
||||
enum class FP4NamedBarriers {
|
||||
QueryEmpty = 1,
|
||||
WarpSpecializedConsumer = 2,
|
||||
WarpSpecializedPingPongConsumer1 = 3,
|
||||
WarpSpecializedPingPongConsumer2 = 4,
|
||||
ProducerEnd = 5,
|
||||
ConsumerEnd = 6,
|
||||
EpilogueBarrier = 7
|
||||
};
|
||||
|
||||
template<int SequenceDepth, int SequenceLength>
|
||||
struct OrderedSequenceBarrierVarGroupSizeSharedStorage {
|
||||
using Barrier = cutlass::arch::ClusterBarrier;
|
||||
Barrier barrier_[SequenceDepth][SequenceLength];
|
||||
};
|
||||
|
||||
template<int SequenceDepth_, int SequenceLength_>
|
||||
class OrderedSequenceBarrierVarGroupSize {
|
||||
public:
|
||||
static constexpr int SequenceDepth = SequenceDepth_;
|
||||
static constexpr int SequenceLength = SequenceLength_;
|
||||
using Barrier = cutlass::arch::ClusterBarrier;
|
||||
using SharedStorage = flash::OrderedSequenceBarrierVarGroupSizeSharedStorage<SequenceDepth, SequenceLength>;
|
||||
|
||||
|
||||
struct Params {
|
||||
uint32_t group_id;
|
||||
uint32_t* group_size_list;
|
||||
};
|
||||
|
||||
private :
|
||||
// In future this Params object can be replaced easily with a CG object
|
||||
Params params_;
|
||||
Barrier *barrier_ptr_;
|
||||
cutlass::PipelineState<SequenceDepth> stage_;
|
||||
|
||||
static constexpr int Depth = SequenceDepth;
|
||||
static constexpr int Length = SequenceLength;
|
||||
|
||||
public:
|
||||
OrderedSequenceBarrierVarGroupSize() = delete;
|
||||
OrderedSequenceBarrierVarGroupSize(const OrderedSequenceBarrierVarGroupSize&) = delete;
|
||||
OrderedSequenceBarrierVarGroupSize(OrderedSequenceBarrierVarGroupSize&&) = delete;
|
||||
OrderedSequenceBarrierVarGroupSize& operator=(const OrderedSequenceBarrierVarGroupSize&) = delete;
|
||||
OrderedSequenceBarrierVarGroupSize& operator=(OrderedSequenceBarrierVarGroupSize&&) = delete;
|
||||
~OrderedSequenceBarrierVarGroupSize() = default;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
OrderedSequenceBarrierVarGroupSize(SharedStorage& storage, Params const& params) :
|
||||
params_(params),
|
||||
barrier_ptr_(&storage.barrier_[0][0]),
|
||||
// Group 0 - starts with an opposite phase
|
||||
stage_({0, params.group_id == 0, 0}) {
|
||||
int warp_idx = cutlass::canonical_warp_idx_sync();
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
|
||||
// Barrier FULL, EMPTY init
|
||||
// Init is done only by the one elected thread of the block
|
||||
if (warp_idx == 0 && lane_predicate) {
|
||||
for (int d = 0; d < Depth; ++d) {
|
||||
for (int l = 0; l < Length; ++l) {
|
||||
barrier_ptr_[d * Length + l].init(*(params.group_size_list + l));
|
||||
}
|
||||
}
|
||||
}
|
||||
cutlass::arch::fence_barrier_init();
|
||||
}
|
||||
|
||||
// Wait on a stage to be unlocked
|
||||
CUTLASS_DEVICE
|
||||
void wait() {
|
||||
get_barrier_for_current_stage(params_.group_id).wait(stage_.phase());
|
||||
}
|
||||
|
||||
// Signal completion of Stage and move to the next stage
|
||||
// (group_id) signals to (group_id+1)
|
||||
CUTLASS_DEVICE
|
||||
void arrive() {
|
||||
int signalling_id = (params_.group_id + 1) % Length;
|
||||
get_barrier_for_current_stage(signalling_id).arrive();
|
||||
++stage_;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
void advance() {
|
||||
++stage_;
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
CUTLASS_DEVICE
|
||||
Barrier& get_barrier_for_current_stage(int group_id) {
|
||||
return barrier_ptr_[stage_.index() * Length + group_id];
|
||||
}
|
||||
};
|
||||
|
||||
} // flash
|
||||
@@ -1,180 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cuda.h>
|
||||
#include <vector>
|
||||
|
||||
#ifdef OLD_GENERATOR_PATH
|
||||
#include <ATen/CUDAGeneratorImpl.h>
|
||||
#else
|
||||
#include <ATen/cuda/CUDAGeneratorImpl.h>
|
||||
#endif
|
||||
|
||||
#include <ATen/cuda/CUDAGraphsUtils.cuh> // For at::cuda::philox::unpack
|
||||
|
||||
#include "cutlass/fast_math.h" // For cutlass::FastDivmod
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
struct Qkv_params {
|
||||
using index_t = int64_t;
|
||||
// The QKV matrices.
|
||||
void *__restrict__ q_ptr;
|
||||
void *__restrict__ k_ptr;
|
||||
void *__restrict__ v_ptr;
|
||||
void *__restrict__ delta_s_ptr;
|
||||
// The QKV scale factor matrices.
|
||||
void *__restrict__ sfq_ptr;
|
||||
void *__restrict__ sfk_ptr;
|
||||
void *__restrict__ sfv_ptr;
|
||||
// The stride between rows of the Q, K and V matrices.
|
||||
index_t q_batch_stride;
|
||||
index_t k_batch_stride;
|
||||
index_t v_batch_stride;
|
||||
index_t q_row_stride;
|
||||
index_t k_row_stride;
|
||||
index_t v_row_stride;
|
||||
index_t q_head_stride;
|
||||
index_t k_head_stride;
|
||||
index_t v_head_stride;
|
||||
index_t ds_batch_stride;
|
||||
index_t ds_row_stride;
|
||||
index_t ds_head_stride;
|
||||
// The stride of the Q, K and V scale factor matrices.
|
||||
index_t sfq_batch_stride;
|
||||
index_t sfk_batch_stride;
|
||||
index_t sfv_batch_stride;
|
||||
index_t sfq_row_stride;
|
||||
index_t sfk_row_stride;
|
||||
index_t sfv_row_stride;
|
||||
index_t sfq_head_stride;
|
||||
index_t sfk_head_stride;
|
||||
index_t sfv_head_stride;
|
||||
|
||||
// The number of heads.
|
||||
int h, h_k;
|
||||
// In the case of multi-query and grouped-query attention (MQA/GQA), nheads_k could be
|
||||
// different from nheads (query).
|
||||
int h_h_k_ratio; // precompute h / h_k,
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
struct Flash_fwd_params : public Qkv_params {
|
||||
|
||||
// The O matrix (output).
|
||||
void * __restrict__ o_ptr;
|
||||
void * __restrict__ oaccum_ptr;
|
||||
void * __restrict__ s_ptr;
|
||||
|
||||
// The stride between rows of O.
|
||||
index_t o_batch_stride;
|
||||
index_t o_row_stride;
|
||||
index_t o_head_stride;
|
||||
|
||||
// The pointer to the P matrix.
|
||||
void * __restrict__ p_ptr;
|
||||
|
||||
// The pointer to the softmax sum.
|
||||
void * __restrict__ softmax_lse_ptr;
|
||||
void * __restrict__ softmax_lseaccum_ptr;
|
||||
|
||||
// The dimensions.
|
||||
int b, seqlen_q, seqlen_k, seqlen_knew, d, seqlen_q_rounded, seqlen_k_rounded, d_rounded, rotary_dim, unpadded_seqlen_k;
|
||||
cutlass::FastDivmod head_divmod, m_block_divmod;
|
||||
int total_blocks;
|
||||
int seqlen_s;
|
||||
|
||||
// The scaling factors for the kernel.
|
||||
float scale_softmax;
|
||||
float scale_softmax_log2;
|
||||
uint32_t scale_softmax_log2_half2;
|
||||
|
||||
// array of length b+1 holding starting offset of each sequence.
|
||||
int * __restrict__ cu_seqlens_q;
|
||||
int * __restrict__ cu_seqlens_k;
|
||||
|
||||
// If provided, the actual length of each k sequence.
|
||||
int * __restrict__ seqused_k;
|
||||
|
||||
int *__restrict__ blockmask;
|
||||
|
||||
// The K_new and V_new matrices.
|
||||
void * __restrict__ knew_ptr;
|
||||
void * __restrict__ vnew_ptr;
|
||||
|
||||
// The stride between rows of the Q, K and V matrices.
|
||||
index_t knew_batch_stride;
|
||||
index_t vnew_batch_stride;
|
||||
index_t knew_row_stride;
|
||||
index_t vnew_row_stride;
|
||||
index_t knew_head_stride;
|
||||
index_t vnew_head_stride;
|
||||
|
||||
// The cos and sin matrices for rotary embedding.
|
||||
void * __restrict__ rotary_cos_ptr;
|
||||
void * __restrict__ rotary_sin_ptr;
|
||||
|
||||
// The indices to index into the KV cache.
|
||||
int * __restrict__ cache_batch_idx;
|
||||
|
||||
// Paged KV cache
|
||||
int * __restrict__ block_table;
|
||||
index_t block_table_batch_stride;
|
||||
int page_block_size;
|
||||
|
||||
// The dropout probability (probability of keeping an activation).
|
||||
float p_dropout;
|
||||
// uint32_t p_dropout_in_uint;
|
||||
// uint16_t p_dropout_in_uint16_t;
|
||||
uint8_t p_dropout_in_uint8_t;
|
||||
|
||||
// Scale factor of 1 / (1 - p_dropout).
|
||||
float rp_dropout;
|
||||
float scale_softmax_rp_dropout;
|
||||
|
||||
// Local window size
|
||||
int window_size_left, window_size_right;
|
||||
|
||||
// Random state.
|
||||
at::PhiloxCudaState philox_args;
|
||||
|
||||
// Pointer to the RNG seed (idx 0) and offset (idx 1).
|
||||
uint64_t * rng_state;
|
||||
|
||||
bool is_bf16;
|
||||
bool is_e4m3;
|
||||
bool is_causal;
|
||||
bool per_block_mean;
|
||||
bool single_level_p_quant; // If true, use single-level 1x16 block scale quantization for P (like V), instead of two-level quantization
|
||||
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
|
||||
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
|
||||
bool is_seqlens_k_cumulative;
|
||||
|
||||
bool is_rotary_interleaved;
|
||||
|
||||
int num_splits; // For split-KV version
|
||||
|
||||
void * __restrict__ alibi_slopes_ptr;
|
||||
index_t alibi_slopes_batch_stride;
|
||||
|
||||
int * __restrict__ tile_count_semaphore;
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -1,190 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <cmath>
|
||||
#include "cute/tensor.hpp"
|
||||
#include "cutlass/numeric_types.h"
|
||||
#include "utils.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <int Rows>
|
||||
struct SoftmaxFused{
|
||||
|
||||
using TensorT = decltype(make_fragment_like<float>(Shape<Int<Rows>>{}));
|
||||
TensorT row_sum, row_max, scores_scale;
|
||||
static constexpr float fp8_scalexfp4_scale = 1.f / (448 * 6);
|
||||
static constexpr float fp8_scalexfp4_scale_log2 = -11.392317422778762f; //log2f(fp8_scalexfp4_scale)
|
||||
static constexpr float fp4_scale_log2 = -2.584962500721156f; // log2f(fp4_scale)
|
||||
static constexpr int RowReductionThr = 4;
|
||||
|
||||
// If true, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly (standard per-block FP4 quantization like V)
|
||||
// If false (default), use two-level quantization: s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1)
|
||||
bool single_level_p_quant;
|
||||
|
||||
CUTLASS_DEVICE SoftmaxFused(bool single_level = false) : single_level_p_quant(single_level) {};
|
||||
|
||||
template<bool FirstTile, bool InfCheck = false, typename TensorAcc, typename TensorMax>
|
||||
CUTLASS_DEVICE auto online_softmax_with_quant(
|
||||
TensorAcc& acc,
|
||||
TensorMax& AbsMaxP,
|
||||
const float softmax_scale_log2
|
||||
) {
|
||||
Tensor acc_reduction_view = make_tensor(acc.data(), flash::convert_to_reduction_layout(acc.layout()));
|
||||
Tensor acc_conversion_view = make_tensor(acc.data(), flash::convert_to_conversion_layout(acc.layout()));
|
||||
Tensor acc_conversion_flatten = group_modes<1, 5>(group_modes<0, 2>(flatten(acc_conversion_view)));
|
||||
|
||||
if constexpr (FirstTile) {
|
||||
fill(row_max, -INFINITY);
|
||||
clear(row_sum);
|
||||
fill(scores_scale, 1.f);
|
||||
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
|
||||
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), acc_reduction_view(mi, make_coord(ei, ni)));
|
||||
}
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), AbsMaxP(mi, ni), 1); // exchange max with neighbour thread of 8 elements
|
||||
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), max_recv);
|
||||
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
|
||||
}
|
||||
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
|
||||
row_max(mi) = fmaxf(row_max(mi), max_recv);
|
||||
|
||||
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
|
||||
// - Pre-scales P to [0, 448×6] range before φ, output scaled by s_P1
|
||||
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
|
||||
// - No s_P1, just standard per-block FP4 quantization φ
|
||||
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
|
||||
const float max_scaled = InfCheck
|
||||
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
|
||||
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
|
||||
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
|
||||
}
|
||||
// s_P2 = max(P_block)/6 — per-block scale factor from φ function (same formula for both modes)
|
||||
// The difference is in max_scaled: two-level includes 448×6 pre-scaling, single-level doesn't
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
|
||||
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
|
||||
}
|
||||
}
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
|
||||
row_sum(mi) += acc_reduction_view(mi, ni);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
Tensor scores_max_prev = make_fragment_like(row_max);
|
||||
cute::copy(row_max, scores_max_prev);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
|
||||
float local_max = -INFINITY;
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
|
||||
local_max = fmaxf(local_max, acc_reduction_view(mi, make_coord(ei, ni)));
|
||||
}
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), local_max, 1); // exchange max with neighbour thread of 8 elements
|
||||
AbsMaxP(mi, ni) = fmaxf(local_max, max_recv);
|
||||
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
|
||||
}
|
||||
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
|
||||
row_max(mi) = fmaxf(row_max(mi), max_recv);
|
||||
|
||||
float scores_max_cur = !InfCheck
|
||||
? row_max(mi)
|
||||
: (row_max(mi) == -INFINITY ? 0.0f : row_max(mi));
|
||||
scores_scale(mi) = flash::ptx_exp2((scores_max_prev(mi) - scores_max_cur) * softmax_scale_log2);
|
||||
|
||||
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
|
||||
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
|
||||
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
|
||||
const float max_scaled = InfCheck
|
||||
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
|
||||
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
|
||||
row_sum(mi) = row_sum(mi) * scores_scale(mi);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
|
||||
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
|
||||
row_sum(mi) += acc_reduction_view(mi, ni);
|
||||
}
|
||||
// s_P2 = max(P_block)/6 — per-block scale factor from φ function
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
|
||||
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
|
||||
}
|
||||
// scores_scale(mi) = max_scaled;
|
||||
}
|
||||
}
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size(AbsMaxP); ++i) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int j = 0; j < size<0>(acc_conversion_flatten); ++j)
|
||||
acc_conversion_flatten(j, i) /= AbsMaxP(i);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename TensorAcc>
|
||||
CUTLASS_DEVICE void finalize(TensorAcc& o_store) {
|
||||
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size(row_max); ++mi) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 1; i < RowReductionThr; i <<= 1) {
|
||||
float sum_recv = __shfl_xor_sync(int32_t(-1), row_sum(mi), i);
|
||||
row_sum(mi) += sum_recv;
|
||||
}
|
||||
float sum = row_sum(mi);
|
||||
float inv_sum = (sum == 0.f || sum != sum) ? 0.f : 1 / sum;
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
|
||||
o_store_reduction_view(mi, ni) *= inv_sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename TensorAcc>
|
||||
CUTLASS_DEVICE void rescale_o(TensorAcc& o_store, TensorAcc const& o_tmp) {
|
||||
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
|
||||
Tensor o_tmp_reduction_view = make_tensor(o_tmp.data(), flash::convert_to_reduction_layout(o_tmp.layout()));
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size(row_max); ++mi) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
|
||||
o_store_reduction_view(mi, ni) = o_store_reduction_view(mi, ni) * scores_scale(mi) + o_tmp_reduction_view(mi, ni);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
};
|
||||
} // namespace flash
|
||||
@@ -1,83 +0,0 @@
|
||||
// Inspired by
|
||||
// https://github.com/NVIDIA/DALI/blob/main/include/dali/core/static_switch.h
|
||||
// and https://github.com/pytorch/pytorch/blob/master/aten/src/ATen/Dispatch.h
|
||||
|
||||
#pragma once
|
||||
|
||||
/// @param COND - a boolean expression to switch by
|
||||
/// @param CONST_NAME - a name given for the constexpr bool variable.
|
||||
/// @param ... - code to execute for true and false
|
||||
///
|
||||
/// Usage:
|
||||
/// ```
|
||||
/// BOOL_SWITCH(flag, BoolConst, [&] {
|
||||
/// some_function<BoolConst>(...);
|
||||
/// });
|
||||
/// ```
|
||||
//
|
||||
|
||||
#define BOOL_SWITCH(COND, CONST_NAME, ...) \
|
||||
[&] { \
|
||||
if (COND) { \
|
||||
constexpr static bool CONST_NAME = true; \
|
||||
return __VA_ARGS__(); \
|
||||
} else { \
|
||||
constexpr static bool CONST_NAME = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
|
||||
#define PREC_SWITCH(PRECTYPE, ...) \
|
||||
[&] { \
|
||||
if (PRECTYPE == 1) { \
|
||||
using kPrecType = cutlass::half_t; \
|
||||
constexpr static bool kSoftFp16 = false; \
|
||||
constexpr static bool kHybrid = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (PRECTYPE == 2) { \
|
||||
using kPrecType = cutlass::float_e4m3_t; \
|
||||
constexpr static bool kSoftFp16 = false; \
|
||||
constexpr static bool kHybrid = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (PRECTYPE == 3) { \
|
||||
using kPrecType = cutlass::float_e4m3_t; \
|
||||
constexpr static bool kSoftFp16 = false; \
|
||||
constexpr static bool kHybrid = true; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (PRECTYPE == 4) { \
|
||||
using kPrecType = cutlass::float_e4m3_t; \
|
||||
constexpr static bool kSoftFp16 = true; \
|
||||
constexpr static bool kHybrid = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
|
||||
#define HEADDIM_SWITCH(HEADDIM, ...) \
|
||||
[&] { \
|
||||
if (HEADDIM == 64) { \
|
||||
constexpr static int kHeadSize = 64; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (HEADDIM == 128) { \
|
||||
constexpr static int kHeadSize = 128; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (HEADDIM == 256) { \
|
||||
constexpr static int kHeadSize = 256; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
|
||||
#define SEQLEN_SWITCH(USE_VAR_SEQ_LEN, SEQ_LEN_OUT_OF_BOUND_CHECK, ...) \
|
||||
[&] { \
|
||||
if (!USE_VAR_SEQ_LEN) { \
|
||||
if (SEQ_LEN_OUT_OF_BOUND_CHECK) { \
|
||||
using kSeqLenTraitsType = FixedSeqLenTraits<true>; \
|
||||
return __VA_ARGS__(); \
|
||||
} else { \
|
||||
using kSeqLenTraitsType = FixedSeqLenTraits<false>; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
} else { \
|
||||
using kSeqLenTraitsType = VarSeqLenTraits; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
@@ -1,304 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
|
||||
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/fast_math.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
class StaticPersistentTileSchedulerOld {
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
private:
|
||||
int current_work_linear_idx_;
|
||||
cutlass::FastDivmod const &m_block_divmod, &head_divmod;
|
||||
int const total_blocks;
|
||||
|
||||
public:
|
||||
struct WorkTileInfo {
|
||||
int M_idx = 0;
|
||||
int H_idx = 0;
|
||||
int B_idx = 0;
|
||||
bool is_valid_tile = false;
|
||||
|
||||
CUTLASS_HOST_DEVICE
|
||||
bool
|
||||
is_valid() const {
|
||||
return is_valid_tile;
|
||||
}
|
||||
|
||||
CUTLASS_HOST_DEVICE
|
||||
static WorkTileInfo
|
||||
invalid_work_tile() {
|
||||
return {-1, -1, -1, false};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
public:
|
||||
|
||||
CUTLASS_DEVICE explicit StaticPersistentTileSchedulerOld(cutlass::FastDivmod const &m_block_divmod_,
|
||||
cutlass::FastDivmod const &head_divmod_,
|
||||
int const total_blocks_) :
|
||||
m_block_divmod(m_block_divmod_), head_divmod(head_divmod_), total_blocks(total_blocks_) {
|
||||
|
||||
// MSVC requires protecting use of CUDA-specific nonstandard syntax,
|
||||
// like blockIdx and gridDim, with __CUDA_ARCH__.
|
||||
#if defined(__CUDA_ARCH__)
|
||||
// current_work_linear_idx_ = blockIdx.x + blockIdx.y * gridDim.x + blockIdx.z * gridDim.x * gridDim.y;
|
||||
current_work_linear_idx_ = blockIdx.x;
|
||||
#else
|
||||
CUTLASS_ASSERT(false && "This line should never be reached");
|
||||
#endif
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_current_work() const {
|
||||
return get_current_work_for_linear_idx(current_work_linear_idx_);
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_current_work_for_linear_idx(int linear_idx) const {
|
||||
if (linear_idx >= total_blocks) {
|
||||
return WorkTileInfo::invalid_work_tile();
|
||||
}
|
||||
|
||||
// Map worker's linear index into the CTA tiled problem shape to the corresponding MHB indices
|
||||
int M_idx, H_idx, B_idx;
|
||||
int quotient = m_block_divmod.divmod(M_idx, linear_idx);
|
||||
B_idx = head_divmod.divmod(H_idx, quotient);
|
||||
return {M_idx, H_idx, B_idx, true};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
void
|
||||
// advance_to_next_work(int advance_count = 1) {
|
||||
advance_to_next_work() {
|
||||
// current_work_linear_idx_ += int(gridDim.x * gridDim.y * gridDim.z);
|
||||
current_work_linear_idx_ += int(gridDim.x);
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
fetch_next_work() {
|
||||
WorkTileInfo new_work_tile_info;
|
||||
advance_to_next_work();
|
||||
new_work_tile_info = get_current_work();
|
||||
return new_work_tile_info;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
class SingleTileScheduler {
|
||||
|
||||
public:
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
int const num_blocks_m, num_head, num_batch;
|
||||
int const* tile_count_semaphore = nullptr;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
return {};
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_grid_dim(Arguments const& args, int num_sm) {
|
||||
return {uint32_t(args.num_blocks_m), uint32_t(args.num_head), uint32_t(args.num_batch)};
|
||||
}
|
||||
|
||||
struct WorkTileInfo {
|
||||
int M_idx = 0;
|
||||
int H_idx = 0;
|
||||
int B_idx = 0;
|
||||
bool is_valid_tile = false;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
bool
|
||||
is_valid(Params const& params) const {
|
||||
return is_valid_tile;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
cute::tuple<int32_t, int32_t, int32_t>
|
||||
get_block_coord(Params const& params) const {
|
||||
return {M_idx, H_idx, B_idx};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params) const {
|
||||
return {-1, -1, -1, false};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_initial_work() const {
|
||||
return {int(blockIdx.x), int(blockIdx.y), int(blockIdx.z), true};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
|
||||
return {-1, -1, -1, false};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
class StaticPersistentTileScheduler {
|
||||
|
||||
public:
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
int const num_blocks_m, num_head, num_batch;
|
||||
int const* tile_count_semaphore = nullptr;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
int total_blocks;
|
||||
cutlass::FastDivmod m_block_divmod, head_divmod;
|
||||
};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
return {args.num_blocks_m * args.num_head * args.num_batch,
|
||||
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head)};
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_grid_dim(Arguments const& args, int num_sm) {
|
||||
return {uint32_t(num_sm)};
|
||||
}
|
||||
|
||||
struct WorkTileInfo {
|
||||
int tile_idx;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
bool
|
||||
is_valid(Params const& params) const {
|
||||
return tile_idx < params.total_blocks;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
cute::tuple<int32_t, int32_t, int32_t>
|
||||
get_block_coord(Params const& params) const {
|
||||
int m_block, bidh, bidb;
|
||||
bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
|
||||
return {m_block, bidh, bidb};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_initial_work() const {
|
||||
return {int(blockIdx.x)};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
|
||||
return {current_work.tile_idx + int(gridDim.x)};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
class DynamicPersistentTileScheduler {
|
||||
|
||||
public:
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
int const num_blocks_m, num_head, num_batch;
|
||||
int const* tile_count_semaphore;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
int const total_blocks;
|
||||
cutlass::FastDivmod const m_block_divmod, head_divmod;
|
||||
int const* tile_count_semaphore;
|
||||
};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
return {args.num_blocks_m * args.num_head * args.num_batch,
|
||||
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head),
|
||||
args.tile_count_semaphore};
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_grid_dim(Arguments const& args, int num_sm) {
|
||||
return {uint32_t(num_sm)};
|
||||
}
|
||||
|
||||
using WorkTileInfo = StaticPersistentTileScheduler::WorkTileInfo;
|
||||
// struct WorkTileInfo {
|
||||
// int tile_idx;
|
||||
|
||||
// CUTLASS_DEVICE
|
||||
// bool
|
||||
// is_valid(Params const& params) const {
|
||||
// return tile_idx < params.total_blocks;
|
||||
// }
|
||||
|
||||
// CUTLASS_DEVICE
|
||||
// cute::tuple<int32_t, int32_t, int32_t>
|
||||
// get_block_coord(Params const& params) const {
|
||||
// int m_block, bidh, bidb;
|
||||
// bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
|
||||
// return {m_block, bidh, bidb};
|
||||
// }
|
||||
|
||||
// };
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_initial_work() const {
|
||||
return {int(blockIdx.x)};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
|
||||
return {current_work.tile_idx + int(gridDim.x)};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
} // flash
|
||||
@@ -1,408 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <assert.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
#include <cuda_fp16.h>
|
||||
|
||||
#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800
|
||||
#include <cuda_bf16.h>
|
||||
#endif
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
|
||||
#include <cutlass/array.h>
|
||||
#include <cutlass/cutlass.h>
|
||||
#include <cutlass/numeric_conversion.h>
|
||||
#include <cutlass/numeric_types.h>
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename T>
|
||||
struct MaxOp {
|
||||
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x > y ? x : y; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct MaxOp<float> {
|
||||
// This is slightly faster
|
||||
__device__ __forceinline__ float operator()(float const &x, float const &y) { return max(x, y); }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename T>
|
||||
struct SumOp {
|
||||
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x + y; }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<int THREADS>
|
||||
struct Allreduce {
|
||||
static_assert(THREADS == 32 || THREADS == 16 || THREADS == 8 || THREADS == 4);
|
||||
template<typename T, typename Operator>
|
||||
static __device__ __forceinline__ T run(T x, Operator &op) {
|
||||
constexpr int OFFSET = THREADS / 2;
|
||||
x = op(x, __shfl_xor_sync(uint32_t(-1), x, OFFSET));
|
||||
return Allreduce<OFFSET>::run(x, op);
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<>
|
||||
struct Allreduce<2> {
|
||||
template<typename T, typename Operator>
|
||||
static __device__ __forceinline__ T run(T x, Operator &op) {
|
||||
x = op(x, __shfl_xor_sync(uint32_t(-1), x, 1));
|
||||
return x;
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
|
||||
__device__ __forceinline__ void thread_reduce_(Tensor<Engine0, Layout0> const &tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
|
||||
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
|
||||
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
|
||||
CUTE_STATIC_ASSERT_V(size<0>(summary) == size<0>(tensor));
|
||||
#pragma unroll
|
||||
for (int mi = 0; mi < size<0>(tensor); mi++) {
|
||||
summary(mi) = zero_init ? tensor(mi, 0) : op(summary(mi), tensor(mi, 0));
|
||||
#pragma unroll
|
||||
for (int ni = 1; ni < size<1>(tensor); ni++) {
|
||||
summary(mi) = op(summary(mi), tensor(mi, ni));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
|
||||
__device__ __forceinline__ void quad_allreduce_(Tensor<Engine0, Layout0> &dst, Tensor<Engine1, Layout1> &src, Operator &op) {
|
||||
CUTE_STATIC_ASSERT_V(size(dst) == size(src));
|
||||
#pragma unroll
|
||||
for (int i = 0; i < size(dst); i++){
|
||||
dst(i) = Allreduce<4>::run(src(i), op);
|
||||
}
|
||||
}
|
||||
|
||||
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
|
||||
__device__ __forceinline__ void reduce_(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
|
||||
thread_reduce_<zero_init>(tensor, summary, op);
|
||||
quad_allreduce_(summary, summary, op);
|
||||
}
|
||||
|
||||
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__device__ __forceinline__ void reduce_max(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &max){
|
||||
MaxOp<float> max_op;
|
||||
reduce_<zero_init>(tensor, max, max_op);
|
||||
}
|
||||
|
||||
template<bool zero_init=true, bool warp_reduce=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__device__ __forceinline__ void reduce_sum(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &sum){
|
||||
SumOp<float> sum_op;
|
||||
thread_reduce_<zero_init>(tensor, sum, sum_op);
|
||||
if constexpr (warp_reduce) { quad_allreduce_(sum, sum, sum_op); }
|
||||
}
|
||||
|
||||
__forceinline__ __device__ __half2 half_exp(__half2 x) {
|
||||
uint32_t tmp_out, tmp_in;
|
||||
tmp_in = reinterpret_cast<uint32_t&>(x);
|
||||
asm ("ex2.approx.f16x2 %0, %1;\n"
|
||||
: "=r"(tmp_out)
|
||||
: "r"(tmp_in));
|
||||
__half2 out = reinterpret_cast<__half2&>(tmp_out);
|
||||
return out;
|
||||
}
|
||||
|
||||
// Apply the exp to all the elements.
|
||||
template <bool zero_init=false, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__forceinline__ __device__ void max_scale_exp2_sum(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> &max, Tensor<Engine1, Layout1> &sum, const float scale) {
|
||||
static_assert(Layout0::rank == 2, "Only support 2D Tensor"); static_assert(Layout1::rank == 1, "Only support 1D Tensor"); CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
|
||||
#pragma unroll
|
||||
for (int mi = 0; mi < size<0>(tensor); ++mi) {
|
||||
MaxOp<float> max_op;
|
||||
max(mi) = zero_init ? tensor(mi, 0) : max_op(max(mi), tensor(mi, 0));
|
||||
#pragma unroll
|
||||
for (int ni = 1; ni < size<1>(tensor); ni++) {
|
||||
max(mi) = max_op(max(mi), tensor(mi, ni));
|
||||
}
|
||||
max(mi) = Allreduce<4>::run(max(mi), max_op);
|
||||
// If max is -inf, then all elements must have been -inf (possibly due to masking).
|
||||
// We don't want (-inf - (-inf)) since that would give NaN.
|
||||
const float max_scaled = max(mi) == -INFINITY ? 0.f : max(mi) * scale;
|
||||
sum(mi) = 0;
|
||||
#pragma unroll
|
||||
for (int ni = 0; ni < size<1>(tensor); ++ni) {
|
||||
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
|
||||
// max * log_2(e)) This allows the compiler to use the ffma
|
||||
// instruction instead of fadd and fmul separately.
|
||||
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
|
||||
sum(mi) += tensor(mi, ni);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Apply the exp to all the elements.
|
||||
template <bool Scale_max=true, bool Check_inf=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__forceinline__ __device__ void scale_apply_exp2(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> const &max, const float scale) {
|
||||
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
|
||||
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
|
||||
CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
|
||||
#pragma unroll
|
||||
for (int mi = 0; mi < size<0>(tensor); ++mi) {
|
||||
// If max is -inf, then all elements must have been -inf (possibly due to masking).
|
||||
// We don't want (-inf - (-inf)) since that would give NaN.
|
||||
// If we don't have float around M_LOG2E the multiplication is done in fp64.
|
||||
const float max_scaled = Check_inf
|
||||
? (max(mi) == -INFINITY ? 0.f : (max(mi) * (Scale_max ? scale : float(M_LOG2E))))
|
||||
: (max(mi) * (Scale_max ? scale : float(M_LOG2E)));
|
||||
#pragma unroll
|
||||
for (int ni = 0; ni < size<1>(tensor); ++ni) {
|
||||
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
|
||||
// max * log_2(e)) This allows the compiler to use the ffma
|
||||
// instruction instead of fadd and fmul separately.
|
||||
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
__forceinline__ __device__ float ptx_exp2(float x) {
|
||||
float y;
|
||||
asm volatile("ex2.approx.ftz.f32 %0, %1;" : "=f"(y) : "f"(x));
|
||||
return y;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
packed_float_to_ue4m3(
|
||||
float const &f0, float const &f1, float const &f2, float const &f3,
|
||||
uint32_t &out
|
||||
) {
|
||||
asm volatile( \
|
||||
"{\n" \
|
||||
".reg .b16 lo;\n" \
|
||||
".reg .b16 hi;\n" \
|
||||
"cvt.rn.satfinite.e4m3x2.f32 lo, %2, %1;\n" \
|
||||
"cvt.rn.satfinite.e4m3x2.f32 hi, %4, %3;\n" \
|
||||
"mov.b32 %0, {lo, hi};\n" \
|
||||
"}" \
|
||||
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
packed_float_to_e2m1(
|
||||
float const &f0, float const &f1, float const &f2, float const& f3,
|
||||
float const &f4, float const &f5, float const &f6, float const& f7,
|
||||
uint32_t &out
|
||||
) {
|
||||
|
||||
asm volatile( \
|
||||
"{\n" \
|
||||
".reg .b8 byte0;\n" \
|
||||
".reg .b8 byte1;\n" \
|
||||
".reg .b8 byte2;\n" \
|
||||
".reg .b8 byte3;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n" \
|
||||
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n" \
|
||||
"}" \
|
||||
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3),
|
||||
"f"(f4), "f"(f5), "f"(f6), "f"(f7));
|
||||
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
add(float2 & c,
|
||||
float2 const& a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("add.f32x2 %0, %1, %2;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(c))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
add_inplace(float2 &a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("add.f32x2 %0, %0, %1;\n"
|
||||
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
|
||||
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
sub(float2 & c,
|
||||
float2 const& a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("sub.f32x2 %0, %1, %2;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(c))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
sub_inplace(float2 &a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("sub.f32x2 %0, %0, %1;\n"
|
||||
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
|
||||
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
mul(float2 & c,
|
||||
float2 const& a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("mul.f32x2 %0, %1, %2;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(c))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
fma(float2 & d,
|
||||
float2 const& a,
|
||||
float2 const& b,
|
||||
float2 const& c)
|
||||
{
|
||||
asm volatile("fma.rn.f32x2 %0, %1, %2, %3;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(d))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(c)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
fma_inplace(float2 &a,
|
||||
float2 const& b,
|
||||
float2 const& c)
|
||||
{
|
||||
asm volatile("fma.rn.f32x2 %0, %0, %1, %2;\n"
|
||||
: "+l"(reinterpret_cast<uint64_t &>(a))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(b)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(c)));
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <
|
||||
class Layout
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_reduction_layout(Layout mma_layout) {
|
||||
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
|
||||
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
|
||||
|
||||
return make_layout(
|
||||
make_layout(get<0,1>(mma_layout), get<1>(mma_layout)),
|
||||
make_layout(get<0,0>(mma_layout), get<2>(mma_layout))
|
||||
);
|
||||
}
|
||||
|
||||
template <
|
||||
class Tensor
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_reduction_tensor(Tensor mma_tensor) {
|
||||
return make_tensor(mma_tensor.data(), convert_to_reduction_layout(mma_tensor.layout()));
|
||||
}
|
||||
|
||||
|
||||
template <
|
||||
class Layout
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_conversion_layout(Layout mma_layout) {
|
||||
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
|
||||
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
|
||||
|
||||
constexpr int MmaAtomN = size<0, 0>(mma_layout);
|
||||
constexpr int MmaAtomM = size<0, 1>(mma_layout);
|
||||
constexpr int MmaM = size<1>(mma_layout);
|
||||
constexpr int MmaN = size<2>(mma_layout);
|
||||
|
||||
static_assert(MmaAtomN == 8, "MmaAtomN should be 8.");
|
||||
static_assert(MmaAtomM == 2, "MmaAtomM should be 2.");
|
||||
static_assert(MmaN % 2 == 0, "MmaN should be multiple of 2.");
|
||||
|
||||
auto mma_n_division = zipped_divide(
|
||||
layout<2>(mma_layout), make_tile(_2{})
|
||||
);
|
||||
return make_layout(
|
||||
make_layout(layout<0,0>(mma_layout), make_layout(layout<0,1>(mma_layout), layout<0>(mma_n_division))),
|
||||
layout<1>(mma_layout), layout<1>(mma_n_division)
|
||||
);
|
||||
}
|
||||
|
||||
template <
|
||||
class Tensor
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_conversion_tensor(Tensor mma_tensor) {
|
||||
return make_tensor(mma_tensor.data(), convert_to_conversion_layout(mma_tensor.layout()));
|
||||
}
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <bool Is_even_MN=true, bool Is_even_K=true, bool Clear_OOB_MN=false, bool Clear_OOB_K=true,
|
||||
typename TiledCopy, typename Engine0, typename Layout0, typename Engine1, typename Layout1,
|
||||
typename Engine2, typename Layout2, typename Engine3, typename Layout3>
|
||||
CUTLASS_DEVICE void copy(TiledCopy tiled_copy, Tensor<Engine0, Layout0> const &S,
|
||||
Tensor<Engine1, Layout1> &D, Tensor<Engine2, Layout2> const &identity_MN,
|
||||
Tensor<Engine3, Layout3> const &predicate_K, const int max_MN=0) {
|
||||
CUTE_STATIC_ASSERT_V(rank(S) == Int<3>{});
|
||||
CUTE_STATIC_ASSERT_V(rank(D) == Int<3>{});
|
||||
CUTE_STATIC_ASSERT_V(size<0>(S) == size<0>(D)); // MMA
|
||||
CUTE_STATIC_ASSERT_V(size<1>(S) == size<1>(D)); // MMA_M
|
||||
CUTE_STATIC_ASSERT_V(size<2>(S) == size<2>(D)); // MMA_K
|
||||
// There's no case where !Clear_OOB_K && Clear_OOB_MN
|
||||
static_assert(!(Clear_OOB_MN && !Clear_OOB_K));
|
||||
#pragma unroll
|
||||
for (int m = 0; m < size<1>(S); ++m) {
|
||||
if (Is_even_MN || get<0>(identity_MN(0, m, 0)) < max_MN) {
|
||||
#pragma unroll
|
||||
for (int k = 0; k < size<2>(S); ++k) {
|
||||
if (Is_even_K || predicate_K(k)) {
|
||||
cute::copy(tiled_copy, S(_, m, k), D(_, m, k));
|
||||
} else if (Clear_OOB_K) {
|
||||
cute::clear(D(_, m, k));
|
||||
}
|
||||
}
|
||||
} else if (Clear_OOB_MN) {
|
||||
cute::clear(D(_, m, _));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace flash
|
||||
@@ -1 +0,0 @@
|
||||
__version__ = "3.0.0.b1"
|
||||
@@ -1,90 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import fp4quant
|
||||
from triton.tools.mxfp import MXFP4Tensor
|
||||
|
||||
from bench_utils import bench_kineto
|
||||
b = 1
|
||||
h = 32
|
||||
n = 16384
|
||||
d = 128
|
||||
|
||||
def test():
|
||||
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
|
||||
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
|
||||
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
|
||||
fp4quant.scaled_fp4_quant_permute(q, o, o_s, 1)
|
||||
|
||||
test()
|
||||
|
||||
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
|
||||
|
||||
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
|
||||
throughput = IO / t * 1e-9
|
||||
|
||||
print(f"Throughput: {throughput:.2f} GB/s")
|
||||
|
||||
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
|
||||
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
|
||||
B, H, M, N = x.shape
|
||||
x = x.view(B, H, M, N // 16, 16)
|
||||
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
|
||||
if all_ones:
|
||||
scales = torch.ones_like(scales)
|
||||
x_scaled = x / scales
|
||||
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
|
||||
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
|
||||
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
|
||||
permuted_fp8_scale = None
|
||||
if permuted:
|
||||
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
|
||||
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
|
||||
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
|
||||
|
||||
b = 2
|
||||
h = 4
|
||||
n = 251
|
||||
n_padded = (n + 127) // 128 * 128
|
||||
d = 128
|
||||
|
||||
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
|
||||
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
|
||||
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
|
||||
|
||||
k_permute = [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
|
||||
o_permuted = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s_permuted = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
fp4quant.scaled_fp4_quant_permute(q, o_permuted, o_s_permuted, 1)
|
||||
|
||||
# padding
|
||||
if n % 128 != 0:
|
||||
o_permuted_gt = torch.cat([o, torch.zeros((b, h, n_padded - n, d // 2), dtype=torch.uint8, device='cuda')], dim=2)
|
||||
o_s_permuted_gt = torch.cat([o_s, torch.zeros((b, h, n_padded - n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')], dim=2)
|
||||
else:
|
||||
o_permuted_gt = o
|
||||
o_s_permuted_gt = o_s
|
||||
|
||||
# use scale_and_fp4_tensor + torch permutation to get the ground truth
|
||||
o_permuted_gt = o_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 2)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 2)
|
||||
o_s_permuted_gt = o_s_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 16)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 16)
|
||||
|
||||
assert((o_permuted - o_permuted_gt).abs().max() == 0)
|
||||
assert((o_s_permuted.float() - o_s_permuted_gt.float()).abs().max() == 0)
|
||||
|
||||
print("All tests passed!")
|
||||
@@ -1,86 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import fp4quant
|
||||
from triton.tools.mxfp import MXFP4Tensor
|
||||
|
||||
from bench_utils import bench_kineto
|
||||
b = 1
|
||||
h = 32
|
||||
n = 16384
|
||||
d = 128
|
||||
|
||||
def test():
|
||||
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
|
||||
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
|
||||
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
|
||||
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
|
||||
|
||||
test()
|
||||
|
||||
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
|
||||
|
||||
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
|
||||
throughput = IO / t * 1e-9
|
||||
|
||||
print(f"Throughput: {throughput:.2f} GB/s")
|
||||
|
||||
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
|
||||
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
|
||||
B, H, M, N = x.shape
|
||||
x = x.view(B, H, M, N // 16, 16)
|
||||
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
|
||||
if all_ones:
|
||||
scales = torch.ones_like(scales)
|
||||
x_scaled = x / scales
|
||||
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
|
||||
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
|
||||
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
|
||||
permuted_fp8_scale = None
|
||||
if permuted:
|
||||
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
|
||||
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
|
||||
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
|
||||
|
||||
b = 2
|
||||
h = 4
|
||||
n = 251
|
||||
d = 128
|
||||
|
||||
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
|
||||
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
|
||||
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
|
||||
|
||||
fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale = scale_and_fp4_tensor(q, packed_dim=3)
|
||||
|
||||
assert((fp8_scale.float() - o_s.float()).abs().max() == 0)
|
||||
|
||||
o_binary = [
|
||||
(int(bin_str[:4], 2), int(bin_str[4:], 2))
|
||||
for bin_str in [format(x.item(), '08b') for x in o.view(-1)]
|
||||
]
|
||||
o_binary_gt = [
|
||||
(int(bin_str[:4], 2), int(bin_str[4:], 2))
|
||||
for bin_str in [format(x.item(), '08b') for x in packed_fp4.view(-1)]
|
||||
]
|
||||
for i in range(len(o_binary)):
|
||||
# check contiguous 4 bits. Difference should be at most one
|
||||
assert(abs(o_binary[i][0] - o_binary_gt[i][0]) <= 1)
|
||||
assert(abs(o_binary[i][1] - o_binary_gt[i][1]) <= 1)
|
||||
|
||||
print("All tests passed!")
|
||||
@@ -1,86 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import fp4quant
|
||||
from triton.tools.mxfp import MXFP4Tensor
|
||||
|
||||
from bench_utils import bench_kineto
|
||||
b = 1
|
||||
h = 32
|
||||
n = 16384
|
||||
d = 128
|
||||
|
||||
def test():
|
||||
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
|
||||
o = torch.empty((b, h, d, n // 2), device="cuda", dtype=torch.uint8)
|
||||
o_s = torch.empty((b, h, d, n // 16), device="cuda", dtype=torch.float8_e4m3fn)
|
||||
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
|
||||
|
||||
test()
|
||||
|
||||
t = bench_kineto(test, "scaled_fp4_quant_trans_kernel", suppress_kineto_output=True)
|
||||
|
||||
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
|
||||
throughput = IO / t * 1e-9
|
||||
|
||||
print(f"Throughput: {throughput:.2f} GB/s")
|
||||
|
||||
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
|
||||
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
|
||||
B, H, M, N = x.shape
|
||||
x = x.view(B, H, M, N // 16, 16)
|
||||
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
|
||||
if all_ones:
|
||||
scales = torch.ones_like(scales)
|
||||
x_scaled = x / scales
|
||||
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
|
||||
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
|
||||
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
|
||||
permuted_fp8_scale = None
|
||||
if permuted:
|
||||
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
|
||||
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
|
||||
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
|
||||
|
||||
b = 2
|
||||
h = 4
|
||||
n = 491
|
||||
n_padded = (n + 127) // 128 * 128
|
||||
d = 128
|
||||
|
||||
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
|
||||
o = torch.empty((b, h, d, n_padded // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s = torch.empty((b, h, d, n_padded // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
|
||||
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
|
||||
|
||||
if n % 128 != 0:
|
||||
q_padded = torch.cat([q, torch.zeros((b, h, n_padded - n, d), dtype=torch.float16, device='cuda')], dim=2)
|
||||
else:
|
||||
q_padded = q
|
||||
|
||||
# use torch transpose + scaled_fp4_quant to get the ground truth
|
||||
q_padded = q_padded.transpose(2, 3).reshape(b, h, n_padded, d).contiguous()
|
||||
o_gt = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s_gt = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
fp4quant.scaled_fp4_quant(q_padded, o_gt, o_s_gt, 1)
|
||||
o_gt = o_gt.reshape(b, h, d, n_padded // 2).contiguous()
|
||||
o_s_gt = o_s_gt.reshape(b, h, d, n_padded // 16).contiguous()
|
||||
|
||||
assert((o_s_gt.float() - o_s.float()).abs().max() == 0)
|
||||
assert((o_gt - o).abs().max() == 0)
|
||||
|
||||
print("All tests passed!")
|
||||
@@ -1,169 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
import torch
|
||||
import torch.distributed as dist
|
||||
|
||||
|
||||
def bench(fn, num_warmups: int = 5, num_tests: int = 10,
|
||||
high_precision: bool = False):
|
||||
# Flush L2 cache with 256 MB data
|
||||
torch.cuda.synchronize()
|
||||
cache = torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda')
|
||||
cache.zero_()
|
||||
|
||||
# Warmup
|
||||
for _ in range(num_warmups):
|
||||
fn()
|
||||
|
||||
# Add a large kernel to eliminate the CPU launch overhead
|
||||
if high_precision:
|
||||
x = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
y = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
x @ y
|
||||
|
||||
# Testing
|
||||
start_event = torch.cuda.Event(enable_timing=True)
|
||||
end_event = torch.cuda.Event(enable_timing=True)
|
||||
start_event.record()
|
||||
for i in range(num_tests):
|
||||
fn()
|
||||
end_event.record()
|
||||
torch.cuda.synchronize()
|
||||
|
||||
return start_event.elapsed_time(end_event) / num_tests
|
||||
|
||||
|
||||
class empty_suppress:
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *_):
|
||||
pass
|
||||
|
||||
|
||||
class suppress_stdout_stderr:
|
||||
def __enter__(self):
|
||||
self.outnull_file = open(os.devnull, 'w')
|
||||
self.errnull_file = open(os.devnull, 'w')
|
||||
|
||||
self.old_stdout_fileno_undup = sys.stdout.fileno()
|
||||
self.old_stderr_fileno_undup = sys.stderr.fileno()
|
||||
|
||||
self.old_stdout_fileno = os.dup(sys.stdout.fileno())
|
||||
self.old_stderr_fileno = os.dup(sys.stderr.fileno())
|
||||
|
||||
self.old_stdout = sys.stdout
|
||||
self.old_stderr = sys.stderr
|
||||
|
||||
os.dup2(self.outnull_file.fileno(), self.old_stdout_fileno_undup)
|
||||
os.dup2(self.errnull_file.fileno(), self.old_stderr_fileno_undup)
|
||||
|
||||
sys.stdout = self.outnull_file
|
||||
sys.stderr = self.errnull_file
|
||||
return self
|
||||
|
||||
def __exit__(self, *_):
|
||||
sys.stdout = self.old_stdout
|
||||
sys.stderr = self.old_stderr
|
||||
|
||||
os.dup2(self.old_stdout_fileno, self.old_stdout_fileno_undup)
|
||||
os.dup2(self.old_stderr_fileno, self.old_stderr_fileno_undup)
|
||||
|
||||
os.close(self.old_stdout_fileno)
|
||||
os.close(self.old_stderr_fileno)
|
||||
|
||||
self.outnull_file.close()
|
||||
self.errnull_file.close()
|
||||
|
||||
|
||||
def bench_kineto(fn, kernel_names, num_tests: int = 30, suppress_kineto_output: bool = False,
|
||||
trace_path: str = None, barrier_comm_profiling: bool = False, flush_l2: bool = False):
|
||||
# Conflict with Nsight Systems
|
||||
using_nsys = os.environ.get('DG_NSYS_PROFILING', False)
|
||||
|
||||
# For some auto-tuning kernels with prints
|
||||
fn()
|
||||
|
||||
# Profile
|
||||
suppress = suppress_stdout_stderr if suppress_kineto_output and not using_nsys else empty_suppress
|
||||
with suppress():
|
||||
schedule = torch.profiler.schedule(wait=0, warmup=1, active=1, repeat=1) if not using_nsys else None
|
||||
profiler = torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CUDA], schedule=schedule) if not using_nsys else empty_suppress()
|
||||
with profiler:
|
||||
for i in range(2):
|
||||
# NOTES: use a large kernel and a barrier to eliminate the unbalanced CPU launch overhead
|
||||
if barrier_comm_profiling:
|
||||
lhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
rhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
lhs @ rhs
|
||||
dist.all_reduce(torch.ones(1, dtype=torch.float, device='cuda'))
|
||||
for _ in range(num_tests):
|
||||
if flush_l2:
|
||||
torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda').zero_()
|
||||
fn()
|
||||
|
||||
if not using_nsys:
|
||||
profiler.step()
|
||||
|
||||
# Return 1 if using Nsight Systems
|
||||
if using_nsys:
|
||||
return 1
|
||||
|
||||
# Parse the profiling table
|
||||
assert isinstance(kernel_names, str) or isinstance(kernel_names, tuple)
|
||||
is_tupled = isinstance(kernel_names, tuple)
|
||||
prof_lines = profiler.key_averages().table(sort_by='cuda_time_total', max_name_column_width=100).split('\n')
|
||||
kernel_names = (kernel_names, ) if isinstance(kernel_names, str) else kernel_names
|
||||
assert all([isinstance(name, str) for name in kernel_names])
|
||||
for name in kernel_names:
|
||||
assert sum([name in line for line in prof_lines]) == 1, f'Errors of the kernel {name} in the profiling table'
|
||||
|
||||
# Save chrome traces
|
||||
if trace_path is not None:
|
||||
profiler.export_chrome_trace(trace_path)
|
||||
|
||||
# Return average kernel times
|
||||
units = {'ms': 1e3, 'us': 1e6}
|
||||
kernel_times = []
|
||||
for name in kernel_names:
|
||||
for line in prof_lines:
|
||||
if name in line:
|
||||
time_str = line.split()[-2]
|
||||
for unit, scale in units.items():
|
||||
if unit in time_str:
|
||||
kernel_times.append(float(time_str.replace(unit, '')) / scale)
|
||||
break
|
||||
break
|
||||
return tuple(kernel_times) if is_tupled else kernel_times[0]
|
||||
|
||||
|
||||
def calc_diff(x, y):
|
||||
x, y = x.double(), y.double()
|
||||
denominator = (x * x + y * y).sum()
|
||||
sim = 2 * (x * y).sum() / denominator
|
||||
return 1 - sim
|
||||
|
||||
|
||||
def count_bytes(tensors):
|
||||
total = 0
|
||||
for t in tensors:
|
||||
if isinstance(t, tuple):
|
||||
total += count_bytes(t)
|
||||
else:
|
||||
total += t.numel() * t.element_size()
|
||||
return total
|
||||
@@ -1,52 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <stdio.h>
|
||||
|
||||
#if defined(__HIPCC__)
|
||||
#define HOST_DEVICE_INLINE __host__ __device__
|
||||
#define DEVICE_INLINE __device__
|
||||
#define HOST_INLINE __host__
|
||||
#elif defined(__CUDACC__) || defined(_NVHPC_CUDA)
|
||||
#define HOST_DEVICE_INLINE __host__ __device__ __forceinline__
|
||||
#define DEVICE_INLINE __device__ __forceinline__
|
||||
#define HOST_INLINE __host__ __forceinline__
|
||||
#else
|
||||
#define HOST_DEVICE_INLINE inline
|
||||
#define DEVICE_INLINE inline
|
||||
#define HOST_INLINE inline
|
||||
#endif
|
||||
|
||||
#define CUDA_CHECK(cmd) \
|
||||
do { \
|
||||
cudaError_t e = cmd; \
|
||||
if (e != cudaSuccess) { \
|
||||
printf("Failed: Cuda error %s:%d '%s'\n", __FILE__, __LINE__, \
|
||||
cudaGetErrorString(e)); \
|
||||
exit(EXIT_FAILURE); \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
int64_t get_device_attribute(int64_t attribute, int64_t device_id) {
|
||||
static int value = [=]() {
|
||||
int device = static_cast<int>(device_id);
|
||||
if (device < 0) {
|
||||
CUDA_CHECK(cudaGetDevice(&device));
|
||||
}
|
||||
int value;
|
||||
CUDA_CHECK(cudaDeviceGetAttribute(
|
||||
&value, static_cast<cudaDeviceAttr>(attribute), device));
|
||||
return static_cast<int>(value);
|
||||
}();
|
||||
|
||||
return value;
|
||||
}
|
||||
|
||||
namespace cuda_utils {
|
||||
|
||||
template <typename T>
|
||||
HOST_DEVICE_INLINE constexpr std::enable_if_t<std::is_integral_v<T>, T>
|
||||
ceil_div(T a, T b) {
|
||||
return (a + b - 1) / b;
|
||||
}
|
||||
|
||||
}; // namespace cuda_utils
|
||||
@@ -1,629 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#include <torch/all.h>
|
||||
#include <torch/python.h>
|
||||
#include <torch/nn/functional.h>
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <c10/cuda/CUDAGuard.h>
|
||||
#include <cuda_runtime_api.h>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <c10/cuda/CUDAGuard.h>
|
||||
|
||||
#include <cuda_fp8.h>
|
||||
|
||||
#include "cuda_utils.h"
|
||||
#include "../blackwell/block_config.h"
|
||||
|
||||
#define DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(pytorch_dtype, c_type, ...) \
|
||||
if (pytorch_dtype == at::ScalarType::Half) { \
|
||||
using c_type = half; \
|
||||
__VA_ARGS__ \
|
||||
} else if (pytorch_dtype == at::ScalarType::BFloat16) { \
|
||||
using c_type = nv_bfloat16; \
|
||||
__VA_ARGS__ \
|
||||
} else { \
|
||||
std::ostringstream oss; \
|
||||
oss << __PRETTY_FUNCTION__ << " failed to dispatch data type " << pytorch_dtype; \
|
||||
TORCH_CHECK(false, oss.str()); \
|
||||
}
|
||||
|
||||
#define DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, ...) \
|
||||
if (head_dim == 64) { \
|
||||
constexpr int HEAD_DIM = 64; \
|
||||
__VA_ARGS__ \
|
||||
} else if (head_dim == 128) { \
|
||||
constexpr int HEAD_DIM = 128; \
|
||||
__VA_ARGS__ \
|
||||
} else { \
|
||||
std::ostringstream err_msg; \
|
||||
err_msg << "Unsupported head dim: " << int(head_dim); \
|
||||
throw std::invalid_argument(err_msg.str()); \
|
||||
}
|
||||
|
||||
#define CHECK_CUDA(x) \
|
||||
TORCH_CHECK(x.is_cuda(), "Tensor " #x " must be on CUDA")
|
||||
#define CHECK_DTYPE(x, true_dtype) \
|
||||
TORCH_CHECK(x.dtype() == true_dtype, \
|
||||
"Tensor " #x " must have dtype (" #true_dtype ")")
|
||||
#define CHECK_DIMS(x, true_dim) \
|
||||
TORCH_CHECK(x.dim() == true_dim, \
|
||||
"Tensor " #x " must have dimension number (" #true_dim ")")
|
||||
#define CHECK_SHAPE(x, ...) \
|
||||
TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), \
|
||||
"Tensor " #x " must have shape (" #__VA_ARGS__ ")")
|
||||
#define CHECK_CONTIGUOUS(x) \
|
||||
TORCH_CHECK(x.is_contiguous(), "Tensor " #x " must be contiguous")
|
||||
#define CHECK_LASTDIM_CONTIGUOUS(x) \
|
||||
TORCH_CHECK(x.stride(-1) == 1, \
|
||||
"Tensor " #x " must be contiguous at the last dimension")
|
||||
|
||||
constexpr int CVT_FP4_ELTS_PER_THREAD = 16;
|
||||
|
||||
// Convert 4 float2 values into 8 e2m1 values (represented as one uint32_t).
|
||||
inline __device__ uint32_t fp32_vec_to_e2m1(float2 *array) {
|
||||
#if defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 1000)
|
||||
uint32_t val;
|
||||
asm volatile(
|
||||
"{\n"
|
||||
".reg .b8 byte0;\n"
|
||||
".reg .b8 byte1;\n"
|
||||
".reg .b8 byte2;\n"
|
||||
".reg .b8 byte3;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n"
|
||||
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n"
|
||||
"}"
|
||||
: "=r"(val)
|
||||
: "f"(array[0].x), "f"(array[0].y), "f"(array[1].x), "f"(array[1].y),
|
||||
"f"(array[2].x), "f"(array[2].y), "f"(array[3].x), "f"(array[3].y));
|
||||
return val;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
// Get type2 from type or vice versa (applied to half and bfloat16)
|
||||
template <typename T>
|
||||
struct TypeConverter {
|
||||
using Type = half2;
|
||||
}; // keep for generality
|
||||
|
||||
template <>
|
||||
struct TypeConverter<half2> {
|
||||
using Type = half;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct TypeConverter<half> {
|
||||
using Type = half2;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct TypeConverter<__nv_bfloat162> {
|
||||
using Type = __nv_bfloat16;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct TypeConverter<__nv_bfloat16> {
|
||||
using Type = __nv_bfloat162;
|
||||
};
|
||||
|
||||
// Define a 32 bytes packed data type.
|
||||
template <class Type>
|
||||
struct PackedVec {
|
||||
typename TypeConverter<Type>::Type elts[8];
|
||||
};
|
||||
|
||||
template <uint32_t head_dim, uint32_t BLOCK_SIZE, bool permute, typename T>
|
||||
__global__ void scaled_fp4_quant_kernel(
|
||||
const T* input, uint8_t* output, uint8_t* output_sf,
|
||||
int batch_size, int num_heads, int num_tokens,
|
||||
int stride_bz_input, int stride_h_input, int stride_seq_input,
|
||||
int stride_bz_output, int stride_h_output, int stride_seq_output,
|
||||
int stride_bz_output_sf, int stride_h_output_sf, int stride_seq_output_sf) {
|
||||
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
|
||||
using PackedVec = PackedVec<T>;
|
||||
|
||||
const int batch_id = blockIdx.y;
|
||||
const int head_id = blockIdx.z;
|
||||
const int token_block_id = blockIdx.x;
|
||||
|
||||
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
|
||||
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
|
||||
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
|
||||
"Vec size is not matched.");
|
||||
|
||||
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
|
||||
|
||||
// load input
|
||||
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
|
||||
|
||||
int load_token_id;
|
||||
if constexpr (!permute) {
|
||||
load_token_id = token_id;
|
||||
} else {
|
||||
int local_token_id = threadIdx.x / NUM_THREADS_PER_TOKEN;
|
||||
int local_token_id_residue = local_token_id % 32;
|
||||
// [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
|
||||
load_token_id = token_block_id * BLOCK_SIZE + (local_token_id / 32) * 32 +
|
||||
(local_token_id_residue / 8) * 2 +
|
||||
((local_token_id_residue % 8) / 2) * 8 +
|
||||
(local_token_id_residue % 8) % 2;
|
||||
}
|
||||
|
||||
PackedVec in_vec;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
|
||||
}
|
||||
|
||||
if (load_token_id < num_tokens) {
|
||||
in_vec = reinterpret_cast<PackedVec const*>(input +
|
||||
batch_id * stride_bz_input + // batch dim
|
||||
head_id * stride_h_input + // head dim
|
||||
load_token_id * stride_seq_input + // seq dim
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
|
||||
}
|
||||
|
||||
// calculate max of every consecutive 16 elements
|
||||
auto localMax = __habs2(in_vec.elts[0]);
|
||||
#pragma unroll
|
||||
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
|
||||
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
|
||||
}
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
|
||||
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
|
||||
}
|
||||
|
||||
float vecMax = float(__hmax(localMax.x, localMax.y));
|
||||
|
||||
// scaling factor
|
||||
float SFValue = vecMax / 6.0f;
|
||||
uint8_t SFValueFP8;
|
||||
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
|
||||
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
|
||||
|
||||
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
|
||||
|
||||
// convert input to float2 and apply scale
|
||||
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
if constexpr (std::is_same<T, half>::value) {
|
||||
fp2Vals[i] = __half22float2(in_vec.elts[i]);
|
||||
} else {
|
||||
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
|
||||
}
|
||||
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
|
||||
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
|
||||
}
|
||||
|
||||
// convert to e2m1
|
||||
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
|
||||
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
|
||||
}
|
||||
|
||||
// save, do not check range
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
|
||||
reinterpret_cast<uint32_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
token_id * stride_seq_output +
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = e2m1Vals[0];
|
||||
} else {
|
||||
reinterpret_cast<uint64_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
token_id * stride_seq_output +
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
|
||||
}
|
||||
|
||||
uint8_t* output_sf_save_base = output_sf + batch_id * stride_bz_output_sf + head_id * stride_h_output_sf + (token_id / 64) * 64 * stride_seq_output_sf;
|
||||
uint32_t token_id_local = token_id % 64;
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
|
||||
uint32_t col_id_local = threadIdx.x % NUM_THREADS_PER_TOKEN;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
} else {
|
||||
if (threadIdx.x % 2 == 0) {
|
||||
uint32_t col_id_local = (threadIdx.x % NUM_THREADS_PER_TOKEN) / 2;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <uint32_t head_dim, uint32_t BLOCK_SIZE, typename T>
|
||||
__global__ void scaled_fp4_quant_trans_kernel(
|
||||
const T* input, uint8_t* output, uint8_t* output_sf,
|
||||
int batch_size, int num_heads, int num_tokens,
|
||||
int stride_bz_input, int stride_h_input, int stride_seq_input,
|
||||
int stride_bz_output, int stride_h_output, int stride_d_output,
|
||||
int stride_bz_output_sf, int stride_h_output_sf, int stride_d_output_sf) {
|
||||
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
|
||||
using PackedVec = PackedVec<T>;
|
||||
|
||||
const int batch_id = blockIdx.y;
|
||||
const int head_id = blockIdx.z;
|
||||
const int token_block_id = blockIdx.x;
|
||||
|
||||
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
|
||||
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
|
||||
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
|
||||
"Vec size is not matched.");
|
||||
|
||||
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
|
||||
constexpr uint32_t NUM_THREADS_PER_SEQ = BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD;
|
||||
|
||||
// load input
|
||||
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
|
||||
|
||||
PackedVec in_vec;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
|
||||
}
|
||||
|
||||
if (token_id < num_tokens) {
|
||||
in_vec = reinterpret_cast<PackedVec const*>(input +
|
||||
batch_id * stride_bz_input + // batch dim
|
||||
head_id * stride_h_input + // head dim
|
||||
token_id * stride_seq_input + // seq dim
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
|
||||
}
|
||||
|
||||
// transpose
|
||||
__shared__ T shared_input[BLOCK_SIZE * head_dim];
|
||||
reinterpret_cast<PackedVec*>(shared_input)[threadIdx.x] = in_vec;
|
||||
__syncthreads();
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
in_vec.elts[i].x = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i) * head_dim];
|
||||
in_vec.elts[i].y = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i + 1) * head_dim];
|
||||
}
|
||||
|
||||
// calculate max of every consecutive 16 elements
|
||||
auto localMax = __habs2(in_vec.elts[0]);
|
||||
#pragma unroll
|
||||
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
|
||||
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
|
||||
}
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
|
||||
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
|
||||
}
|
||||
|
||||
float vecMax = float(__hmax(localMax.x, localMax.y));
|
||||
|
||||
// scaling factor
|
||||
float SFValue = vecMax / 6.0f;
|
||||
uint8_t SFValueFP8;
|
||||
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
|
||||
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
|
||||
|
||||
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
|
||||
|
||||
// convert input to float2 and apply scale
|
||||
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
if constexpr (std::is_same<T, half>::value) {
|
||||
fp2Vals[i] = __half22float2(in_vec.elts[i]);
|
||||
} else {
|
||||
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
|
||||
}
|
||||
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
|
||||
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
|
||||
}
|
||||
|
||||
// convert to e2m1
|
||||
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
|
||||
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
|
||||
}
|
||||
|
||||
// save
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
|
||||
reinterpret_cast<uint32_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
|
||||
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = e2m1Vals[0];
|
||||
} else {
|
||||
reinterpret_cast<uint64_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
|
||||
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
|
||||
}
|
||||
|
||||
uint8_t *output_sf_save_base = output_sf +
|
||||
batch_id * stride_bz_output_sf +
|
||||
head_id * stride_h_output_sf +
|
||||
(threadIdx.x / NUM_THREADS_PER_SEQ / 64) * 64 * stride_d_output_sf;
|
||||
uint32_t row_id_local = (threadIdx.x / NUM_THREADS_PER_SEQ) % 64;
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
|
||||
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + threadIdx.x % NUM_THREADS_PER_SEQ;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
} else {
|
||||
if (threadIdx.x % 2 == 0) {
|
||||
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + (threadIdx.x % NUM_THREADS_PER_SEQ) / 2;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void scaled_fp4_quant(torch::Tensor const& input,
|
||||
torch::Tensor const& output,
|
||||
torch::Tensor const& output_sf,
|
||||
int tensor_layout) {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
|
||||
CHECK_CUDA(input);
|
||||
CHECK_CUDA(output);
|
||||
CHECK_CUDA(output_sf);
|
||||
|
||||
CHECK_LASTDIM_CONTIGUOUS(input);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output_sf);
|
||||
|
||||
CHECK_DTYPE(output, at::ScalarType::Byte);
|
||||
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
|
||||
|
||||
CHECK_DIMS(input, 4);
|
||||
CHECK_DIMS(output, 4);
|
||||
CHECK_DIMS(output_sf, 4);
|
||||
|
||||
const int batch_size = input.size(0);
|
||||
const int head_dim = input.size(3);
|
||||
|
||||
const int stride_bz_input = input.stride(0);
|
||||
const int stride_bz_output = output.stride(0);
|
||||
const int stride_bz_output_sf = output_sf.stride(0);
|
||||
|
||||
int num_tokens, num_heads;
|
||||
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
|
||||
int stride_h_input, stride_h_output, stride_h_output_sf;
|
||||
if (tensor_layout == 0) {
|
||||
num_tokens = input.size(1);
|
||||
num_heads = input.size(2);
|
||||
stride_seq_input = input.stride(1);
|
||||
stride_seq_output = output.stride(1);
|
||||
stride_seq_output_sf = output_sf.stride(1);
|
||||
stride_h_input = input.stride(2);
|
||||
stride_h_output = output.stride(2);
|
||||
stride_h_output_sf = output_sf.stride(2);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_tokens, num_heads, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_tokens, num_heads, head_dim / 16);
|
||||
} else {
|
||||
num_tokens = input.size(2);
|
||||
num_heads = input.size(1);
|
||||
stride_seq_input = input.stride(2);
|
||||
stride_seq_output = output.stride(2);
|
||||
stride_seq_output_sf = output_sf.stride(2);
|
||||
stride_h_input = input.stride(1);
|
||||
stride_h_output = output.stride(1);
|
||||
stride_h_output_sf = output_sf.stride(1);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_heads, num_tokens, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_heads, num_tokens, head_dim / 16);
|
||||
}
|
||||
|
||||
auto input_dtype = input.scalar_type();
|
||||
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
|
||||
|
||||
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
|
||||
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
|
||||
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
|
||||
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
|
||||
|
||||
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, false, c_type>
|
||||
<<<grid, block, 0, stream>>>(
|
||||
reinterpret_cast<c_type*>(input.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
|
||||
batch_size, num_heads, num_tokens,
|
||||
stride_bz_input, stride_h_input, stride_seq_input,
|
||||
stride_bz_output, stride_h_output, stride_seq_output,
|
||||
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
void scaled_fp4_quant_permute(torch::Tensor const& input,
|
||||
torch::Tensor const& output,
|
||||
torch::Tensor const& output_sf,
|
||||
int tensor_layout) {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
|
||||
CHECK_CUDA(input);
|
||||
CHECK_CUDA(output);
|
||||
CHECK_CUDA(output_sf);
|
||||
|
||||
CHECK_LASTDIM_CONTIGUOUS(input);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output_sf);
|
||||
|
||||
CHECK_DTYPE(output, at::ScalarType::Byte);
|
||||
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
|
||||
|
||||
CHECK_DIMS(input, 4);
|
||||
CHECK_DIMS(output, 4);
|
||||
CHECK_DIMS(output_sf, 4);
|
||||
|
||||
const int batch_size = input.size(0);
|
||||
const int head_dim = input.size(3);
|
||||
|
||||
const int stride_bz_input = input.stride(0);
|
||||
const int stride_bz_output = output.stride(0);
|
||||
const int stride_bz_output_sf = output_sf.stride(0);
|
||||
|
||||
int num_tokens, num_heads;
|
||||
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
|
||||
int stride_h_input, stride_h_output, stride_h_output_sf;
|
||||
if (tensor_layout == 0) {
|
||||
num_tokens = input.size(1);
|
||||
num_heads = input.size(2);
|
||||
stride_seq_input = input.stride(1);
|
||||
stride_seq_output = output.stride(1);
|
||||
stride_seq_output_sf = output_sf.stride(1);
|
||||
stride_h_input = input.stride(2);
|
||||
stride_h_output = output.stride(2);
|
||||
stride_h_output_sf = output_sf.stride(2);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 16);
|
||||
} else {
|
||||
num_tokens = input.size(2);
|
||||
num_heads = input.size(1);
|
||||
stride_seq_input = input.stride(2);
|
||||
stride_seq_output = output.stride(2);
|
||||
stride_seq_output_sf = output_sf.stride(2);
|
||||
stride_h_input = input.stride(1);
|
||||
stride_h_output = output.stride(1);
|
||||
stride_h_output_sf = output_sf.stride(1);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 16);
|
||||
}
|
||||
|
||||
auto input_dtype = input.scalar_type();
|
||||
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
|
||||
|
||||
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
|
||||
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
|
||||
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
|
||||
|
||||
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, true, c_type>
|
||||
<<<grid, block, 0, stream>>>(
|
||||
reinterpret_cast<c_type*>(input.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
|
||||
batch_size, num_heads, num_tokens,
|
||||
stride_bz_input, stride_h_input, stride_seq_input,
|
||||
stride_bz_output, stride_h_output, stride_seq_output,
|
||||
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
void scaled_fp4_quant_trans(torch::Tensor const& input,
|
||||
torch::Tensor const& output,
|
||||
torch::Tensor const& output_sf,
|
||||
int tensor_layout) {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
|
||||
CHECK_CUDA(input);
|
||||
CHECK_CUDA(output);
|
||||
CHECK_CUDA(output_sf);
|
||||
|
||||
CHECK_LASTDIM_CONTIGUOUS(input);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output_sf);
|
||||
|
||||
CHECK_DTYPE(output, at::ScalarType::Byte);
|
||||
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
|
||||
|
||||
CHECK_DIMS(input, 4);
|
||||
CHECK_DIMS(output, 4);
|
||||
CHECK_DIMS(output_sf, 4);
|
||||
|
||||
const int batch_size = input.size(0);
|
||||
const int head_dim = input.size(3);
|
||||
|
||||
const int stride_bz_input = input.stride(0);
|
||||
const int stride_bz_output = output.stride(0);
|
||||
const int stride_bz_output_sf = output_sf.stride(0);
|
||||
|
||||
int num_tokens, num_heads;
|
||||
int stride_seq_input;
|
||||
int stride_d_output, stride_d_output_sf;
|
||||
int stride_h_input, stride_h_output, stride_h_output_sf;
|
||||
if (tensor_layout == 0) {
|
||||
num_tokens = input.size(1);
|
||||
num_heads = input.size(2);
|
||||
stride_seq_input = input.stride(1);
|
||||
stride_d_output = output.stride(1);
|
||||
stride_d_output_sf = output_sf.stride(1);
|
||||
stride_h_input = input.stride(2);
|
||||
stride_h_output = output.stride(2);
|
||||
stride_h_output_sf = output_sf.stride(2);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
|
||||
} else {
|
||||
num_tokens = input.size(2);
|
||||
num_heads = input.size(1);
|
||||
stride_seq_input = input.stride(2);
|
||||
stride_d_output = output.stride(2);
|
||||
stride_d_output_sf = output_sf.stride(2);
|
||||
stride_h_input = input.stride(1);
|
||||
stride_h_output = output.stride(1);
|
||||
stride_h_output_sf = output_sf.stride(1);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
|
||||
}
|
||||
|
||||
auto input_dtype = input.scalar_type();
|
||||
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
|
||||
|
||||
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
|
||||
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
|
||||
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
|
||||
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
|
||||
|
||||
scaled_fp4_quant_trans_kernel<HEAD_DIM, BLOCK_SIZE, c_type>
|
||||
<<<grid, block, 0, stream>>>(
|
||||
reinterpret_cast<c_type*>(input.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
|
||||
batch_size, num_heads, num_tokens,
|
||||
stride_bz_input, stride_h_input, stride_seq_input,
|
||||
stride_bz_output, stride_h_output, stride_d_output,
|
||||
stride_bz_output_sf, stride_h_output_sf, stride_d_output_sf);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
m.def("scaled_fp4_quant", &scaled_fp4_quant);
|
||||
m.def("scaled_fp4_quant_permute", &scaled_fp4_quant_permute);
|
||||
m.def("scaled_fp4_quant_trans", &scaled_fp4_quant_trans);
|
||||
}
|
||||
@@ -1,14 +0,0 @@
|
||||
"""Make local benchmark scripts runnable from common repo entrypoints."""
|
||||
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
|
||||
BENCHMARKS_DIR = Path(__file__).resolve().parent
|
||||
KERNEL_ROOT = BENCHMARKS_DIR.parent
|
||||
REPO_ROOT = KERNEL_ROOT.parent
|
||||
|
||||
for path in (KERNEL_ROOT, REPO_ROOT):
|
||||
path_str = str(path)
|
||||
if path_str not in sys.path:
|
||||
sys.path.insert(0, path_str)
|
||||
@@ -1,287 +0,0 @@
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import torch
|
||||
|
||||
from attn_qat_infer.api import (
|
||||
blockscaled_fp4_attn,
|
||||
preprocess_qkv,
|
||||
scale_and_quant_fp4,
|
||||
scale_and_quant_fp4_permute,
|
||||
scale_and_quant_fp4_transpose,
|
||||
)
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def benchmark_blockscaled_fp4_attn(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
per_block_mean=True, single_level_p_quant=False,
|
||||
num_warmups=100, num_tests=1000):
|
||||
"""
|
||||
Benchmark blockscaled_fp4_attn function (excluding quantization overhead).
|
||||
|
||||
This benchmarks ONLY the core FP4 attention kernel, with pre-quantized inputs.
|
||||
The quantization step is performed once before benchmarking.
|
||||
|
||||
Args:
|
||||
batch_size: Batch size
|
||||
num_heads: Number of attention heads
|
||||
seq_len: Sequence length (same for Q, K, V)
|
||||
head_dim: Head dimension
|
||||
is_causal: Whether to use causal masking
|
||||
dtype: Data type (torch.bfloat16 or torch.float16)
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization for P matrix
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
|
||||
Returns:
|
||||
dict with performance metrics
|
||||
"""
|
||||
device = 'cuda'
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
# Create input tensors
|
||||
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
|
||||
# Pre-process and quantize inputs (done once, not included in benchmark)
|
||||
is_bf16 = dtype == torch.bfloat16
|
||||
KL = k.size(2)
|
||||
q_processed, k_processed, v_processed, delta_s = preprocess_qkv(q, k, v, per_block_mean)
|
||||
qlist = scale_and_quant_fp4(q_processed)
|
||||
klist = scale_and_quant_fp4_permute(k_processed)
|
||||
vlist = scale_and_quant_fp4_transpose(v_processed)
|
||||
|
||||
# Synchronize to ensure quantization is complete
|
||||
torch.cuda.synchronize()
|
||||
|
||||
# Create closure for benchmarking (only the attention kernel)
|
||||
def run_attention():
|
||||
return blockscaled_fp4_attn(
|
||||
qlist, klist, vlist,
|
||||
delta_s,
|
||||
KL,
|
||||
is_causal=is_causal,
|
||||
per_block_mean=per_block_mean,
|
||||
is_bf16=is_bf16,
|
||||
single_level_p_quant=single_level_p_quant
|
||||
)
|
||||
|
||||
# Benchmark using the bench utility (handles warmup and timing)
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
avg_time_s = avg_time_ms / 1000.0
|
||||
|
||||
# Calculate FLOPs (use processed sequence length after padding)
|
||||
processed_seq_len = q_processed.size(2)
|
||||
total_flops = calculate_attention_flops(
|
||||
batch_size, num_heads, processed_seq_len, processed_seq_len, head_dim, is_causal
|
||||
)
|
||||
|
||||
# Calculate TFLOPs
|
||||
tflops = total_flops / (avg_time_s * 1e12)
|
||||
|
||||
# Calculate throughput (tokens/sec) - use original seq_len for meaningful metric
|
||||
tokens_per_second = (batch_size * seq_len) / avg_time_s
|
||||
|
||||
return {
|
||||
'batch_size': batch_size,
|
||||
'num_heads': num_heads,
|
||||
'seq_len': seq_len,
|
||||
'processed_seq_len': processed_seq_len,
|
||||
'head_dim': head_dim,
|
||||
'is_causal': is_causal,
|
||||
'dtype': str(dtype),
|
||||
'per_block_mean': per_block_mean,
|
||||
'single_level_p_quant': single_level_p_quant,
|
||||
'avg_time_ms': avg_time_ms,
|
||||
'avg_time_s': avg_time_s,
|
||||
'total_flops': total_flops,
|
||||
'tflops': tflops,
|
||||
'tokens_per_second': tokens_per_second,
|
||||
}
|
||||
|
||||
|
||||
def print_results(results):
|
||||
"""Print benchmark results in a formatted table."""
|
||||
print("\n" + "="*100)
|
||||
print("blockscaled_fp4_attn Benchmark Results (Kernel Only, No Quantization)")
|
||||
print("="*100)
|
||||
print(f"Configuration:")
|
||||
print(f" Batch Size: {results['batch_size']}")
|
||||
print(f" Num Heads: {results['num_heads']}")
|
||||
print(f" Sequence Length: {results['seq_len']}")
|
||||
print(f" Processed Seq Length: {results['processed_seq_len']} (after padding)")
|
||||
print(f" Head Dimension: {results['head_dim']}")
|
||||
print(f" Causal: {results['is_causal']}")
|
||||
print(f" Data Type: {results['dtype']}")
|
||||
print(f" Per Block Mean: {results['per_block_mean']}")
|
||||
print(f" Single Level P Quant: {results['single_level_p_quant']}")
|
||||
print(f"\nPerformance:")
|
||||
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
|
||||
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
|
||||
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
|
||||
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
|
||||
print("="*100 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def run_benchmark_suite():
|
||||
"""Run a comprehensive benchmark suite with various configurations."""
|
||||
print("Starting blockscaled_fp4_attn Benchmark Suite (Kernel Only)...")
|
||||
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}\n")
|
||||
print("Note: This benchmark measures only the FP4 attention kernel,")
|
||||
print(" excluding quantization overhead.\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
# Default configurations to test
|
||||
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
|
||||
configs = [
|
||||
(1, 16, 512, 64, False, torch.bfloat16),
|
||||
(1, 16, 1024, 64, False, torch.bfloat16),
|
||||
(1, 16, 2048, 64, False, torch.bfloat16),
|
||||
(1, 16, 4096, 64, False, torch.bfloat16),
|
||||
(1, 16, 8192, 64, False, torch.bfloat16),
|
||||
(1, 16, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 16, 512, 128, False, torch.bfloat16),
|
||||
(1, 16, 1024, 128, False, torch.bfloat16),
|
||||
(1, 16, 2048, 128, False, torch.bfloat16),
|
||||
(1, 16, 4096, 128, False, torch.bfloat16),
|
||||
(1, 16, 8192, 128, False, torch.bfloat16),
|
||||
(1, 16, 16384, 128, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 512, 64, False, torch.bfloat16),
|
||||
(1, 32, 1024, 64, False, torch.bfloat16),
|
||||
(1, 32, 2048, 64, False, torch.bfloat16),
|
||||
(1, 32, 4096, 64, False, torch.bfloat16),
|
||||
(1, 32, 8192, 64, False, torch.bfloat16),
|
||||
(1, 32, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 512, 128, False, torch.bfloat16),
|
||||
(1, 32, 1024, 128, False, torch.bfloat16),
|
||||
(1, 32, 2048, 128, False, torch.bfloat16),
|
||||
(1, 32, 4096, 128, False, torch.bfloat16),
|
||||
(1, 32, 8192, 128, False, torch.bfloat16),
|
||||
(1, 32, 16384, 128, False, torch.bfloat16),
|
||||
]
|
||||
|
||||
all_results = []
|
||||
|
||||
for config in configs:
|
||||
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
|
||||
|
||||
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
|
||||
f"Causal={is_causal}, dtype={dtype}...")
|
||||
sys.stdout.flush()
|
||||
|
||||
try:
|
||||
results = benchmark_blockscaled_fp4_attn(
|
||||
batch_size=batch_size,
|
||||
num_heads=num_heads,
|
||||
seq_len=seq_len,
|
||||
head_dim=head_dim,
|
||||
is_causal=is_causal,
|
||||
dtype=dtype,
|
||||
num_warmups=10,
|
||||
num_tests=50
|
||||
)
|
||||
|
||||
print_results(results)
|
||||
all_results.append(results)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error benchmarking configuration {config}:")
|
||||
print(f" Exception: {e}")
|
||||
traceback.print_exc()
|
||||
sys.stdout.flush()
|
||||
continue
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*120)
|
||||
print("Summary Table (blockscaled_fp4_attn Kernel Only)")
|
||||
print("="*120)
|
||||
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
|
||||
print("-"*120)
|
||||
|
||||
for r in all_results:
|
||||
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
|
||||
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
|
||||
f"{r['tokens_per_second']:<15,.0f}")
|
||||
|
||||
print("="*120 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description='Benchmark blockscaled_fp4_attn kernel (excluding quantization)')
|
||||
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
|
||||
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
|
||||
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
|
||||
help='Data type')
|
||||
parser.add_argument('--per-block-mean', action='store_true', default=True,
|
||||
help='Use per-block mean for Q smoothing (default: True)')
|
||||
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
|
||||
help='Disable per-block mean for Q smoothing')
|
||||
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
|
||||
help='Use single-level P quantization (default: False)')
|
||||
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
|
||||
help='Use two-level P quantization')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
dtype_map = {
|
||||
'bfloat16': torch.bfloat16,
|
||||
'float16': torch.float16
|
||||
}
|
||||
|
||||
if args.suite:
|
||||
run_benchmark_suite()
|
||||
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
|
||||
results = benchmark_blockscaled_fp4_attn(
|
||||
batch_size=args.batch_size,
|
||||
num_heads=args.num_heads,
|
||||
seq_len=args.seq_len,
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
dtype=dtype_map[args.dtype],
|
||||
per_block_mean=args.per_block_mean,
|
||||
single_level_p_quant=args.single_level_p_quant,
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests
|
||||
)
|
||||
print_results(results)
|
||||
else:
|
||||
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
|
||||
print(
|
||||
"Example: python benchmarks/benchmark_blockscaled_fp4_attn.py "
|
||||
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
|
||||
)
|
||||
print("\nNote: This benchmark measures only the FP4 attention kernel,")
|
||||
print(" excluding quantization overhead.")
|
||||
sys.stdout.flush()
|
||||
run_benchmark_suite()
|
||||
@@ -1,380 +0,0 @@
|
||||
import argparse
|
||||
import sys
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import matplotlib
|
||||
matplotlib.use('Agg') # Use non-interactive backend for server environments
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from flash_attn import flash_attn_func
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
# Import SageAttn components for direct control
|
||||
from attn_qat_infer.api import (
|
||||
preprocess_qkv,
|
||||
scale_and_quant_fp4,
|
||||
scale_and_quant_fp4_permute,
|
||||
scale_and_quant_fp4_transpose,
|
||||
blockscaled_fp4_attn
|
||||
)
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def sageattn_blackwell_configurable(q, k, v, is_causal=False, per_block_mean=True,
|
||||
single_level_p_quant=True,
|
||||
enable_smoothing_q=False, enable_smoothing_k=False):
|
||||
"""
|
||||
Configurable SageAttention3 Blackwell kernel with explicit smoothing control.
|
||||
|
||||
Args:
|
||||
q: Query tensor [B, H, L, D]
|
||||
k: Key tensor [B, H, L, D]
|
||||
v: Value tensor [B, H, L, D]
|
||||
is_causal: Whether to use causal masking
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization for P matrix
|
||||
enable_smoothing_q: Enable Q smoothing
|
||||
enable_smoothing_k: Enable K smoothing
|
||||
|
||||
Returns:
|
||||
Output tensor [B, H, L, D]
|
||||
"""
|
||||
QL = q.size(2)
|
||||
KL = k.size(2)
|
||||
is_bf16 = q.dtype == torch.bfloat16
|
||||
|
||||
# Preprocess with explicit smoothing control
|
||||
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean, enable_smoothing_q, enable_smoothing_k)
|
||||
|
||||
qlist_from_cuda = scale_and_quant_fp4(q)
|
||||
klist_from_cuda = scale_and_quant_fp4_permute(k)
|
||||
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
|
||||
|
||||
o_fp4 = blockscaled_fp4_attn(
|
||||
qlist_from_cuda,
|
||||
klist_from_cuda,
|
||||
vlist_from_cuda,
|
||||
delta_s,
|
||||
KL,
|
||||
is_causal,
|
||||
per_block_mean,
|
||||
is_bf16,
|
||||
single_level_p_quant
|
||||
)[0][:, :, :QL, :].contiguous()
|
||||
|
||||
return o_fp4
|
||||
|
||||
|
||||
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
num_warmups=10, num_tests=50):
|
||||
"""Benchmark FlashAttention2."""
|
||||
device = 'cuda'
|
||||
|
||||
# FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
|
||||
q = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
|
||||
|
||||
def run_attention():
|
||||
return flash_attn_func(q, k, v, causal=is_causal)
|
||||
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
return avg_time_ms
|
||||
|
||||
|
||||
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
per_block_mean=True, single_level_p_quant=False,
|
||||
enable_smoothing_q=True, enable_smoothing_k=True,
|
||||
num_warmups=10, num_tests=50):
|
||||
"""Benchmark SageAttention3 with configurable smoothing."""
|
||||
device = 'cuda'
|
||||
|
||||
# SageAttn expects (batch, num_heads, seq_len, head_dim)
|
||||
q = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
|
||||
|
||||
def run_attention():
|
||||
return sageattn_blackwell_configurable(
|
||||
q, k, v,
|
||||
is_causal=is_causal,
|
||||
per_block_mean=per_block_mean,
|
||||
single_level_p_quant=single_level_p_quant,
|
||||
enable_smoothing_q=enable_smoothing_q,
|
||||
enable_smoothing_k=enable_smoothing_k
|
||||
)
|
||||
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
return avg_time_ms
|
||||
|
||||
|
||||
def time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal=False):
|
||||
"""Convert time to TFLOPs (Tera FLOPs per Second)."""
|
||||
total_flops = calculate_attention_flops(batch_size, num_heads, seq_len, seq_len, head_dim, is_causal)
|
||||
time_s = time_ms / 1000.0
|
||||
tflops = total_flops / (time_s * 1e12)
|
||||
return tflops
|
||||
|
||||
|
||||
def run_benchmark_suite(head_dim=64, is_causal=False, num_heads=12, batch_size=1,
|
||||
num_warmups=10, num_tests=50,
|
||||
seq_lens=None, output_file="benchmark_attention.png"):
|
||||
"""
|
||||
Run comprehensive benchmark suite and generate plot.
|
||||
|
||||
Args:
|
||||
head_dim: Head dimension (64 or 128)
|
||||
is_causal: Whether to use causal attention
|
||||
num_heads: Number of attention heads
|
||||
batch_size: Batch size
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
seq_lens: List of sequence lengths to test
|
||||
output_file: Output plot filename
|
||||
"""
|
||||
if seq_lens is None:
|
||||
seq_lens = [1024, 2048, 4096, 8192, 16384, 32768]
|
||||
|
||||
device_name = torch.cuda.get_device_name(0)
|
||||
# Extract short name (e.g., "RTX5090" from full name)
|
||||
short_name = device_name.split()[-1] if 'RTX' in device_name or 'A100' in device_name else device_name[:20]
|
||||
|
||||
print(f"Starting Combined Attention Benchmark Suite...")
|
||||
print(f"CUDA Device: {device_name}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}")
|
||||
print(f"Head Dim: {head_dim}, Causal: {is_causal}, Num Heads: {num_heads}, Batch Size: {batch_size}")
|
||||
print("="*80)
|
||||
sys.stdout.flush()
|
||||
|
||||
# Results storage: {method_name: {seq_len: tflops}}
|
||||
results: Dict[str, Dict[int, Optional[float]]] = {
|
||||
'FlashAttn': {},
|
||||
'SageAttn3': {},
|
||||
'FP4': {},
|
||||
}
|
||||
|
||||
dtype = torch.bfloat16
|
||||
|
||||
for seq_len in seq_lens:
|
||||
print(f"\n--- Sequence Length: {seq_len} ---")
|
||||
sys.stdout.flush()
|
||||
|
||||
# FlashAttention2
|
||||
print(f" Benchmarking FlashAttn2...", end=" ")
|
||||
sys.stdout.flush()
|
||||
try:
|
||||
time_ms = benchmark_flashattn2(
|
||||
batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=is_causal, dtype=dtype,
|
||||
num_warmups=num_warmups, num_tests=num_tests
|
||||
)
|
||||
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
|
||||
results['FlashAttn'][seq_len] = tflops
|
||||
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
|
||||
except Exception as e:
|
||||
print(f"OOM or Error: {e}")
|
||||
results['FlashAttn'][seq_len] = None
|
||||
sys.stdout.flush()
|
||||
|
||||
# SageAttn3 (with smoothing: single_level_p_quant=False, enable_smoothing_q=True, enable_smoothing_k=True)
|
||||
print(f" Benchmarking SageAttn3 (smoothing ON)...", end=" ")
|
||||
sys.stdout.flush()
|
||||
try:
|
||||
time_ms = benchmark_sageattn3(
|
||||
batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=is_causal, dtype=dtype,
|
||||
per_block_mean=True,
|
||||
single_level_p_quant=False,
|
||||
enable_smoothing_q=True,
|
||||
enable_smoothing_k=True,
|
||||
num_warmups=num_warmups, num_tests=num_tests
|
||||
)
|
||||
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
|
||||
results['SageAttn3'][seq_len] = tflops
|
||||
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
|
||||
except Exception as e:
|
||||
print(f"OOM or Error: {e}")
|
||||
results['SageAttn3'][seq_len] = None
|
||||
sys.stdout.flush()
|
||||
|
||||
# FP4 (no smoothing: single_level_p_quant=True, enable_smoothing_q=False, enable_smoothing_k=False)
|
||||
print(f" Benchmarking FP4 (smoothing OFF)...", end=" ")
|
||||
sys.stdout.flush()
|
||||
try:
|
||||
time_ms = benchmark_sageattn3(
|
||||
batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=is_causal, dtype=dtype,
|
||||
per_block_mean=True,
|
||||
single_level_p_quant=True,
|
||||
enable_smoothing_q=False,
|
||||
enable_smoothing_k=False,
|
||||
num_warmups=num_warmups, num_tests=num_tests
|
||||
)
|
||||
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
|
||||
results['FP4'][seq_len] = tflops
|
||||
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
|
||||
except Exception as e:
|
||||
print(f"OOM or Error: {e}")
|
||||
results['FP4'][seq_len] = None
|
||||
sys.stdout.flush()
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*100)
|
||||
print("Summary Table (TFLOPs)")
|
||||
print("="*100)
|
||||
header = f"{'SeqLen':<10}"
|
||||
for method in results.keys():
|
||||
header += f"{method:<15}"
|
||||
print(header)
|
||||
print("-"*100)
|
||||
|
||||
for seq_len in seq_lens:
|
||||
row = f"{seq_len:<10}"
|
||||
for method in results.keys():
|
||||
val = results[method].get(seq_len)
|
||||
if val is not None:
|
||||
row += f"{val:<15.0f}"
|
||||
else:
|
||||
row += f"{'OOM':<15}"
|
||||
print(row)
|
||||
print("="*100)
|
||||
sys.stdout.flush()
|
||||
|
||||
# Generate plot
|
||||
generate_plot(results, seq_lens, head_dim, is_causal, short_name, output_file)
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def generate_plot(results: Dict[str, Dict[int, Optional[float]]],
|
||||
seq_lens: List[int],
|
||||
head_dim: int,
|
||||
is_causal: bool,
|
||||
device_name: str,
|
||||
output_file: str):
|
||||
"""Generate bar plot comparing attention implementations."""
|
||||
|
||||
# Prepare data
|
||||
methods = list(results.keys())
|
||||
x_labels = [f"{sl//1024}K" for sl in seq_lens]
|
||||
|
||||
# Colors for each method (red, blue, green scheme)
|
||||
colors = {
|
||||
'FlashAttn': '#1E90FF', # Blue (Dodger Blue)
|
||||
'SageAttn3': '#228B22', # Green (Forest Green)
|
||||
'FP4': '#DC143C', # Red (Crimson)
|
||||
}
|
||||
|
||||
# Number of methods and positions
|
||||
n_methods = len(methods)
|
||||
n_positions = len(seq_lens)
|
||||
|
||||
# Bar width and positions
|
||||
bar_width = 0.25
|
||||
x = np.arange(n_positions)
|
||||
|
||||
# Create figure
|
||||
fig, ax = plt.subplots(figsize=(12, 6))
|
||||
|
||||
# Plot bars for each method
|
||||
for i, method in enumerate(methods):
|
||||
values = []
|
||||
for seq_len in seq_lens:
|
||||
val = results[method].get(seq_len)
|
||||
values.append(val if val is not None else 0)
|
||||
|
||||
offset = (i - n_methods/2 + 0.5) * bar_width
|
||||
bars = ax.bar(x + offset, values, bar_width,
|
||||
label=method, color=colors.get(method, f'C{i}'),
|
||||
edgecolor='black', linewidth=0.5)
|
||||
|
||||
# Add value labels on top of bars
|
||||
for bar, val, seq_len in zip(bars, values, seq_lens):
|
||||
if results[method].get(seq_len) is None:
|
||||
label = 'OOM'
|
||||
else:
|
||||
label = f'{int(val)}'
|
||||
|
||||
height = bar.get_height()
|
||||
ax.annotate(label,
|
||||
xy=(bar.get_x() + bar.get_width() / 2, height),
|
||||
xytext=(0, 3), # 3 points vertical offset
|
||||
textcoords="offset points",
|
||||
ha='center', va='bottom',
|
||||
fontsize=8, rotation=0)
|
||||
|
||||
# Customize plot
|
||||
ax.set_xlabel('Sequence Length', fontsize=12, fontweight='bold')
|
||||
ax.set_ylabel('Speed (TFLOPs)', fontsize=12, fontweight='bold')
|
||||
ax.set_title(f'{device_name}, (Head dim = {head_dim}, causal = {is_causal})', fontsize=14, fontweight='bold')
|
||||
ax.set_xticks(x)
|
||||
ax.set_xticklabels(x_labels)
|
||||
legend = ax.legend(loc='upper left', ncol=len(methods), fontsize=10)
|
||||
# Make legend text bold
|
||||
for text in legend.get_texts():
|
||||
text.set_fontweight('bold')
|
||||
|
||||
# Set y-axis to start from 0
|
||||
ax.set_ylim(bottom=0)
|
||||
|
||||
# Add grid for readability
|
||||
ax.yaxis.grid(True, linestyle='--', alpha=0.7)
|
||||
ax.set_axisbelow(True)
|
||||
|
||||
# Tight layout
|
||||
plt.tight_layout()
|
||||
|
||||
# Save plot
|
||||
plt.savefig(output_file, dpi=150, bbox_inches='tight')
|
||||
print(f"\nPlot saved to: {output_file}")
|
||||
|
||||
# Also save as PDF for high quality
|
||||
pdf_file = output_file.rsplit('.', 1)[0] + '.pdf'
|
||||
plt.savefig(pdf_file, bbox_inches='tight')
|
||||
print(f"PDF saved to: {pdf_file}")
|
||||
|
||||
plt.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description='Combined Attention Benchmark (FlashAttn2 vs SageAttn3 vs FP4)')
|
||||
parser.add_argument('--batch-size', type=int, default=1, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=16, help='Number of attention heads')
|
||||
parser.add_argument('--head-dim', type=int, default=64, choices=[64, 128], help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--seq-lens', type=int, nargs='+',
|
||||
default=[1024, 2048, 4096, 8192, 16384, 32768],
|
||||
help='Sequence lengths to benchmark')
|
||||
parser.add_argument('--output', type=str, default='benchmark_attention.png',
|
||||
help='Output plot filename')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
run_benchmark_suite(
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
num_heads=args.num_heads,
|
||||
batch_size=args.batch_size,
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests,
|
||||
seq_lens=args.seq_lens,
|
||||
output_file=args.output
|
||||
)
|
||||
@@ -1,234 +0,0 @@
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import torch
|
||||
|
||||
from flash_attn import flash_attn_func
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
num_warmups=100, num_tests=1000):
|
||||
"""
|
||||
Benchmark FlashAttention2 and return performance metrics.
|
||||
|
||||
Args:
|
||||
batch_size: Batch size
|
||||
num_heads: Number of attention heads
|
||||
seq_len: Sequence length (same for Q, K, V)
|
||||
head_dim: Head dimension
|
||||
is_causal: Whether to use causal masking
|
||||
dtype: Data type (torch.bfloat16 or torch.float16)
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
|
||||
Returns:
|
||||
dict with performance metrics
|
||||
"""
|
||||
device = 'cuda'
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
# Create input tensors - FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
|
||||
q = torch.randn(batch_size, seq_len, num_heads, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, seq_len, num_heads, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, seq_len, num_heads, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
|
||||
# Create closure for benchmarking
|
||||
def run_attention():
|
||||
return flash_attn_func(
|
||||
q, k, v,
|
||||
causal=is_causal,
|
||||
)
|
||||
|
||||
# Benchmark using the bench utility (handles warmup and timing)
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
avg_time_s = avg_time_ms / 1000.0
|
||||
|
||||
# Calculate FLOPs
|
||||
total_flops = calculate_attention_flops(
|
||||
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
|
||||
)
|
||||
|
||||
# Calculate TFLOPs
|
||||
tflops = total_flops / (avg_time_s * 1e12)
|
||||
|
||||
# Calculate throughput (tokens/sec)
|
||||
tokens_per_second = (batch_size * seq_len) / avg_time_s
|
||||
|
||||
return {
|
||||
'batch_size': batch_size,
|
||||
'num_heads': num_heads,
|
||||
'seq_len': seq_len,
|
||||
'head_dim': head_dim,
|
||||
'is_causal': is_causal,
|
||||
'dtype': str(dtype),
|
||||
'avg_time_ms': avg_time_ms,
|
||||
'avg_time_s': avg_time_s,
|
||||
'total_flops': total_flops,
|
||||
'tflops': tflops,
|
||||
'tokens_per_second': tokens_per_second,
|
||||
}
|
||||
|
||||
|
||||
def print_results(results):
|
||||
"""Print benchmark results in a formatted table."""
|
||||
print("\n" + "="*100)
|
||||
print("FlashAttention2 Benchmark Results")
|
||||
print("="*100)
|
||||
print(f"Configuration:")
|
||||
print(f" Batch Size: {results['batch_size']}")
|
||||
print(f" Num Heads: {results['num_heads']}")
|
||||
print(f" Sequence Length: {results['seq_len']}")
|
||||
print(f" Head Dimension: {results['head_dim']}")
|
||||
print(f" Causal: {results['is_causal']}")
|
||||
print(f" Data Type: {results['dtype']}")
|
||||
print(f"\nPerformance:")
|
||||
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
|
||||
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
|
||||
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
|
||||
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
|
||||
print("="*100 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def run_benchmark_suite():
|
||||
"""Run a comprehensive benchmark suite with various configurations."""
|
||||
print("Starting FlashAttention2 Benchmark Suite...")
|
||||
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
# Default configurations to test
|
||||
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
|
||||
configs = [
|
||||
(1, 16, 1024, 64, False, torch.bfloat16),
|
||||
(1, 16, 2048, 64, False, torch.bfloat16),
|
||||
(1, 16, 4096, 64, False, torch.bfloat16),
|
||||
(1, 16, 8192, 64, False, torch.bfloat16),
|
||||
(1, 16, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 16, 1024, 128, False, torch.bfloat16),
|
||||
(1, 16, 2048, 128, False, torch.bfloat16),
|
||||
(1, 16, 4096, 128, False, torch.bfloat16),
|
||||
(1, 16, 8192, 128, False, torch.bfloat16),
|
||||
(1, 16, 16384, 128, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 64, False, torch.bfloat16),
|
||||
(1, 32, 2048, 64, False, torch.bfloat16),
|
||||
(1, 32, 4096, 64, False, torch.bfloat16),
|
||||
(1, 32, 8192, 64, False, torch.bfloat16),
|
||||
(1, 32, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 128, False, torch.bfloat16),
|
||||
(1, 32, 2048, 128, False, torch.bfloat16),
|
||||
(1, 32, 4096, 128, False, torch.bfloat16),
|
||||
(1, 32, 8192, 128, False, torch.bfloat16),
|
||||
(1, 32, 16384, 128, False, torch.bfloat16),
|
||||
]
|
||||
|
||||
all_results = []
|
||||
|
||||
for config in configs:
|
||||
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
|
||||
|
||||
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
|
||||
f"Causal={is_causal}, dtype={dtype}...")
|
||||
sys.stdout.flush()
|
||||
|
||||
try:
|
||||
results = benchmark_flashattn2(
|
||||
batch_size=batch_size,
|
||||
num_heads=num_heads,
|
||||
seq_len=seq_len,
|
||||
head_dim=head_dim,
|
||||
is_causal=is_causal,
|
||||
dtype=dtype,
|
||||
num_warmups=10,
|
||||
num_tests=50
|
||||
)
|
||||
|
||||
print_results(results)
|
||||
all_results.append(results)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error benchmarking configuration {config}:")
|
||||
print(f" Exception: {e}")
|
||||
traceback.print_exc()
|
||||
sys.stdout.flush()
|
||||
continue
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*120)
|
||||
print("Summary Table")
|
||||
print("="*120)
|
||||
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
|
||||
print("-"*120)
|
||||
|
||||
for r in all_results:
|
||||
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
|
||||
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
|
||||
f"{r['tokens_per_second']:<15,.0f}")
|
||||
|
||||
print("="*120 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description='Benchmark FlashAttention2 in TFLOPs')
|
||||
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
|
||||
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
|
||||
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
|
||||
help='Data type')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
dtype_map = {
|
||||
'bfloat16': torch.bfloat16,
|
||||
'float16': torch.float16
|
||||
}
|
||||
|
||||
if args.suite:
|
||||
run_benchmark_suite()
|
||||
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
|
||||
results = benchmark_flashattn2(
|
||||
batch_size=args.batch_size,
|
||||
num_heads=args.num_heads,
|
||||
seq_len=args.seq_len,
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
dtype=dtype_map[args.dtype],
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests
|
||||
)
|
||||
print_results(results)
|
||||
else:
|
||||
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
|
||||
print(
|
||||
"Example: python benchmarks/benchmark_flashattn2.py "
|
||||
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
|
||||
)
|
||||
sys.stdout.flush()
|
||||
run_benchmark_suite()
|
||||
@@ -1,253 +0,0 @@
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import torch
|
||||
|
||||
from attn_qat_infer.api import sageattn_blackwell
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
per_block_mean=True, single_level_p_quant=False,
|
||||
num_warmups=100, num_tests=1000):
|
||||
"""
|
||||
Benchmark SageAttention3 and return performance metrics.
|
||||
|
||||
Args:
|
||||
batch_size: Batch size
|
||||
num_heads: Number of attention heads
|
||||
seq_len: Sequence length (same for Q, K, V)
|
||||
head_dim: Head dimension
|
||||
is_causal: Whether to use causal masking
|
||||
dtype: Data type (torch.bfloat16 or torch.float16)
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization for P matrix
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
|
||||
Returns:
|
||||
dict with performance metrics
|
||||
"""
|
||||
device = 'cuda'
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
# Create input tensors
|
||||
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
|
||||
# Create closure for benchmarking (no extra stream needed - bench handles synchronization)
|
||||
def run_attention():
|
||||
return sageattn_blackwell(
|
||||
q, k, v,
|
||||
is_causal=is_causal,
|
||||
per_block_mean=per_block_mean,
|
||||
single_level_p_quant=single_level_p_quant
|
||||
)
|
||||
|
||||
# Benchmark using the bench utility (handles warmup and timing)
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
avg_time_s = avg_time_ms / 1000.0
|
||||
|
||||
# Calculate FLOPs
|
||||
total_flops = calculate_attention_flops(
|
||||
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
|
||||
)
|
||||
|
||||
# Calculate TFLOPs
|
||||
tflops = total_flops / (avg_time_s * 1e12)
|
||||
|
||||
# Calculate throughput (tokens/sec)
|
||||
tokens_per_second = (batch_size * seq_len) / avg_time_s
|
||||
|
||||
return {
|
||||
'batch_size': batch_size,
|
||||
'num_heads': num_heads,
|
||||
'seq_len': seq_len,
|
||||
'head_dim': head_dim,
|
||||
'is_causal': is_causal,
|
||||
'dtype': str(dtype),
|
||||
'per_block_mean': per_block_mean,
|
||||
'single_level_p_quant': single_level_p_quant,
|
||||
'avg_time_ms': avg_time_ms,
|
||||
'avg_time_s': avg_time_s,
|
||||
'total_flops': total_flops,
|
||||
'tflops': tflops,
|
||||
'tokens_per_second': tokens_per_second,
|
||||
}
|
||||
|
||||
|
||||
def print_results(results):
|
||||
"""Print benchmark results in a formatted table."""
|
||||
print("\n" + "="*100)
|
||||
print("SageAttention3 Benchmark Results")
|
||||
print("="*100)
|
||||
print(f"Configuration:")
|
||||
print(f" Batch Size: {results['batch_size']}")
|
||||
print(f" Num Heads: {results['num_heads']}")
|
||||
print(f" Sequence Length: {results['seq_len']}")
|
||||
print(f" Head Dimension: {results['head_dim']}")
|
||||
print(f" Causal: {results['is_causal']}")
|
||||
print(f" Data Type: {results['dtype']}")
|
||||
print(f" Per Block Mean: {results['per_block_mean']}")
|
||||
print(f" Single Level P Quant: {results['single_level_p_quant']}")
|
||||
print(f"\nPerformance:")
|
||||
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
|
||||
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
|
||||
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
|
||||
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
|
||||
print("="*100 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def run_benchmark_suite():
|
||||
"""Run a comprehensive benchmark suite with various configurations."""
|
||||
print("Starting SageAttention3 Benchmark Suite...")
|
||||
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
# Default configurations to test
|
||||
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
|
||||
configs = [
|
||||
(1, 16, 1024, 64, False, torch.bfloat16),
|
||||
(1, 16, 2048, 64, False, torch.bfloat16),
|
||||
(1, 16, 4096, 64, False, torch.bfloat16),
|
||||
(1, 16, 8192, 64, False, torch.bfloat16),
|
||||
(1, 16, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 16, 1024, 128, False, torch.bfloat16),
|
||||
(1, 16, 2048, 128, False, torch.bfloat16),
|
||||
(1, 16, 4096, 128, False, torch.bfloat16),
|
||||
(1, 16, 8192, 128, False, torch.bfloat16),
|
||||
(1, 16, 16384, 128, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 64, False, torch.bfloat16),
|
||||
(1, 32, 2048, 64, False, torch.bfloat16),
|
||||
(1, 32, 4096, 64, False, torch.bfloat16),
|
||||
(1, 32, 8192, 64, False, torch.bfloat16),
|
||||
(1, 32, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 128, False, torch.bfloat16),
|
||||
(1, 32, 2048, 128, False, torch.bfloat16),
|
||||
(1, 32, 4096, 128, False, torch.bfloat16),
|
||||
(1, 32, 8192, 128, False, torch.bfloat16),
|
||||
(1, 32, 16384, 128, False, torch.bfloat16),
|
||||
]
|
||||
|
||||
all_results = []
|
||||
|
||||
for config in configs:
|
||||
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
|
||||
|
||||
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
|
||||
f"Causal={is_causal}, dtype={dtype}...")
|
||||
sys.stdout.flush()
|
||||
|
||||
try:
|
||||
results = benchmark_sageattn3(
|
||||
batch_size=batch_size,
|
||||
num_heads=num_heads,
|
||||
seq_len=seq_len,
|
||||
head_dim=head_dim,
|
||||
is_causal=is_causal,
|
||||
dtype=dtype,
|
||||
num_warmups=10,
|
||||
num_tests=50
|
||||
)
|
||||
|
||||
print_results(results)
|
||||
all_results.append(results)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error benchmarking configuration {config}:")
|
||||
print(f" Exception: {e}")
|
||||
traceback.print_exc()
|
||||
sys.stdout.flush()
|
||||
continue
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*120)
|
||||
print("Summary Table")
|
||||
print("="*120)
|
||||
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
|
||||
print("-"*120)
|
||||
|
||||
for r in all_results:
|
||||
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
|
||||
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
|
||||
f"{r['tokens_per_second']:<15,.0f}")
|
||||
|
||||
print("="*120 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description='Benchmark SageAttention3 in TFLOPs')
|
||||
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
|
||||
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
|
||||
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
|
||||
help='Data type')
|
||||
parser.add_argument('--per-block-mean', action='store_true', default=True,
|
||||
help='Use per-block mean for Q smoothing (default: True)')
|
||||
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
|
||||
help='Disable per-block mean for Q smoothing')
|
||||
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
|
||||
help='Use single-level P quantization (default: True)')
|
||||
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
|
||||
help='Use two-level P quantization')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
dtype_map = {
|
||||
'bfloat16': torch.bfloat16,
|
||||
'float16': torch.float16
|
||||
}
|
||||
|
||||
if args.suite:
|
||||
run_benchmark_suite()
|
||||
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
|
||||
results = benchmark_sageattn3(
|
||||
batch_size=args.batch_size,
|
||||
num_heads=args.num_heads,
|
||||
seq_len=args.seq_len,
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
dtype=dtype_map[args.dtype],
|
||||
per_block_mean=args.per_block_mean,
|
||||
single_level_p_quant=args.single_level_p_quant,
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests
|
||||
)
|
||||
print_results(results)
|
||||
else:
|
||||
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
|
||||
print(
|
||||
"Example: python benchmarks/benchmark_sageattn3.py "
|
||||
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
|
||||
)
|
||||
sys.stdout.flush()
|
||||
run_benchmark_suite()
|
||||
@@ -10,6 +10,35 @@ set -ex
|
||||
|
||||
echo "Building fastvideo-kernel..."
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Neutralise conda-injected compiler toolchains.
|
||||
#
|
||||
# Conda compiler packages (gcc_linux-aarch64, gxx_linux-64, etc.) set
|
||||
# CMAKE_ARGS, CFLAGS, CXXFLAGS, and LDFLAGS on activation. When multiple
|
||||
# toolchains are installed the variables can reference a *cross*-compiler
|
||||
# that doesn't match the host (e.g. aarch64-conda-linux-gnu-c++ on x86_64).
|
||||
# Even when the correct toolchain is active, the flags it injects
|
||||
# (-march=nocona, -mtune=haswell, …) can conflict with nvcc's host-compiler
|
||||
# expectations. Clear them so CMake discovers the system compiler instead.
|
||||
# ---------------------------------------------------------------------------
|
||||
if [[ -n "${CONDA_PREFIX:-}" ]]; then
|
||||
_need_clean=0
|
||||
# Detect conda cross-compiler that doesn't match the host.
|
||||
_host_arch="$(uname -m)"
|
||||
if [[ "${CXX:-}" == *"conda"* ]] || [[ "${CC:-}" == *"conda"* ]]; then
|
||||
_need_clean=1
|
||||
fi
|
||||
if [[ "${CMAKE_ARGS:-}" == *"conda"* ]]; then
|
||||
_need_clean=1
|
||||
fi
|
||||
if (( _need_clean )); then
|
||||
echo "NOTE: Clearing conda-injected compiler settings (CC/CXX/CMAKE_ARGS/CFLAGS/...)"
|
||||
echo " to use the system compiler for CUDA extension builds."
|
||||
unset CC CXX CMAKE_ARGS CFLAGS CXXFLAGS LDFLAGS
|
||||
fi
|
||||
unset _need_clean _host_arch
|
||||
fi
|
||||
|
||||
# Ensure submodules are initialized if needed (tk)
|
||||
git submodule update --init --recursive
|
||||
|
||||
@@ -32,7 +61,16 @@ has_cmake_arg() {
|
||||
}
|
||||
|
||||
detect_with_torch() {
|
||||
uv run --active --no-project python -c "import torch
|
||||
# Prefer the active venv's python directly over `uv run --active --no-project`,
|
||||
# which on some uv versions provisions its own interpreter and misses packages
|
||||
# installed into VIRTUAL_ENV.
|
||||
local py
|
||||
if [[ -n "${VIRTUAL_ENV:-}" && -x "${VIRTUAL_ENV}/bin/python" ]]; then
|
||||
py="${VIRTUAL_ENV}/bin/python"
|
||||
else
|
||||
py="$(command -v python3 || command -v python)"
|
||||
fi
|
||||
"${py}" -c "import torch
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError('torch.cuda.is_available() is false')
|
||||
mj, mn = torch.cuda.get_device_capability(0)
|
||||
|
||||
@@ -32,4 +32,4 @@ dependencies = [
|
||||
[tool.scikit-build]
|
||||
cmake.build-type = "Release"
|
||||
minimum-version = "build-system.requires"
|
||||
wheel.packages = ["python/fastvideo_kernel", "attn_qat_infer"]
|
||||
wheel.packages = ["python/fastvideo_kernel"]
|
||||
|
||||
@@ -5,6 +5,11 @@ from fastvideo_kernel.ops import (
|
||||
video_sparse_attn,
|
||||
)
|
||||
|
||||
from fastvideo_kernel.block_sparse_attn import (
|
||||
block_sparse_attn,
|
||||
block_sparse_attn_from_indices,
|
||||
)
|
||||
|
||||
from fastvideo_kernel.vmoba import (
|
||||
moba_attn_varlen,
|
||||
process_moba_input,
|
||||
@@ -22,6 +27,8 @@ from fastvideo_kernel.turbodiffusion_ops import (
|
||||
__all__ = [
|
||||
"sliding_tile_attention",
|
||||
"video_sparse_attn",
|
||||
"block_sparse_attn",
|
||||
"block_sparse_attn_from_indices",
|
||||
"moba_attn_varlen",
|
||||
"process_moba_input",
|
||||
"process_moba_output",
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
"""Autograd-enabled block-sparse attention. Index-native ops with a bool-mask compat shim."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
@@ -6,6 +8,11 @@ from typing import Tuple
|
||||
import torch
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Backend selection helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _get_sm90_ops():
|
||||
try:
|
||||
from fastvideo_kernel._C import fastvideo_kernel_ops # type: ignore
|
||||
@@ -25,38 +32,66 @@ def _is_sm90() -> bool:
|
||||
|
||||
|
||||
def _force_triton() -> bool:
|
||||
# Force Triton even on SM90 and even if the compiled extension is available.
|
||||
# Useful for CI / debugging / parity testing.
|
||||
return os.environ.get("FASTVIDEO_KERNEL_VSA_FORCE_TRITON", "0") == "1"
|
||||
|
||||
|
||||
def _map_to_index(block_map: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
"""
|
||||
Preferred map->index conversion used by the wrapper.
|
||||
# ---------------------------------------------------------------------------
|
||||
# Index helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
This wrapper **requires** the Triton implementation.
|
||||
If Triton (or the Triton map_to_index module) is not available, it raises.
|
||||
"""
|
||||
|
||||
def _map_to_index(block_map: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
"""Compact a bool block_map to (q2k_idx, q2k_num). Legacy path only."""
|
||||
if block_map.dim() == 3:
|
||||
block_map = block_map.unsqueeze(0)
|
||||
if block_map.dim() != 4:
|
||||
raise ValueError(f"block_map must be [B,H,Q,KV] (or [H,Q,KV]), got shape={tuple(block_map.shape)}")
|
||||
raise ValueError(
|
||||
f"block_map must be [B,H,Q,KV] (or [H,Q,KV]), "
|
||||
f"got shape={tuple(block_map.shape)}"
|
||||
)
|
||||
if block_map.dtype != torch.bool:
|
||||
block_map = block_map.to(torch.bool)
|
||||
|
||||
if not block_map.is_cuda:
|
||||
raise RuntimeError("block_map must be a CUDA tensor (Triton map_to_index required).")
|
||||
raise RuntimeError(
|
||||
"block_map must be a CUDA tensor (Triton map_to_index required)."
|
||||
)
|
||||
|
||||
try:
|
||||
from fastvideo_kernel.triton_kernels.index import map_to_index as triton_map_to_index # local import
|
||||
except Exception as e:
|
||||
from fastvideo_kernel.triton_kernels.index import map_to_index as triton_map_to_index
|
||||
except Exception as e: # pragma: no cover - environment issue
|
||||
raise ImportError(
|
||||
"Triton map_to_index is required but not available. "
|
||||
"Ensure Triton is installed and fastvideo_kernel.triton_kernels.index is importable."
|
||||
"Ensure Triton is installed and "
|
||||
"fastvideo_kernel.triton_kernels.index is importable."
|
||||
) from e
|
||||
return triton_map_to_index(block_map)
|
||||
|
||||
|
||||
def _invert_indices_for_backward(
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
num_kv_blocks: int,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
from fastvideo_kernel.triton_kernels.index import invert_indices
|
||||
return invert_indices(q2k_idx, q2k_num, num_kv_blocks=num_kv_blocks)
|
||||
|
||||
|
||||
def _as_int32_contig(t: torch.Tensor, name: str) -> torch.Tensor:
|
||||
"""Return `t` as a contiguous int32 tensor, raising a clear error on CPU input."""
|
||||
if not t.is_cuda:
|
||||
raise RuntimeError(f"{name} must be a CUDA tensor, got device={t.device}")
|
||||
if t.dtype != torch.int32:
|
||||
t = t.to(torch.int32)
|
||||
if not t.is_contiguous():
|
||||
t = t.contiguous()
|
||||
return t
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Triton backend custom ops (index-native)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@torch.library.custom_op(
|
||||
"fastvideo_kernel::block_sparse_attn_triton",
|
||||
mutates_args=(),
|
||||
@@ -66,34 +101,40 @@ def block_sparse_attn_triton(
|
||||
q: torch.Tensor,
|
||||
k: torch.Tensor,
|
||||
v: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
q = q.contiguous()
|
||||
k = k.contiguous()
|
||||
v = v.contiguous()
|
||||
block_map = block_map.to(torch.bool)
|
||||
q2k_idx, q2k_num = _map_to_index(block_map)
|
||||
|
||||
from fastvideo_kernel.triton_kernels.block_sparse_attn_triton import ( # local import
|
||||
from fastvideo_kernel.triton_kernels.block_sparse_attn_triton import (
|
||||
triton_block_sparse_attn_forward,
|
||||
)
|
||||
|
||||
o, M = triton_block_sparse_attn_forward(q, k, v, q2k_idx, q2k_num, variable_block_sizes)
|
||||
o, M = triton_block_sparse_attn_forward(
|
||||
q.contiguous(),
|
||||
k.contiguous(),
|
||||
v.contiguous(),
|
||||
q2k_idx,
|
||||
q2k_num,
|
||||
variable_block_sizes,
|
||||
)
|
||||
return o, M
|
||||
|
||||
|
||||
|
||||
@torch.library.register_fake("fastvideo_kernel::block_sparse_attn_triton")
|
||||
def _block_sparse_attn_triton_fake(
|
||||
q: torch.Tensor,
|
||||
k: torch.Tensor,
|
||||
v: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
o = torch.empty_like(q)
|
||||
M = torch.empty((q.shape[0], q.shape[1], q.shape[2]), device=q.device, dtype=torch.float32)
|
||||
M = torch.empty(
|
||||
(q.shape[0], q.shape[1], q.shape[2]),
|
||||
device=q.device,
|
||||
dtype=torch.float32,
|
||||
)
|
||||
return o, M
|
||||
|
||||
|
||||
@@ -109,20 +150,32 @@ def block_sparse_attn_backward_triton(
|
||||
v: torch.Tensor,
|
||||
o: torch.Tensor,
|
||||
M: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
grad_output = grad_output.contiguous()
|
||||
block_map = block_map.to(torch.bool)
|
||||
q2k_idx, q2k_num = _map_to_index(block_map)
|
||||
k2q_idx, k2q_num = _map_to_index(block_map.transpose(-1, -2).contiguous())
|
||||
|
||||
from fastvideo_kernel.triton_kernels.block_sparse_attn_triton import ( # local import
|
||||
from fastvideo_kernel.triton_kernels.block_sparse_attn_triton import (
|
||||
triton_block_sparse_attn_backward,
|
||||
)
|
||||
|
||||
num_kv_blocks = int(variable_block_sizes.numel())
|
||||
k2q_idx, k2q_num = _invert_indices_for_backward(
|
||||
q2k_idx, q2k_num, num_kv_blocks
|
||||
)
|
||||
# q/k/v are saved from the user-facing inputs and may be non-contiguous;
|
||||
# o/M are kernel outputs so are already contiguous.
|
||||
dq, dk, dv = triton_block_sparse_attn_backward(
|
||||
grad_output, q, k, v, o, M, q2k_idx, q2k_num, k2q_idx, k2q_num, variable_block_sizes
|
||||
grad_output.contiguous(),
|
||||
q.contiguous(),
|
||||
k.contiguous(),
|
||||
v.contiguous(),
|
||||
o,
|
||||
M,
|
||||
q2k_idx,
|
||||
q2k_num,
|
||||
k2q_idx,
|
||||
k2q_num,
|
||||
variable_block_sizes,
|
||||
)
|
||||
return dq, dk, dv
|
||||
|
||||
@@ -135,7 +188,8 @@ def _block_sparse_attn_backward_triton_fake(
|
||||
v: torch.Tensor,
|
||||
o: torch.Tensor,
|
||||
M: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
dq = torch.empty_like(q)
|
||||
@@ -144,19 +198,28 @@ def _block_sparse_attn_backward_triton_fake(
|
||||
return dq, dk, dv
|
||||
|
||||
|
||||
def _backward_triton(ctx, grad_o, grad_M):
|
||||
q, k, v, o, M, block_map, variable_block_sizes = ctx.saved_tensors
|
||||
dq, dk, dv = block_sparse_attn_backward_triton(grad_o, q, k, v, o, M, block_map, variable_block_sizes)
|
||||
return dq, dk, dv, None, None
|
||||
|
||||
|
||||
def _setup_context_triton(ctx, inputs, output):
|
||||
q, k, v, block_map, variable_block_sizes = inputs
|
||||
q, k, v, q2k_idx, q2k_num, variable_block_sizes = inputs
|
||||
o, M = output
|
||||
ctx.save_for_backward(q, k, v, o, M, block_map, variable_block_sizes)
|
||||
ctx.save_for_backward(q, k, v, o, M, q2k_idx, q2k_num, variable_block_sizes)
|
||||
|
||||
|
||||
block_sparse_attn_triton.register_autograd(_backward_triton, setup_context=_setup_context_triton)
|
||||
def _backward_triton(ctx, grad_o, grad_M):
|
||||
q, k, v, o, M, q2k_idx, q2k_num, variable_block_sizes = ctx.saved_tensors
|
||||
dq, dk, dv = block_sparse_attn_backward_triton(
|
||||
grad_o, q, k, v, o, M, q2k_idx, q2k_num, variable_block_sizes
|
||||
)
|
||||
return dq, dk, dv, None, None, None
|
||||
|
||||
|
||||
block_sparse_attn_triton.register_autograd(
|
||||
_backward_triton, setup_context=_setup_context_triton
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# SM90 backend custom ops (index-native)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@torch.library.custom_op(
|
||||
@@ -168,21 +231,21 @@ def block_sparse_attn_sm90(
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
block_sparse_fwd, _ = _get_sm90_ops()
|
||||
if block_sparse_fwd is None:
|
||||
raise ImportError("fastvideo_kernel_ops.block_sparse_fwd is not available")
|
||||
|
||||
q_padded = q_padded.contiguous()
|
||||
k_padded = k_padded.contiguous()
|
||||
v_padded = v_padded.contiguous()
|
||||
block_map = block_map.to(torch.bool)
|
||||
q2k_idx, q2k_num = _map_to_index(block_map)
|
||||
|
||||
o_padded, lse_padded = block_sparse_fwd(
|
||||
q_padded, k_padded, v_padded, q2k_idx, q2k_num, variable_block_sizes.int()
|
||||
q_padded.contiguous(),
|
||||
k_padded.contiguous(),
|
||||
v_padded.contiguous(),
|
||||
q2k_idx,
|
||||
q2k_num,
|
||||
variable_block_sizes,
|
||||
)
|
||||
return o_padded, lse_padded
|
||||
|
||||
@@ -192,11 +255,16 @@ def _block_sparse_attn_sm90_fake(
|
||||
q_padded: torch.Tensor,
|
||||
k_padded: torch.Tensor,
|
||||
v_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
o = torch.empty_like(q_padded)
|
||||
lse = torch.empty((q_padded.shape[0], q_padded.shape[1], q_padded.shape[2], 1), device=q_padded.device, dtype=torch.float32)
|
||||
lse = torch.empty(
|
||||
(q_padded.shape[0], q_padded.shape[1], q_padded.shape[2], 1),
|
||||
device=q_padded.device,
|
||||
dtype=torch.float32,
|
||||
)
|
||||
return o, lse
|
||||
|
||||
|
||||
@@ -212,30 +280,34 @@ def block_sparse_attn_backward_sm90(
|
||||
v_padded: torch.Tensor,
|
||||
o_padded: torch.Tensor,
|
||||
lse_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
_, block_sparse_bwd = _get_sm90_ops()
|
||||
if block_sparse_bwd is None:
|
||||
raise ImportError("fastvideo_kernel_ops.block_sparse_bwd is not available")
|
||||
|
||||
grad_output_padded = grad_output_padded.contiguous()
|
||||
block_map = block_map.to(torch.bool)
|
||||
k2q_idx, k2q_num = _map_to_index(block_map.transpose(-1, -2).contiguous())
|
||||
num_kv_blocks = int(variable_block_sizes.numel())
|
||||
k2q_idx, k2q_num = _invert_indices_for_backward(
|
||||
q2k_idx, q2k_num, num_kv_blocks
|
||||
)
|
||||
|
||||
# q/k/v are saved from user-facing inputs; o/lse are kernel outputs.
|
||||
dq, dk, dv = block_sparse_bwd(
|
||||
q_padded,
|
||||
k_padded,
|
||||
v_padded,
|
||||
q_padded.contiguous(),
|
||||
k_padded.contiguous(),
|
||||
v_padded.contiguous(),
|
||||
o_padded,
|
||||
lse_padded,
|
||||
grad_output_padded,
|
||||
grad_output_padded.contiguous(),
|
||||
k2q_idx,
|
||||
k2q_num,
|
||||
variable_block_sizes.int(),
|
||||
variable_block_sizes,
|
||||
)
|
||||
# C++ kernel returns fp32 grads; cast back to match PyTorch convention if needed
|
||||
return dq.to(grad_output_padded.dtype), dk.to(grad_output_padded.dtype), dv.to(grad_output_padded.dtype)
|
||||
# C++ kernel returns fp32 grads; cast back to the input dtype.
|
||||
out_dtype = grad_output_padded.dtype
|
||||
return dq.to(out_dtype), dk.to(out_dtype), dv.to(out_dtype)
|
||||
|
||||
|
||||
@torch.library.register_fake("fastvideo_kernel::block_sparse_attn_backward_sm90")
|
||||
@@ -246,7 +318,8 @@ def _block_sparse_attn_backward_sm90_fake(
|
||||
v_padded: torch.Tensor,
|
||||
o_padded: torch.Tensor,
|
||||
lse_padded: torch.Tensor,
|
||||
block_map: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
dq = torch.empty_like(q_padded)
|
||||
@@ -255,21 +328,57 @@ def _block_sparse_attn_backward_sm90_fake(
|
||||
return dq, dk, dv
|
||||
|
||||
|
||||
def _backward_sm90(ctx, grad_o, grad_lse):
|
||||
q, k, v, o, lse, block_map, variable_block_sizes = ctx.saved_tensors
|
||||
dq, dk, dv = block_sparse_attn_backward_sm90(
|
||||
grad_o, q, k, v, o, lse, block_map, variable_block_sizes
|
||||
)
|
||||
return dq, dk, dv, None, None
|
||||
|
||||
|
||||
def _setup_context_sm90(ctx, inputs, output):
|
||||
q, k, v, block_map, variable_block_sizes = inputs
|
||||
q, k, v, q2k_idx, q2k_num, variable_block_sizes = inputs
|
||||
o, lse = output
|
||||
ctx.save_for_backward(q, k, v, o, lse, block_map, variable_block_sizes)
|
||||
ctx.save_for_backward(q, k, v, o, lse, q2k_idx, q2k_num, variable_block_sizes)
|
||||
|
||||
|
||||
block_sparse_attn_sm90.register_autograd(_backward_sm90, setup_context=_setup_context_sm90)
|
||||
def _backward_sm90(ctx, grad_o, grad_lse):
|
||||
q, k, v, o, lse, q2k_idx, q2k_num, variable_block_sizes = ctx.saved_tensors
|
||||
dq, dk, dv = block_sparse_attn_backward_sm90(
|
||||
grad_o, q, k, v, o, lse, q2k_idx, q2k_num, variable_block_sizes
|
||||
)
|
||||
return dq, dk, dv, None, None, None
|
||||
|
||||
|
||||
block_sparse_attn_sm90.register_autograd(
|
||||
_backward_sm90, setup_context=_setup_context_sm90
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Public API
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def block_sparse_attn_from_indices(
|
||||
q: torch.Tensor,
|
||||
k: torch.Tensor,
|
||||
v: torch.Tensor,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
"""Block-sparse attention with autograd, taking compact per-row KV indices."""
|
||||
# Normalize index tensors once at the public boundary so the custom ops
|
||||
# and their fakes can assume int32/contiguous. No-op on well-formed input.
|
||||
q2k_idx = _as_int32_contig(q2k_idx, "q2k_idx")
|
||||
q2k_num = _as_int32_contig(q2k_num, "q2k_num")
|
||||
variable_block_sizes = _as_int32_contig(variable_block_sizes, "variable_block_sizes")
|
||||
|
||||
block_sparse_fwd, block_sparse_bwd = _get_sm90_ops()
|
||||
use_sm90 = (
|
||||
(not _force_triton())
|
||||
and _is_sm90()
|
||||
and block_sparse_fwd is not None
|
||||
and block_sparse_bwd is not None
|
||||
)
|
||||
if use_sm90:
|
||||
return block_sparse_attn_sm90(q, k, v, q2k_idx, q2k_num, variable_block_sizes)
|
||||
# Triton path: supports q_seq_len != kv_seq_len as long as both are padded
|
||||
# to a multiple of the block size (64 tokens).
|
||||
return block_sparse_attn_triton(q, k, v, q2k_idx, q2k_num, variable_block_sizes)
|
||||
|
||||
|
||||
def block_sparse_attn(
|
||||
@@ -279,16 +388,8 @@ def block_sparse_attn(
|
||||
block_map: torch.Tensor,
|
||||
variable_block_sizes: torch.Tensor,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
"""
|
||||
Unified block-sparse attention op with autograd support.
|
||||
- On SM90 with compiled extension present: uses fastvideo_kernel_ops.block_sparse_fwd/bwd.
|
||||
- Otherwise: uses Triton implementation (requires q/k/v to have same padded length today).
|
||||
"""
|
||||
block_sparse_fwd, block_sparse_bwd = _get_sm90_ops()
|
||||
if (not _force_triton()) and _is_sm90() and (block_sparse_fwd is not None) and (block_sparse_bwd is not None):
|
||||
return block_sparse_attn_sm90(q, k, v, block_map, variable_block_sizes)
|
||||
# Triton path: supports q_seq_len != kv_seq_len as long as both are padded
|
||||
# to a multiple of the block size (64 tokens).
|
||||
return block_sparse_attn_triton(q, k, v, block_map, variable_block_sizes)
|
||||
|
||||
|
||||
"""Bool-mask compat wrapper; prefer block_sparse_attn_from_indices."""
|
||||
q2k_idx, q2k_num = _map_to_index(block_map)
|
||||
return block_sparse_attn_from_indices(
|
||||
q, k, v, q2k_idx, q2k_num, variable_block_sizes
|
||||
)
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user