Compare commits

..
Author SHA1 Message Date
Jinzhe Pan 3a7925c674 [ci] Add /test pre-commit slash command
- /test pre-commit triggers pre-commit via workflow_call
- Posts commit status to PR head SHA so check appears on PR page
- Mergify check-success~=pre-commit condition can now be satisfied
2026-03-31 08:18:06 -04:00
276 changed files with 454 additions and 35980 deletions
+2 -186
View File
@@ -9,183 +9,11 @@ notify:
- github_commit_status:
context: "full-suite-passed"
if: build.env("TEST_SCOPE") == "full"
- github_commit_status:
context: "direct-test-completed"
if: build.env("TEST_SCOPE") == "direct"
steps:
# ============================================================
# Direct test: triggered by /test <name> slash command.
# Labels match fastcheck/full-suite counterparts so the GitHub
# check status overwrites the original failed check.
# Only ONE step executes per build (gated by TEST_TYPE).
# ============================================================
# --- Fastcheck-scope direct tests ---
- label: ":microscope: Encoder Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "encoder"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: VAE Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "vae"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: Transformer Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "transformer"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: Kernel Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "kernel_tests"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: Unit Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "unit_test"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
# --- Full-suite-scope direct tests ---
- label: ":bar_chart: SSIM Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "ssim"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
- exit_status: 1
limit: 2
agents:
queue: "default"
- label: ":test_tube: LoRA Inference Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_lora"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Training Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Distillation DMD Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "distillation_dmd"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Self-Forcing Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "self_forcing"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: LoRA Training Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_lora"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
- exit_status: 1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Training Tests VSA"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_vsa"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
- exit_status: 1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Inference Tests VMoBA"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_vmoba"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Performance Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "performance"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: API Server Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "api_server"
- label: ":dart: Direct Test (${TEST_TYPE})"
if: build.env("TEST_SCOPE") == "direct"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
@@ -307,10 +135,6 @@ steps:
label: ":bar_chart: SSIM Tests"
env:
- TEST_TYPE=ssim
retry:
automatic:
- exit_status: 1
limit: 2
agents:
queue: "default"
- path:
@@ -371,10 +195,6 @@ steps:
label: ":test_tube: LoRA Training Tests"
env:
- TEST_TYPE=training_lora
retry:
automatic:
- exit_status: 1
limit: 2
agents:
queue: "default"
- path:
@@ -387,10 +207,6 @@ steps:
label: ":test_tube: Training Tests VSA"
env:
- TEST_TYPE=training_vsa
retry:
automatic:
- exit_status: 1
limit: 2
agents:
queue: "default"
- path:
View File
+11 -6
View File
@@ -4,10 +4,8 @@ merge_protections:
- base = main
success_conditions:
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model)\\]"
- "#approved-reviews-by>=1"
- check-success~=pre-commit
- check-success=fastcheck-passed
- check-success=full-suite-passed
pull_request_rules:
@@ -274,15 +272,24 @@ pull_request_rules:
merge:
method: squash
- name: auto-update when ready
- name: auto-rebase when ready and Full Suite passed
conditions:
- label=ready
- "#approved-reviews-by>=1"
- check-success=full-suite-passed
- -conflict
- -closed
- -draft
actions:
update: {}
rebase: {}
- name: remove ready label on Full Suite failure
conditions:
- label=ready
- check-failure=full-suite-passed
actions:
label:
remove: [ready]
# ============================================================
# PR title format help
@@ -312,5 +319,3 @@ pull_request_rules:
Please update your PR title and the merge protection check will pass automatically.
merge_protections_settings:
reporting_method: check-runs
-80
View File
@@ -1,80 +0,0 @@
name: Aggregate Test Status
on:
status:
permissions:
statuses: write
jobs:
aggregate:
if: >-
github.event.context == 'direct-test-completed'
&& github.event.state == 'success'
runs-on: ubuntu-latest
steps:
- name: Check and update aggregate status
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const sha = context.payload.sha;
const { data } = await github.rest.repos.getCombinedStatusForRef({
owner: context.repo.owner,
repo: context.repo.repo,
ref: sha,
per_page: 100,
});
const bkStatuses = data.statuses.filter(
s => s.context.startsWith('buildkite/ci/')
);
const FASTCHECK_PREFIX = 'buildkite/ci/microscope-';
const FULL_SUITE_PREFIXES = [
'buildkite/ci/test-tube-',
'buildkite/ci/bar-chart-',
];
const fastcheck = bkStatuses.filter(
s => s.context.startsWith(FASTCHECK_PREFIX)
);
const fullSuite = bkStatuses.filter(
s => FULL_SUITE_PREFIXES.some(p => s.context.startsWith(p))
);
if (
fastcheck.length > 0
&& fastcheck.every(s => s.state === 'success')
) {
core.info(
`All ${fastcheck.length} fastcheck tests passed — updating fastcheck-passed`
);
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha,
state: 'success',
context: 'fastcheck-passed',
description:
`All ${fastcheck.length} fastcheck tests passed`,
});
}
if (
fullSuite.length > 0
&& fullSuite.every(s => s.state === 'success')
) {
core.info(
`All ${fullSuite.length} full suite tests passed — updating full-suite-passed`
);
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha,
state: 'success',
context: 'full-suite-passed',
description:
`All ${fullSuite.length} full suite tests passed`,
});
}
+4 -7
View File
@@ -4,11 +4,10 @@ on:
pull_request:
branches: [main]
workflow_call:
inputs:
ref:
description: 'Git ref to checkout (defaults to github.ref)'
required: false
type: string
concurrency:
group: pre-commit-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
permissions:
contents: read
@@ -19,8 +18,6 @@ jobs:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
with:
ref: ${{ inputs.ref || '' }}
- uses: actions/setup-python@v5
with:
python-version: "3.12"
+14 -55
View File
@@ -7,7 +7,6 @@ on:
permissions:
contents: read
pull-requests: write
statuses: write
jobs:
handle-merge:
@@ -33,7 +32,6 @@ jobs:
core.setOutput('has_write', String(hasWrite));
- name: Add ready label and react
id: label
if: steps.perm.outputs.has_write == 'true'
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
@@ -41,6 +39,7 @@ jobs:
const owner = context.repo.owner;
const repo = context.repo.repo;
const prNumber = context.payload.issue.number;
// Remove ready first to allow re-trigger (labeled event fires on add, not if already present)
try { await github.rest.issues.removeLabel({ owner, repo, issue_number: prNumber, name: 'ready' }); } catch {}
await github.rest.issues.addLabels({ owner, repo, issue_number: prNumber, labels: ['ready'] });
await github.rest.reactions.createForIssueComment({
@@ -48,44 +47,6 @@ jobs:
comment_id: context.payload.comment.id,
content: 'rocket',
});
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: prNumber });
core.setOutput('pr_sha', pr.head.sha);
core.setOutput('pr_branch', pr.head.ref);
core.setOutput('pr_number', String(prNumber));
- name: Trigger Full Suite
if: steps.perm.outputs.has_write == 'true'
env:
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
PR_SHA: ${{ steps.label.outputs.pr_sha }}
PR_BRANCH: ${{ steps.label.outputs.pr_branch }}
PR_NUMBER: ${{ steps.label.outputs.pr_number }}
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
run: |
curl -sS --fail-with-body -X POST \
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
-H "Content-Type: application/json" \
--data-raw "$(jq -n \
--arg commit "$PR_SHA" \
--arg branch "$PR_BRANCH" \
--arg message "Full Suite for PR #${PR_NUMBER} (via /merge)" \
--argjson pr_id "$PR_NUMBER" \
'{
commit: $commit,
branch: $branch,
message: $message,
ignore_pipeline_branch_filters: true,
pull_request_id: $pr_id,
pull_request_base_branch: "main",
env: {
TEST_SCOPE: "full",
FULL_SUITE: "true",
PR_NUMBER: ($pr_id | tostring)
}
}')"
parse-command:
if: >-
github.event.issue.pull_request != null
@@ -181,33 +142,20 @@ jobs:
core.setOutput('sha', pr.head.sha);
core.setOutput('branch', pr.head.ref);
- name: React to comment
if: steps.perm.outputs.has_write == 'true'
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
await github.rest.reactions.createForIssueComment({
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: context.payload.comment.id,
content: 'rocket',
});
pre-commit:
needs: parse-command
if: >-
needs.parse-command.outputs.has_write == 'true'
&& needs.parse-command.outputs.test_scope == 'precommit'
uses: ./.github/workflows/ci-precommit.yml
with:
ref: refs/pull/${{ github.event.issue.number }}/merge
post-precommit-status:
needs: [parse-command, pre-commit]
if: always() && needs.parse-command.outputs.test_scope == 'precommit'
runs-on: ubuntu-latest
steps:
- uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
- name: Post commit status
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
env:
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
RESULT: ${{ needs.pre-commit.result }}
@@ -230,6 +178,17 @@ jobs:
&& needs.parse-command.outputs.test_type != ''
runs-on: ubuntu-latest
steps:
- name: React to comment
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
await github.rest.reactions.createForIssueComment({
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: context.payload.comment.id,
content: 'rocket',
});
- name: Trigger Buildkite
env:
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
+3 -3
View File
@@ -1,7 +1,7 @@
name: Trigger Full Suite
on:
pull_request_target:
pull_request:
types: [labeled, synchronize]
permissions:
@@ -10,7 +10,7 @@ permissions:
concurrency:
group: full-suite-${{ github.event.pull_request.number }}
cancel-in-progress: false
cancel-in-progress: true
jobs:
trigger:
@@ -42,7 +42,7 @@ jobs:
# Find running builds for this branch with TEST_SCOPE=full and cancel them
builds=$(curl -sS -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds?branch=${PR_BRANCH}&state=running,scheduled" \
| jq -r '.[] | select(try (.env.TEST_SCOPE == "full") catch false) | .number')
| jq -r '.[] | select(.env.TEST_SCOPE == "full") | .number')
for build_num in $builds; do
echo "Cancelling Buildkite build #$build_num"
curl -sS -X PUT -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
-1
View File
@@ -85,7 +85,6 @@ docs/distillation/examples/
dmd_t2v_output/
preprocess_output_text/
# Next.js / Node artifacts under ui/: see ui/.gitignore
.claude/
.codex/
-1
View File
@@ -1 +0,0 @@
WRN 2026-03-26T13:46:33.469 ?.19646 server_start:193: Failed to start server: operation not permitted: /var/folders/z_/h_6myyk14d1b7z87z3vy4mjh0000gn/T/nvim.dsynkd/iSe0el/nvim.19646.0
-1
View File
@@ -1 +0,0 @@
3.12
-318
View File
@@ -1,318 +0,0 @@
# Attention QAT
Attention QAT in FastVideo covers two related, but different, backends:
- `ATTN_QAT_INFER`: the inference-oriented CUDA kernel path
- `ATTN_QAT_TRAIN`: the training-oriented Triton attention path
Both are selected with `FASTVIDEO_ATTENTION_BACKEND`, but they are not
interchangeable. The main practical split is:
- use `ATTN_QAT_INFER` for standalone inference with the dedicated inference
kernel
- use `ATTN_QAT_TRAIN` for finetuning, validation during training, or when you
specifically want to reproduce the training-side attention path
## Quick Start
If your goal is "run Wan 2.1 14B with Attention QAT inference weights", this is
the shortest path:
1. Build the in-repo kernel package so FastVideo can import `attn_qat_infer`.
2. Download the Wan 2.1 14B QAT checkpoint.
3. Edit the provided inference example to point at the 14B base model and the
downloaded QAT safetensors.
4. Run the example with `ATTN_QAT_INFER`.
### Step 1. Build the kernel package
Before using either Attention QAT backend, build the in-repo
`fastvideo-kernel` package from source:
```bash
git submodule update --init --recursive
cd fastvideo-kernel
./build.sh
```
After a successful build:
- `ATTN_QAT_TRAIN` should be able to import `fastvideo_kernel`
- `ATTN_QAT_INFER` should be able to import `attn_qat_infer`
`ATTN_QAT_INFER` currently targets the Blackwell CUDA path under
`fastvideo-kernel/attn_qat_infer/` and requires CUDA 12.8+.
### Step 2. Download the Wan 2.1 14B QAT checkpoint
FastVideo includes a helper script:
- `examples/inference/optimizations/download_14B_qat.sh`
By default it downloads:
- Hugging Face repo: `FastVideo/14B_qat_400`
- local directory: `checkpoints/14B_qat_400`
Prerequisites:
- `huggingface_hub` installed, for example:
`uv pip install huggingface_hub`
- access to the model repo if it is private or gated:
`huggingface-cli login`
Run the downloader:
```bash
bash examples/inference/optimizations/download_14B_qat.sh
```
To download into a custom directory:
```bash
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
```
The script prints a ready-to-copy `init_weights_from_safetensors=...` value at
the end.
### Step 3. Edit the provided inference example
The example to start from is:
- `examples/inference/optimizations/attn_qat_inference_example.py`
Open that file and update these two values:
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
`Wan-AI/Wan2.1-T2V-14B-Diffusers`
2. Replace
`init_weights_from_safetensors="safetensors_path"` with the directory that
contains the downloaded `.safetensors` files
Example:
```python
import os
from fastvideo import VideoGenerator
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
num_gpus=1,
use_fsdp_inference=True,
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False,
init_weights_from_safetensors="checkpoints/14B_qat_400",
)
```
Important:
- the checked-in example currently uses the `1.3B` base model until you edit it
- do not load the 14B QAT weights on top of the `1.3B` base model; the weights
and model config will not match
### Step 4. Run the inference example
```bash
python examples/inference/optimizations/attn_qat_inference_example.py
```
Generated videos are written to `video_samples/` by default.
## Backend Overview
| Backend | Best for | Package requirement | Primary kernel location |
|---------|----------|---------------------|-------------------------|
| `ATTN_QAT_TRAIN` | finetuning, training-time validation, reproducing the training path | `fastvideo_kernel` | `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` |
| `ATTN_QAT_INFER` | standalone inference with the dedicated CUDA kernel | `attn_qat_infer` from the in-repo `fastvideo-kernel` checkout | `fastvideo-kernel/attn_qat_infer/` |
FastVideo routes backend selection through:
- `fastvideo/envs.py`
- `fastvideo/platforms/cuda.py`
- `fastvideo/attention/backends/attn_qat_train.py`
- `fastvideo/attention/backends/attn_qat_infer.py`
The legacy training pipeline also contains explicit Attention QAT integration:
- `fastvideo/training/training_pipeline.py`
That pipeline forces generator loading through `ATTN_QAT_TRAIN` when
`FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN` or `--generator_4bit_attn` is
enabled.
## Inference Workflows
For standalone inference, prefer `ATTN_QAT_INFER` when the CUDA kernel is
available. Use `ATTN_QAT_TRAIN` for inference only if you intentionally want to
exercise the training-side attention path for debugging or parity checks.
### Minimal Python example
```python
import os
from fastvideo import VideoGenerator
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
num_gpus=1,
)
generator.generate_video(
"A cinematic close-up of rain on a neon street at night.",
output_path="video_samples",
save_video=True,
)
```
### Loading custom safetensors during inference
FastVideo supports loading custom transformer weights through
`init_weights_from_safetensors`.
This value can point to either:
- a directory containing one or more `.safetensors` files
- a single `.safetensors` file
For Wan 2.1 14B QAT inference, the common pattern is:
```python
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
num_gpus=1,
use_fsdp_inference=True,
init_weights_from_safetensors="checkpoints/14B_qat_400",
)
```
### CLI example
You can also force the backend from the command line:
```bash
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
fastvideo generate \
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
--num-gpus 1 \
--sp-size 1 \
--tp-size 1 \
--height 480 \
--width 832 \
--num-frames 77 \
--num-inference-steps 50 \
--guidance-scale 6.0 \
--prompt "A cinematic close-up of rain on a neon street at night." \
--output-path outputs_video/
```
If you want to use custom QAT transformer weights from the CLI, pass the same
custom weight override that the Python API uses:
```bash
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
fastvideo generate \
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
--init-weights-from-safetensors checkpoints/14B_qat_400 \
--num-gpus 1 \
--output-path outputs_video/ \
--prompt "A cinematic close-up of rain on a neon street at night."
```
## Training Workflows
Today the checked-in Attention QAT training launchers use the legacy training
pipeline in `fastvideo/training/wan_training_pipeline.py`.
### Ready-made launchers
Use the provided SLURM scripts directly:
```bash
sbatch examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh
sbatch examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh
```
Both scripts already set:
```bash
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
```
Before launching, update the script-local values that depend on your
environment:
- `WANDB_API_KEY`
- `MODEL_PATH`
- `DATA_DIR`
- `VALIDATION_DATASET_FILE`
- output directory and SLURM resource requests
### What the launchers run
The training scripts eventually invoke:
```bash
torchrun fastvideo/training/wan_training_pipeline.py ...
```
If you are adapting the workflow to your own cluster or running outside SLURM,
the main Attention QAT requirement is still:
```bash
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
```
Then launch the normal Wan training pipeline with your preferred `torchrun`
arguments and training flags.
## Where The Code Lives
Use these paths when you want to trace or modify the Attention QAT flow:
| Location | Purpose |
|----------|---------|
| `fastvideo/attention/backends/attn_qat_train.py` | FastVideo wrapper that imports and calls the Triton training kernel |
| `fastvideo/attention/backends/attn_qat_infer.py` | FastVideo wrapper that imports and calls the inference kernel |
| `fastvideo-kernel/CMakeLists.txt` | Kernel build definition that compiles the `attn_qat_infer` inference extensions |
| `fastvideo/platforms/cuda.py` | Chooses the concrete attention backend at runtime |
| `fastvideo/envs.py` | Documents supported `FASTVIDEO_ATTENTION_BACKEND` values |
| `fastvideo/training/training_pipeline.py` | Training-time forcing logic for the generator attention backend |
| `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` | Triton implementation for `ATTN_QAT_TRAIN` |
| `fastvideo-kernel/attn_qat_infer/api.py` | Python API entrypoint for the inference kernel |
| `fastvideo-kernel/benchmarks/benchmark_*.py` | Kernel-side benchmark scripts for FlashAttn2, SageAttention3, FP4, and comparison plots |
| `fastvideo-kernel/attn_qat_infer/blackwell/api.cu` | CUDA implementation behind `ATTN_QAT_INFER` |
| `fastvideo-kernel/tests/test_attn_qat_train.py` | Kernel-level test coverage for the training path |
| `examples/inference/optimizations/attn_qat_inference_example.py` | Ready-to-edit inference example for custom Attention QAT weights |
| `examples/inference/optimizations/download_14B_qat.sh` | Helper script for downloading the Wan 2.1 14B QAT checkpoint |
| `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 1.3B Attention QAT finetune launcher |
| `examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 14B Attention QAT finetune launcher |
## Troubleshooting
- If `ATTN_QAT_TRAIN` fails to import, verify that `fastvideo-kernel` built
successfully and exposes `fastvideo_kernel`.
- If `ATTN_QAT_INFER` fails to import, verify that the local build exposes the
`attn_qat_infer` package.
- If the Wan 2.1 14B example fails after you changed only the checkpoint path,
make sure you also changed the base model to
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
- If you hit issues with CPU memory pressure or obscure CUDA argument errors in
the example script, try setting `pin_cpu_memory=False`.
- If you want a known-safe fallback for debugging, use
`FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA`.
## Related Pages
- [Attention Overview](../index.md)
- [Inference Optimizations](../../inference/optimizations.md)
- [Debugging](../../utilities/debugging.md)
-2
View File
@@ -5,8 +5,6 @@ FastVideo provides highly optimized custom attention kernels to accelerate video
## Supported Kernels
* **[Video Sparse Attention (VSA)](vsa/index.md)**: Sparse attention mechanism selecting top-k blocks.
* **[Attention QAT](attn_qat/index.md)**: Dedicated guide for Attention QAT
inference, training, checkpoint loading, and troubleshooting.
* **[Sliding Tile Attention (STA)](sta/index.md)**: STA kernel support is kept in
`fastvideo-kernel`; full FastVideo STA pipeline workflow is archived in
`sta_do_not_delete`.
+5 -30
View File
@@ -24,7 +24,7 @@ PR push
Runs on the PR branch directly
│
pass ──► Mergify auto-squash-merges to main, branch deleted
fail ──► fix the regression, push, and /merge again
fail ──► Mergify removes 'ready' label; fix and /merge again
```
---
@@ -102,8 +102,8 @@ failing test's output.
| Performance Tests | `performance` | 30 min |
| API Server Tests | `api_server` | 30 min |
If a Full Suite test fails, check the Buildkite build log for the failing step's output.
Fix the regression, push, and comment `/merge` again to re-trigger.
A Full Suite failure removes the `ready` label automatically. A Mergify comment links to
the Buildkite build. Fix the regression, push, and comment `/merge` again.
---
@@ -129,8 +129,8 @@ Suite passing directly on the PR branch.
- No merge conflicts
5. If all conditions pass, Mergify squash-merges to `main` automatically. The branch is
deleted after merge.
6. If the Full Suite fails, the developer fixes the issue, pushes, and comments `/merge`
again to re-trigger.
6. If the Full Suite fails, Mergify removes the `ready` label and posts a comment linking to
the Buildkite build. The developer fixes the issue, pushes, and comments `/merge` again.
**Merge conditions summary:**
@@ -279,30 +279,6 @@ Triggers a specific Buildkite test or suite on the current PR branch.
| `/test api` | API server integration tests | `api_server` |
| `/test full` | Entire Full Suite | all (with `TEST_SCOPE=full`) |
| `/test fastcheck` | Entire Fastcheck suite | fastcheck (with `TEST_SCOPE=fastcheck`) |
| `/test pre-commit` | Pre-commit checks on PR code | — (runs `ci-precommit.yml` via `workflow_call`) |
**Re-running failed tests:** When you use `/test <name>` to re-run a specific failed test,
the resulting Buildkite check uses the same name as the original (e.g., `/test encoder`
creates `buildkite/ci/microscope-encoder-tests`). This overwrites the failed check status.
Once all tests in a tier pass, the aggregate status (`fastcheck-passed` or
`full-suite-passed`) is automatically updated to `success` by the `ci-aggregate-status.yml`
workflow.
**How aggregate status refresh works:**
1. `/test <name>` triggers a Buildkite build with `TEST_SCOPE=direct`. The test step uses
the same label as its fastcheck/full-suite counterpart, so the resulting GitHub check
overwrites the original.
2. When the build completes, Buildkite's `notify` posts a `direct-test-completed` commit
status. This is the only signal that triggers the aggregate workflow — intermediate step
status updates do not trigger it.
3. `ci-aggregate-status.yml` fires, calls `getCombinedStatusForRef` to fetch the latest
status for every context on that commit (each context returns only its most recent
state), groups them by prefix (`microscope-*` → fastcheck, `test-tube-*`/`bar-chart-*`
→ full suite), and posts `fastcheck-passed: success` or `full-suite-passed: success` if
all entries in the group are `success`.
4. Tests that were never triggered (skipped by monorepo-diff) have no status entry and do
not block the aggregate.
---
@@ -320,7 +296,6 @@ Protected branches (`main`, `master`, `release/*`) are never deleted.
| `ci-precommit.yml` | Every push / PR against `main` | Runs pre-commit hooks (yapf, ruff, mypy, codespell, pymarkdown, actionlint, check-filenames) |
| `ci-trigger-full-suite.yml` | `ready` label added to a PR | Calls Buildkite API to run Full Suite on the PR branch |
| `ci-slash-commands.yml` | PR comment starting with `/merge` or `/test` | Handles slash commands; adds `ready` label or triggers Buildkite |
| `ci-aggregate-status.yml` | Any Buildkite commit status update | Checks if all tests in a tier passed; updates `fastcheck-passed` or `full-suite-passed` |
| `community-issue-labeler.yml` | Issue opened or edited | Auto-labels issues by keyword matching against title and body |
| `community-welcome.yml` | First contribution | Posts a welcome comment for first-time contributors |
| `community-stale.yml` | Scheduled | Marks and closes stale issues and PRs |
+5 -10
View File
@@ -104,9 +104,8 @@ distillation, self-forcing, VSA, VMoBA, performance benchmarks, and API server t
8. If all Full Suite tests pass and all merge conditions are met (approval, valid title,
pre-commit green, fastcheck green, no draft, no conflicts), Mergify squash-merges to
`main` automatically. Your branch is deleted.
9. If a Full Suite test fails, check the Buildkite build log for the failing step. Fix the
issue, push, and comment `/merge` again. You can also re-run individual failed tests
with `/test <name>` — see below.
9. If a Full Suite test fails, Mergify removes the `ready` label and posts a comment with a
link to the Buildkite build. Fix the issue, push, and comment `/merge` again.
!!! note
Only contributors with write permission to the repository can trigger slash commands.
@@ -150,15 +149,10 @@ Comment on your PR to trigger specific tests independently of the auto-merge flo
/test vmoba # VMoBA inference tests
/test performance # Performance benchmarks
/test api # API server integration tests
/test pre-commit # Pre-commit checks on PR code
```
The workflow reacts with a 🚀 emoji to confirm the command was received.
When you re-run an individual test with `/test <name>`, the new result overwrites the
original failed check (same Buildkite check name). Once all tests in a tier pass, the
`fastcheck-passed` or `full-suite-passed` status is automatically updated.
---
## Troubleshooting
@@ -205,8 +199,9 @@ Mergify removes the `needs-rebase` label automatically once conflicts are resolv
### Full Suite failed after `/merge`
The Full Suite found a regression. Check the failing Buildkite step's output for assertion
errors or tracebacks.
The Full Suite found a regression. Mergify removes the `ready` label and posts a comment
linking to the Buildkite build. Check the failing step's output for assertion errors or
tracebacks.
Common causes:
@@ -1,720 +0,0 @@
status_definitions:
kept: "Public field remains on a public adapter surface with the same meaning."
moved: "Public field remains supported but normalizes into a different nested path."
profile_owned: "Public field remains supported only through a model/profile-specific surface."
compatibility_only: "Legacy public field remains adapter-only during migration and is not part of the canonical typed schema."
private_only: "Field should only be handled by private adapters and is not a public FastVideo compatibility promise."
internal_only: "Field is runtime/config plumbing and should not be part of the new public typed inference API."
surfaces:
fastvideo_args:
moved:
model_path: generator.model_path
workload_type: generator.pipeline.workload_type
distributed_executor_backend: generator.engine.execution_backend
trust_remote_code: generator.trust_remote_code
revision: generator.revision
num_gpus: generator.engine.num_gpus
tp_size: generator.engine.parallelism.tp_size
sp_size: generator.engine.parallelism.sp_size
hsdp_replicate_dim: generator.engine.parallelism.hsdp_replicate_dim
hsdp_shard_dim: generator.engine.parallelism.hsdp_shard_dim
dist_timeout: generator.engine.parallelism.dist_timeout
lora_path: generator.pipeline.components.lora_path
dit_cpu_offload: generator.engine.offload.dit
use_fsdp_inference: generator.engine.use_fsdp_inference
dit_layerwise_offload: generator.engine.offload.dit_layerwise
text_encoder_cpu_offload: generator.engine.offload.text_encoder
image_encoder_cpu_offload: generator.engine.offload.image_encoder
vae_cpu_offload: generator.engine.offload.vae
pin_cpu_memory: generator.engine.offload.pin_cpu_memory
enable_torch_compile: generator.engine.compile.enabled
torch_compile_kwargs: generator.engine.compile.kwargs
disable_autocast: generator.engine.disable_autocast
enable_stage_verification: generator.engine.enable_stage_verification
prompt_txt: request.inputs.prompt_path
override_text_encoder_safetensors: generator.pipeline.components.text_encoder_weights
override_text_encoder_quant: generator.engine.quantization.text_encoder_quant
transformer_quant: generator.engine.quantization.transformer_quant
override_transformer_cls_name: generator.pipeline.components.override_transformer_cls_name
init_weights_from_safetensors: generator.pipeline.components.transformer_weights
init_weights_from_safetensors_2: generator.pipeline.components.transformer_2_weights
override_pipeline_cls_name: generator.pipeline.components.override_pipeline_cls_name
boundary_ratio: request.sampling.boundary_ratio
profile_owned:
ltx2_vae_tiling: generator.pipeline.profile_overrides.ltx2.vae_tiling
ltx2_vae_spatial_tile_size_in_pixels: generator.pipeline.profile_overrides.ltx2.vae.spatial_tile_size_in_pixels
ltx2_vae_spatial_tile_overlap_in_pixels: generator.pipeline.profile_overrides.ltx2.vae.spatial_tile_overlap_in_pixels
ltx2_vae_temporal_tile_size_in_frames: generator.pipeline.profile_overrides.ltx2.vae.temporal_tile_size_in_frames
ltx2_vae_temporal_tile_overlap_in_frames: generator.pipeline.profile_overrides.ltx2.vae.temporal_tile_overlap_in_frames
ltx2_initial_latent_path: request.extensions.ltx2.initial_latent_path
compatibility_only:
mode: "Legacy multi-mode FastVideoArgs switch; typed inference config should not expose execution mode."
inference_mode: "Legacy boolean mirror of mode; kept only through adapters while FastVideoArgs remains."
lora_nickname: "Legacy adapter-selection surface pending LoRA API cleanup."
lora_target_modules: "Legacy LoRA configuration surface pending dedicated component API."
output_type: "Legacy output formatting surface pending GenerationResult cleanup."
VSA_sparsity: "Model-specific inference optimization not yet represented in the typed public schema."
moba_config_path: "Model-specific MoBA optimization surface not yet represented in the typed public schema."
master_port: "Executor/bootstrap compatibility field; not part of the canonical inference schema."
private_only:
ray_placement_group: "Ray deployment-only field."
ray_runtime_env: "Ray deployment-only field."
internal_only:
pipeline_config: "Legacy internal carrier object."
preprocess_config: "Legacy preprocess carrier object."
moba_config: "Derived runtime config loaded from moba_config_path."
model_paths: "Runtime bookkeeping."
model_loaded: "Runtime bookkeeping."
pipeline_config_base:
moved:
pipeline_config_path: generator.pipeline.components.pipeline_config_path
profile_owned:
embedded_cfg_scale: generator.pipeline.profile_overrides.embedded_cfg_scale
flow_shift: generator.pipeline.profile_overrides.flow_shift
flow_shift_sr: generator.pipeline.profile_overrides.flow_shift_sr
is_causal: generator.pipeline.profile_overrides.is_causal
vae_tiling: generator.pipeline.profile_overrides.vae_tiling
vae_sp: generator.pipeline.profile_overrides.vae_sp
dmd_denoising_steps: generator.pipeline.profile_overrides.dmd_denoising_steps
ti2v_task: generator.pipeline.profile_overrides.ti2v_task
boundary_ratio: generator.pipeline.profile_overrides.boundary_ratio
compatibility_only:
model_path: "Redundant with generator.model_path."
disable_autocast: "Duplicated by generator.engine.disable_autocast during migration."
dit_precision: "Precision override pending dedicated typed component precision design."
upsampler_precision: "Precision override pending dedicated typed component precision design."
vae_precision: "Precision override pending dedicated typed component precision design."
image_encoder_precision: "Precision override pending dedicated typed component precision design."
text_encoder_precisions: "Precision override pending dedicated typed component precision design."
internal_only:
dit_config: "Legacy internal component config object."
upsampler_config: "Legacy internal component config object."
vae_config: "Legacy internal component config object."
image_encoder_config: "Legacy internal component config object."
text_encoder_configs: "Legacy internal component config object."
preprocess_text_funcs: "Internal text preprocessing hooks."
postprocess_text_funcs: "Internal text postprocessing hooks."
pipeline_config_extensions:
profile_owned:
conditioning_strategy:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
max_num_conditional_frames:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
min_num_conditional_frames:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
sigma_conditional:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
sigma_data:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
state_ch:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
state_t:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
text_encoder_class:
sources:
- fastvideo.configs.pipelines.cosmos.CosmosConfig
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
autoregressive_chunk_frames:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
autoregressive_overlap_frames:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
cfg_behavior:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
default_camera_rotation:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
default_movement_distance:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
default_negative_prompt:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
default_trajectory_type:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
filter_points_threshold:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
fps:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
frame_buffer_max:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
moge_model_name:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
noise_aug_strength:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
num_frames:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
offload_moge_after_depth:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
use_moge_depth:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
video_resolution:
sources:
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
text_encoder_crop_start:
sources:
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V480PStepDistilledConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V720PConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15SR1080PConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V480PConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V720PConfig
- fastvideo.configs.pipelines.hyworld.HYWorldConfig
- fastvideo.configs.pipelines.hyworld.Hunyuan15T2V480PConfig
text_encoder_max_lengths:
sources:
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V480PStepDistilledConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V720PConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15SR1080PConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V480PConfig
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V720PConfig
- fastvideo.configs.pipelines.hyworld.HYWorldConfig
- fastvideo.configs.pipelines.hyworld.Hunyuan15T2V480PConfig
precision:
sources:
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
- fastvideo.configs.pipelines.wan.MatrixGameBaseI2V480PConfig
- fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig
- fastvideo.configs.pipelines.wan.SelfForcingWan2_2_T2V480PConfig
- fastvideo.configs.pipelines.wan.SelfForcingWanT2V480PConfig
- fastvideo.configs.pipelines.wan.WANV2VConfig
- fastvideo.configs.pipelines.wan.Wan2_2_I2V_A14B_Config
- fastvideo.configs.pipelines.wan.Wan2_2_T2V_A14B_Config
- fastvideo.configs.pipelines.wan.Wan2_2_TI2V_5B_Config
- fastvideo.configs.pipelines.wan.WanI2V480PConfig
- fastvideo.configs.pipelines.wan.WanI2V720PConfig
- fastvideo.configs.pipelines.wan.WanT2V480PConfig
- fastvideo.configs.pipelines.wan.WanT2V720PConfig
warp_denoising_step:
sources:
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
- fastvideo.configs.pipelines.wan.MatrixGameBaseI2V480PConfig
- fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig
- fastvideo.configs.pipelines.wan.SelfForcingWan2_2_T2V480PConfig
- fastvideo.configs.pipelines.wan.SelfForcingWanT2V480PConfig
- fastvideo.configs.pipelines.wan.WANV2VConfig
- fastvideo.configs.pipelines.wan.Wan2_2_I2V_A14B_Config
- fastvideo.configs.pipelines.wan.Wan2_2_T2V_A14B_Config
- fastvideo.configs.pipelines.wan.Wan2_2_TI2V_5B_Config
- fastvideo.configs.pipelines.wan.WanI2V480PConfig
- fastvideo.configs.pipelines.wan.WanI2V720PConfig
- fastvideo.configs.pipelines.wan.WanT2V480PConfig
- fastvideo.configs.pipelines.wan.WanT2V720PConfig
bsa_cdf_threshold:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
bsa_chunk_k:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
bsa_chunk_q:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
bsa_params:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
bsa_sparsity:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
enable_bsa:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
enable_kv_cache:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
enhance_hf:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
offload_kv_cache:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
t_thresh:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
use_distill:
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
scheduler_arch:
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
text_encoder_archs:
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
tokenizer_archs:
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
transformer_arch:
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
vae_arch:
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
expand_timesteps:
sources:
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
- fastvideo.configs.pipelines.wan.Wan2_2_TI2V_5B_Config
context_noise:
sources: [fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig]
num_frames_per_block:
sources: [fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig]
compatibility_only:
batch_size: "Gen3C inference-only tuning field pending typed batching design."
gradient_checkpointing: "Gen3C inference-only compatibility field pending typed batching design."
guidance_scale: "Gen3C pipeline-level default pending profile/default-request cleanup."
num_inference_steps: "Gen3C pipeline-level default pending profile/default-request cleanup."
internal_only:
audio_decoder_config: "Legacy internal component config object."
audio_decoder_precision: "Precision override pending dedicated component precision design."
vocoder_config: "Legacy internal component config object."
vocoder_precision: "Precision override pending dedicated component precision design."
sampling_param_base:
moved:
image_path: request.inputs.image_path
pil_image: request.inputs.pil_image
video_path: request.inputs.video_path
mouse_cond: request.inputs.mouse_cond
keyboard_cond: request.inputs.keyboard_cond
grid_sizes: request.inputs.grid_sizes
pose: request.inputs.pose
c2ws_plucker_emb: request.inputs.c2ws_plucker_emb
refine_from: request.inputs.refine_from
stage1_video: request.inputs.stage1_video
prompt: request.prompt
negative_prompt: request.negative_prompt
prompt_path: request.inputs.prompt_path
output_path: request.output.output_path
output_video_name: request.output.output_video_name
num_videos_per_prompt: request.sampling.num_videos_per_prompt
seed: request.sampling.seed
num_frames: request.sampling.num_frames
height: request.sampling.height
width: request.sampling.width
height_sr: request.sampling.height_sr
width_sr: request.sampling.width_sr
fps: request.sampling.fps
num_inference_steps: request.sampling.num_inference_steps
num_inference_steps_sr: request.sampling.num_inference_steps_sr
guidance_scale: request.sampling.guidance_scale
guidance_scale_2: request.sampling.guidance_scale_2
guidance_rescale: request.sampling.guidance_rescale
boundary_ratio: request.sampling.boundary_ratio
sigmas: request.sampling.sigmas
enable_teacache: request.runtime.enable_teacache
save_video: request.output.save_video
return_frames: request.output.return_frames
return_trajectory_latents: request.runtime.return_trajectory_latents
return_trajectory_decoded: request.runtime.return_trajectory_decoded
profile_owned:
t_thresh: request.stage_overrides.refine.t_thresh
spatial_refine_only: request.stage_overrides.refine.spatial_refine_only
num_cond_frames: request.stage_overrides.refine.num_cond_frames
trajectory_type: request.extensions.gen3c.trajectory_type
movement_distance: request.extensions.gen3c.movement_distance
camera_rotation: request.extensions.gen3c.camera_rotation
internal_only:
data_type: "Derived from the request shape and not a public input."
sampling_param_extensions:
moved: {}
profile_owned:
action_list:
target: request.extensions.hunyuangamecraft.action_list
sources:
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
action_speed_list:
target: request.extensions.hunyuangamecraft.action_speed_list
sources:
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
camera_states:
target: request.extensions.hunyuangamecraft.camera_states
sources:
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
camera_trajectory:
target: request.extensions.hunyuangamecraft.camera_trajectory
sources:
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
conditioning_mask:
target: request.extensions.hunyuangamecraft.conditioning_mask
sources:
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
gt_latents:
target: request.extensions.hunyuangamecraft.gt_latents
sources:
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
prompt_attention_mask:
target: request.extensions.hyworld.prompt_attention_mask
sources: [fastvideo.configs.sample.hyworld.HYWorld_SamplingParam]
negative_attention_mask:
target: request.extensions.hyworld.negative_attention_mask
sources: [fastvideo.configs.sample.hyworld.HYWorld_SamplingParam]
ltx2_cfg_scale_audio:
target: request.extensions.ltx2.cfg_scale_audio
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_cfg_scale_video:
target: request.extensions.ltx2.cfg_scale_video
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_modality_scale_audio:
target: request.extensions.ltx2.modality_scale_audio
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_modality_scale_video:
target: request.extensions.ltx2.modality_scale_video
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_rescale_scale:
target: request.extensions.ltx2.rescale_scale
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_stg_blocks_audio:
target: request.extensions.ltx2.stg_blocks_audio
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_stg_blocks_video:
target: request.extensions.ltx2.stg_blocks_video
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_stg_scale_audio:
target: request.extensions.ltx2.stg_scale_audio
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
ltx2_stg_scale_video:
target: request.extensions.ltx2.stg_scale_video
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
openai_image_request:
kept:
model: "HTTP adapter model-routing field."
response_format: "HTTP adapter response formatting field."
output_format: "HTTP adapter output-format field."
background: "HTTP adapter output-format field."
quality: "Compatibility field currently accepted by the adapter."
style: "Compatibility field currently accepted by the adapter."
user: "Compatibility field currently accepted by the adapter."
moved:
prompt: request.prompt
n: request.sampling.num_videos_per_prompt
size:
target: request.sampling.width,height
note: "Adapter parses OpenAI size strings as WIDTHxHEIGHT and forwards width then height."
num_inference_steps: request.sampling.num_inference_steps
guidance_scale: request.sampling.guidance_scale
true_cfg_scale: request.sampling.true_cfg_scale
seed: request.sampling.seed
negative_prompt: request.negative_prompt
enable_teacache: request.runtime.enable_teacache
openai_video_request:
kept:
model: "HTTP adapter model-routing field."
moved:
prompt: request.prompt
input_reference: request.inputs.image_path
reference_url: request.inputs.image_path
size:
target: request.sampling.width,height
note: "Adapter parses OpenAI size strings as WIDTHxHEIGHT and forwards width then height."
fps: request.sampling.fps
num_frames: request.sampling.num_frames
seed: request.sampling.seed
num_inference_steps: request.sampling.num_inference_steps
guidance_scale: request.sampling.guidance_scale
guidance_scale_2: request.sampling.guidance_scale_2
true_cfg_scale: request.sampling.true_cfg_scale
negative_prompt: request.negative_prompt
enable_teacache: request.runtime.enable_teacache
output_path: request.output.output_path
compatibility_only:
seconds:
target: request.sampling.num_frames
note: "HTTP adapter duration convenience field. If num_frames is omitted, the adapter computes num_frames = fps * seconds."
cli:
notes:
- "CLI parity is checked against the actual generate/serve parser dest sets."
- "The inventory tracks parser dest names, excluding argparse's implicit help action."
generate:
explicit_local_fields:
- config
expected_dests:
- VSA_sparsity
- boundary_ratio
- bsa_cdf_threshold
- bsa_chunk_k
- bsa_chunk_q
- bsa_sparsity
- config
- disable_autocast
- dist_timeout
- distributed_executor_backend
- dit_config.prefix
- dit_config.quant_config
- dit_cpu_offload
- dit_layerwise_offload
- dit_precision
- dmd_denoising_steps
- embedded_cfg_scale
- enable_bsa
- enable_stage_verification
- enable_torch_compile
- flow_shift
- fps
- guidance_rescale
- guidance_scale
- height
- hsdp_replicate_dim
- hsdp_shard_dim
- image_encoder_cpu_offload
- image_encoder_precision
- image_path
- inference_mode
- init_weights_from_safetensors
- init_weights_from_safetensors_2
- lora_nickname
- lora_path
- lora_target_modules
- ltx2_initial_latent_path
- ltx2_vae_spatial_tile_overlap_in_pixels
- ltx2_vae_spatial_tile_size_in_pixels
- ltx2_vae_temporal_tile_overlap_in_frames
- ltx2_vae_temporal_tile_size_in_frames
- ltx2_vae_tiling
- master_port
- moba_config_path
- mode
- model_path
- negative_prompt
- num_cond_frames
- num_frames
- num_gpus
- num_inference_steps
- num_videos_per_prompt
- output_path
- output_type
- output_video_name
- override_pipeline_cls_name
- override_text_encoder_quant
- override_text_encoder_safetensors
- override_transformer_cls_name
- pin_cpu_memory
- pipeline_config_path
- preprocess.dataloader_num_workers
- preprocess.dataset_output_dir
- preprocess.dataset_path
- preprocess.dataset_type
- preprocess.do_temporal_sample
- preprocess.drop_short_ratio
- preprocess.flush_frequency
- preprocess.max_height
- preprocess.max_width
- preprocess.model_path
- preprocess.num_frames
- preprocess.preprocess_video_batch_size
- preprocess.samples_per_file
- preprocess.seed
- preprocess.speed_factor
- preprocess.train_fps
- preprocess.training_cfg_rate
- preprocess.video_length_tolerance_range
- preprocess.video_loader_type
- preprocess.with_audio
- prompt
- prompt_path
- prompt_txt
- refine_from
- return_frames
- return_trajectory_decoded
- return_trajectory_latents
- revision
- save_video
- seed
- sp_size
- spatial_refine_only
- t_thresh
- text_encoder_configs
- text_encoder_cpu_offload
- text_encoder_precisions
- torch_compile_kwargs
- transformer_quant
- tp_size
- trust_remote_code
- use_fsdp_inference
- vae_config.blend_num_frames
- vae_config.load_decoder
- vae_config.load_encoder
- vae_config.tile_sample_min_height
- vae_config.tile_sample_min_num_frames
- vae_config.tile_sample_min_width
- vae_config.tile_sample_stride_height
- vae_config.tile_sample_stride_num_frames
- vae_config.tile_sample_stride_width
- vae_config.use_parallel_tiling
- vae_config.use_temporal_tiling
- vae_config.use_tiling
- vae_cpu_offload
- vae_precision
- vae_sp
- vae_tiling
- video_path
- width
- workload_type
serve:
explicit_local_fields:
- config
- host
- output_dir
- port
expected_dests:
- VSA_sparsity
- bsa_cdf_threshold
- bsa_chunk_k
- bsa_chunk_q
- bsa_sparsity
- config
- disable_autocast
- dist_timeout
- distributed_executor_backend
- dit_config.prefix
- dit_config.quant_config
- dit_cpu_offload
- dit_layerwise_offload
- dit_precision
- dmd_denoising_steps
- embedded_cfg_scale
- enable_bsa
- enable_stage_verification
- enable_torch_compile
- flow_shift
- host
- hsdp_replicate_dim
- hsdp_shard_dim
- image_encoder_cpu_offload
- image_encoder_precision
- inference_mode
- init_weights_from_safetensors
- init_weights_from_safetensors_2
- lora_nickname
- lora_path
- lora_target_modules
- ltx2_initial_latent_path
- ltx2_vae_spatial_tile_overlap_in_pixels
- ltx2_vae_spatial_tile_size_in_pixels
- ltx2_vae_temporal_tile_overlap_in_frames
- ltx2_vae_temporal_tile_size_in_frames
- ltx2_vae_tiling
- master_port
- mode
- model_path
- num_gpus
- output_dir
- output_type
- override_pipeline_cls_name
- override_text_encoder_quant
- override_text_encoder_safetensors
- override_transformer_cls_name
- pin_cpu_memory
- pipeline_config_path
- port
- preprocess.dataloader_num_workers
- preprocess.dataset_output_dir
- preprocess.dataset_path
- preprocess.dataset_type
- preprocess.do_temporal_sample
- preprocess.drop_short_ratio
- preprocess.flush_frequency
- preprocess.max_height
- preprocess.max_width
- preprocess.model_path
- preprocess.num_frames
- preprocess.preprocess_video_batch_size
- preprocess.samples_per_file
- preprocess.seed
- preprocess.speed_factor
- preprocess.train_fps
- preprocess.training_cfg_rate
- preprocess.video_length_tolerance_range
- preprocess.video_loader_type
- preprocess.with_audio
- prompt_txt
- revision
- sp_size
- text_encoder_cpu_offload
- text_encoder_precisions
- torch_compile_kwargs
- transformer_quant
- tp_size
- trust_remote_code
- use_fsdp_inference
- vae_config.blend_num_frames
- vae_config.load_decoder
- vae_config.load_encoder
- vae_config.tile_sample_min_height
- vae_config.tile_sample_min_num_frames
- vae_config.tile_sample_min_width
- vae_config.tile_sample_stride_height
- vae_config.tile_sample_stride_num_frames
- vae_config.tile_sample_stride_width
- vae_config.use_parallel_tiling
- vae_config.use_temporal_tiling
- vae_config.use_tiling
- vae_cpu_offload
- vae_precision
- vae_sp
- vae_tiling
- workload_type
-5
View File
@@ -167,11 +167,6 @@ How this maps to FastVideo:
- Attention backends live in `fastvideo/attention/` and can be selected via
`FASTVIDEO_ATTENTION_BACKEND`.
- SageAttention3 is split into two selectable backends:
`SAGE_ATTN_THREE` for the regular upstream package and
`ATTN_QAT_INFER` for the FastVideoKernel-backed inference variant.
- `ATTN_QAT_TRAIN` is a separate FastVideoKernel Triton backend for the QAT attention
path.
- `LocalAttention` is used for cross-attention and most attention layers.
- `DistributedAttention` is used for full-sequence self-attention in the DiT.
- Tensor-parallel layers live in `fastvideo/layers/`.
-129
View File
@@ -1,129 +0,0 @@
# GEN3C: 3D-Informed Camera-Controlled Video Generation
[GEN3C](https://arxiv.org/abs/2503.03751) is NVIDIA's Cosmos-7B-based video model for camera-controlled generation from a single image. The FastVideo integration supports the GEN3C I2V workflow, including 3D cache conditioning and tokenizer-based conditioning latents.
## Key Features
- **Camera trajectory control**: `left/right/up/down/zoom_in/zoom_out/clockwise/counterclockwise`
- **3D cache conditioning**: depth prediction -> point cloud cache -> forward warping -> latent conditioning
- **Single-image to video generation**: 121-frame generation with camera motion
- **Official raw checkpoint conversion**: `model.pt` -> Diffusers/FastVideo layout
## Model Sources
- Official raw checkpoint (not Diffusers): `nvidia/GEN3C-Cosmos-7B`
- Diffusers-format checkpoint: `FastVideo/GEN3C-Cosmos-7B-Diffusers`
## Prerequisites
- Install MoGe:
```bash
pip install git+https://github.com/microsoft/MoGe.git
```
- If you hit `ImportError: libGL.so.1` (common on Ubuntu/headless nodes), you can try installing OpenCV runtime libs:
```bash
sudo apt-get update
sudo apt-get install -y libgl1 libglib2.0-0 libsm6 libxext6 libxrender1
```
## Quick Start
### Option A: Use Diffusers-format weights directly
```bash
python examples/inference/basic/basic_gen3c.py \
--model_path FastVideo/GEN3C-Cosmos-7B-Diffusers \
--image_path /path/to/input.png \
--prompt "" \
--trajectory left \
--movement_distance 0.3 \
--camera_rotation center_facing \
--num_inference_steps 35 \
--guidance_scale 1.0 \
--output_path outputs_video/gen3c_output.mp4
```
### Option B: Convert official raw checkpoint locally
1. Download:
```bash
huggingface-cli download nvidia/GEN3C-Cosmos-7B --local-dir official_weights/GEN3C-Cosmos-7B
```
1. Convert:
```bash
python scripts/checkpoint_conversion/convert_gen3c_to_fastvideo.py \
--source official_weights/GEN3C-Cosmos-7B/model.pt \
--output converted_weights/GEN3C-Cosmos-7B
```
1. Run:
```bash
python examples/inference/basic/basic_gen3c.py \
--model_path converted_weights/GEN3C-Cosmos-7B \
--image_path /path/to/input.png \
--prompt "" \
--trajectory left \
--movement_distance 0.3 \
--camera_rotation center_facing \
--num_inference_steps 35 \
--guidance_scale 1.0 \
--output_path outputs_video/gen3c_output.mp4
```
## FastVideo Defaults
GEN3C defaults in FastVideo:
- `height=704`, `width=1280`
- `num_frames=121`
- `num_inference_steps=35`
- `guidance_scale=1.0`
- `fps=24`
These values are defined in:
- `fastvideo/configs/sample/gen3c.py`
- `fastvideo/configs/pipelines/gen3c.py`
and align with the official GEN3C inference defaults in:
- `tmp/GEN3C/cosmos_predict1/diffusion/inference/inference_utils.py`
## Scheduler Note
The converted GEN3C Diffusers layout may include a FlowMatch scheduler config, but GEN3C denoising uses EDM preconditioning behavior. FastVideo's GEN3C pipeline enforces an EDM scheduler at runtime for parity with official inference behavior.
Implementation path:
- `fastvideo/pipelines/basic/gen3c/gen3c_pipeline.py`
## 3D Cache Conditioning Path
FastVideo GEN3C conditioning stage performs:
1. MoGe depth estimation from input image
2. 3D cache initialization
3. Camera trajectory generation
4. Forward rendering of warped frames + masks
5. VAE/tokenizer encoding of conditioning buffers
6. Denoising with condition mask + condition pose channels
Main implementation:
- `fastvideo/pipelines/basic/gen3c/gen3c_pipeline.py`
- `fastvideo/pipelines/basic/gen3c/cache_3d.py`
- `fastvideo/pipelines/basic/gen3c/depth_estimation.py`
- `fastvideo/models/vaes/gen3c_tokenizer_vae.py`
## References
- [GEN3C Paper](https://arxiv.org/abs/2503.03751)
- [Official Repository](https://github.com/nv-tlabs/GEN3C)
- [Official Checkpoint (raw)](https://huggingface.co/nvidia/GEN3C-Cosmos-7B)
-2
View File
@@ -107,8 +107,6 @@ If you encounter CUDA out of memory errors:
(single GPU) or `use_fsdp_inference=True` (multi-GPU)
- Try a smaller model or use distilled versions
- Use `num_gpus` > 1 if multiple GPUs are available
- Try enabling FSDP inference with `use_fsdp_inference=True` (may slow down generation)
- Try enabling DiT layerwise offload with `dit_layerwise_offload=True` (now only a few models support this, but may introduce less overhead than FSDP)
### Slow Generation
-57
View File
@@ -21,8 +21,6 @@ This page describes the various options for speeding up generation times in Fast
- Video Sparse Attention: `FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN`
- Sage Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN`
- Sage Attention 3: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN_THREE`
- Attn QAT Infer: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER`
- Attn QAT Train: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN`
- Video MoBA Attention: `FASTVIDEO_ATTENTION_BACKEND=VMOBA_ATTN`
- Sparse Linear Attention: `FASTVIDEO_ATTENTION_BACKEND=SLA_ATTN`
- SageSLA Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_SLA_ATTN`
@@ -105,14 +103,6 @@ python setup.py install # or pip install -e .
### Sage Attention 3
FastVideo now exposes two SageAttention3-compatible backends with distinct
environment variable values:
- `SAGE_ATTN_THREE`: the regular upstream SageAttention3 backend imported from
the `sageattn3` package.
- `ATTN_QAT_INFER`: the inference CUDA-kernel backend imported from the
in-repo `attn_qat_infer` package.
**`SAGE_ATTN_THREE`**
[SageAttention 3](https://github.com/thu-ml/SageAttention/tree/main/sageattention3_blackwell) is an advanced attention mechanism that leverages FP4 quantization and Blackwell GPU Tensor Cores for significant performance improvements.
@@ -127,53 +117,6 @@ Note that Sage Attention 3 requires `python>=3.13`, `torch>=2.8.0`, `CUDA >=12.8
To use Sage Attention 3 in FastVideo, follow the `README.md` in the linked repository to install the package from source.
### Attn QAT Infer
**`ATTN_QAT_INFER`**
This backend uses the `attn_qat_infer` implementation that lives in the
`fastvideo-kernel` repository alongside the `fastvideo_kernel` Triton kernels.
Use this backend when you want to run the dedicated FP4 inference CUDA kernel
directly during inference.
For the full Attention QAT guide, including Wan 2.1 14B checkpoint download,
example editing steps, training launchers, and troubleshooting, see
[Attention QAT](../attention/attn_qat/index.md).
This backend currently assumes access to the in-repo `fastvideo-kernel`
checkout or an equivalent editable/source install that exposes:
- `attn_qat_infer`
Example:
```python
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
```
### QAT Attention
**`ATTN_QAT_TRAIN`**
This backend uses the FastVideoKernel Triton attention implementation from
`fastvideo_kernel.triton_kernels.attn_qat_train`. Use it when you specifically
want the training-oriented Triton attention path rather than the
`attn_qat_infer` CUDA kernel path.
The dedicated [Attention QAT](../attention/attn_qat/index.md) page covers when
to use `ATTN_QAT_TRAIN` versus `ATTN_QAT_INFER`, the ready-made training
launchers, and the end-to-end Wan 2.1 14B inference workflow.
This backend currently assumes access to an install that exposes:
- `fastvideo_kernel`
Example:
```python
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_TRAIN"
```
### V-MoBA / SLA / SageSLA
These backends are model-specific and require the corresponding kernels and
-6
View File
@@ -73,7 +73,6 @@ pipeline initialization and sampling.
| Matrix Game 2.0 Base | `FastVideo/Matrix-Game-2.0-Base-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
| Matrix Game 2.0 GTA | `FastVideo/Matrix-Game-2.0-GTA-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
| Matrix Game 2.0 TempleRun | `FastVideo/Matrix-Game-2.0-TempleRun-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
| GEN3C Cosmos 7B | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | 704px1280p | ❌ | ❌ | ❌ | ⭕ | ⭕ |
**Note**: Wan2.2 TI2V 5B has some quality issues when performing I2V generation. We are working on fixing this issue.
@@ -86,11 +85,6 @@ The authoritative source for model-ID recognition is
`fastvideo/registry.py`. If a model ID is registered there, FastVideo can
resolve default pipeline and sampling configuration for it.
**Note (GEN3C)**: The official `nvidia/GEN3C-Cosmos-7B` repo provides a raw
`model.pt` checkpoint. Use a Diffusers-format repo (for example,
`FastVideo/GEN3C-Cosmos-7B-Diffusers`) or convert locally with
`scripts/checkpoint_conversion/convert_gen3c_to_fastvideo.py`.
## Special requirements
### Sliding Tile Attention
+2 -7
View File
@@ -27,8 +27,7 @@ Useful variables:
- `FASTVIDEO_LOGGING_LEVEL`: `DEBUG`, `INFO`, `WARNING`, `ERROR`
- `FASTVIDEO_STAGE_LOGGING`: print per-stage timings during pipeline execution
- `FASTVIDEO_ATTENTION_BACKEND`: force an attention backend (for example
`TORCH_SDPA`, `FLASH_ATTN`, `SAGE_ATTN_THREE`, or
`ATTN_QAT_INFER`, or `ATTN_QAT_TRAIN`)
`TORCH_SDPA` or `FLASH_ATTN`)
## Common Failure Modes
@@ -53,11 +52,7 @@ If forcing a backend fails, verify optional dependencies are installed:
- `VIDEO_SPARSE_ATTN`: `fastvideo-kernel`
- `SLIDING_TILE_ATTN`: STA legacy workflow in
`sta_do_not_delete` + `fastvideo-kernel`
- `SAGE_ATTN`: SageAttention package
- `SAGE_ATTN_THREE`: upstream `sageattn3` package
- `ATTN_QAT_INFER`: `fastvideo-kernel` checkout/source install that exposes
`attn_qat_infer`
- `ATTN_QAT_TRAIN`: `fastvideo-kernel` install exposing `fastvideo_kernel`
- `SAGE_ATTN` / `SAGE_ATTN_THREE`: SageAttention packages
As a fallback, use:
@@ -22,7 +22,7 @@ export NODE_RANK=$SLURM_PROCID
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
export MASTER_ADDR=${nodes[0]}
export TOKENIZERS_PARALLELISM=false
export WANDB_API_KEY=YOUR_WANDB_API_KEY
export WANDB_API_KEY="2f25ad37933894dbf0966c838c0b8494987f9f2f"
# export WANDB_API_KEY='your_wandb_api_key_here'
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
@@ -9,7 +9,7 @@ pip install vsa
### 1. Download dataset:
```bash
bash examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/download_dataset.sh
bash examples/distill/Wan-Syn-480P/download_dataset.sh
```
### 2. Configure and run distillation:
@@ -1,3 +1,3 @@
#!/bin/bash
mkdir -p data
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "data/Wan-Syn_77x448x832_600k" --repo_type "dataset"
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "FastVideo/Wan-Syn_77x448x832_600k" --repo_type "dataset"
-5
View File
@@ -28,11 +28,6 @@ For an example running DMD+VSA inference:
python examples/inference/basic/basic_dmd.py
```
For the typed config/request path added during the inference API refactor:
```
python examples/inference/basic/basic_dmd_new_api.py
```
## Basic Walkthrough
All you need to generate videos using multi-gpus from state-of-the-art diffusion pipelines is the following few lines!
@@ -1,98 +0,0 @@
import os
import time
from fastvideo import VideoGenerator
from fastvideo.api import (
EngineConfig,
GenerationRequest,
GeneratorConfig,
OffloadConfig,
OutputConfig,
PipelineSelection,
)
OUTPUT_PATH = "video_samples_dmd2_typed"
def main():
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
model_name = "FastVideo/FastWan2.1-T2V-1.3B-Diffusers"
generator_config = GeneratorConfig(
model_path=model_name,
engine=EngineConfig(
num_gpus=1,
use_fsdp_inference=False,
offload=OffloadConfig(
text_encoder=True,
pin_cpu_memory=True,
dit=False,
vae=False,
),
),
# PR 2 still routes a few advanced inference knobs through the
# compatibility bridge until they get first-class typed fields.
pipeline=PipelineSelection(
experimental={
"VSA_sparsity": 0.8,
},
),
)
load_start_time = time.perf_counter()
generator = VideoGenerator.from_config(generator_config)
load_end_time = time.perf_counter()
load_time = load_end_time - load_start_time
prompt = (
"A neon-lit alley in futuristic Tokyo during a heavy rainstorm at night. "
"The puddles reflect glowing signs in kanji, advertising ramen, karaoke, "
"and VR arcades. A woman in a translucent raincoat walks briskly with an "
"LED umbrella. Steam rises from a street food cart, and a cat darts "
"across the screen. Raindrops are visible on the camera lens, creating "
"a cinematic bokeh effect."
)
request = GenerationRequest(
prompt=prompt,
output=OutputConfig(
output_path=OUTPUT_PATH,
save_video=True,
return_frames=False,
),
)
start_time = time.perf_counter()
result = generator.generate(request)
end_time = time.perf_counter()
gen_time = end_time - start_time
prompt2 = (
"A majestic lion strides across the golden savanna, its powerful frame "
"glistening under the warm afternoon sun. The tall grass ripples gently "
"in the breeze, enhancing the lion's commanding presence. The tone is "
"vibrant, embodying the raw energy of the wild. Low angle, steady "
"tracking shot, cinematic."
)
request2 = GenerationRequest(
prompt=prompt2,
output=OutputConfig(
output_path=OUTPUT_PATH,
save_video=True,
return_frames=False,
),
)
start_time = time.perf_counter()
result2 = generator.generate(request2)
end_time = time.perf_counter()
gen_time2 = end_time - start_time
print(f"Time taken to load model: {load_time} seconds")
print(f"Time taken to generate video: {gen_time} seconds")
print(f"First output written to: {result.video_path}")
print(f"Time taken to generate video2: {gen_time2} seconds")
print(f"Second output written to: {result2.video_path}")
if __name__ == "__main__":
main()
-109
View File
@@ -1,109 +0,0 @@
"""
GEN3C: 3D-aware camera-controlled video generation.
This example generates a video from a single input image with camera control.
The pipeline uses MoGe depth estimation, 3D point cloud forward warping,
and the GEN3C diffusion model.
Requirements:
1. Install MoGe:
pip install git+https://github.com/microsoft/MoGe.git
If you hit `ImportError: libGL.so.1`, install:
sudo apt-get update && sudo apt-get install -y libgl1 libglib2.0-0 libsm6 libxext6 libxrender1
2. Download and convert weights:
huggingface-cli download nvidia/GEN3C-Cosmos-7B --local-dir official_weights/GEN3C-Cosmos-7B
python scripts/checkpoint_conversion/convert_gen3c_to_fastvideo.py \
--source ./official_weights/GEN3C-Cosmos-7B/model.pt \
--output ./converted_weights/GEN3C-Cosmos-7B \
--components-source nvidia/Cosmos-Predict2-2B-Video2World
3. Provide an input image for 3D-conditioned generation.
"""
import argparse
from fastvideo import VideoGenerator
def main():
parser = argparse.ArgumentParser(description="GEN3C video generation")
parser.add_argument("--model_path",
type=str,
default="converted_weights/GEN3C-Cosmos-7B")
parser.add_argument("--image_path",
type=str,
default=None,
help="Input image for 3D cache conditioning")
parser.add_argument("--prompt",
type=str,
default="A slow camera pan over a sunlit landscape.")
parser.add_argument(
"--negative_prompt",
type=str,
default=(
"The video captures a series of frames showing ugly scenes, static with no motion, motion blur, "
"over-saturation, shaky footage, low resolution, grainy texture, pixelated images, poorly lit areas, "
"underexposed and overexposed scenes, poor color balance, washed out colors, choppy sequences, "
"jerky movements, low frame rate, artifacting, color banding, unnatural transitions, outdated special "
"effects, fake elements, unconvincing visuals, poorly edited content, jump cuts, visual noise, and "
"flickering. Overall, the video is of poor quality."
),
)
parser.add_argument("--trajectory",
type=str,
default="left",
choices=[
"left", "right", "up", "down", "zoom_in",
"zoom_out", "clockwise", "counterclockwise", "none"
])
parser.add_argument("--movement_distance", type=float, default=0.3)
parser.add_argument("--camera_rotation",
type=str,
default="center_facing",
choices=[
"center_facing", "no_rotation",
"trajectory_aligned"
])
parser.add_argument("--height", type=int, default=704)
parser.add_argument("--width", type=int, default=1280)
parser.add_argument("--num_frames", type=int, default=121)
parser.add_argument("--num_inference_steps", type=int, default=35)
parser.add_argument("--guidance_scale", type=float, default=1.0)
parser.add_argument("--output_path",
type=str,
default="outputs_video/gen3c.mp4")
parser.add_argument("--seed", type=int, default=42)
args = parser.parse_args()
generator = VideoGenerator.from_pretrained(
args.model_path,
num_gpus=1,
use_fsdp_inference=False,
dit_cpu_offload=False,
vae_cpu_offload=True,
text_encoder_cpu_offload=True,
pin_cpu_memory=True,
)
video = generator.generate_video(
args.prompt,
negative_prompt=args.negative_prompt,
image_path=args.image_path,
trajectory_type=args.trajectory,
movement_distance=args.movement_distance,
camera_rotation=args.camera_rotation,
height=args.height,
width=args.width,
num_frames=args.num_frames,
num_inference_steps=args.num_inference_steps,
guidance_scale=args.guidance_scale,
fps=24,
seed=args.seed,
output_path=args.output_path,
save_video=True,
)
generator.shutdown()
if __name__ == "__main__":
main()
+1 -75
View File
@@ -1,79 +1,5 @@
# Optimization Examples
## Wan 2.1 QAT Attention 14B Inference
Use these files for Wan 2.1 14B inference with the `ATTN_QAT_INFER` backend:
- `examples/inference/optimizations/download_14B_qat.sh`
- `examples/inference/optimizations/attn_qat_inference_example.py`
### 1. Download the 14B QAT checkpoint
The helper script downloads the QAT safetensors from
`FastVideo/14B_qat_400` into `checkpoints/14B_qat_400` by default.
Prerequisites:
- `huggingface_hub` installed, for example: `uv pip install huggingface_hub`
- access to the model repo if it is private or gated: `huggingface-cli login`
Run:
```bash
bash examples/inference/optimizations/download_14B_qat.sh
python examples/inference/optimizations/attention_example.py
```
To download into a custom directory, pass it as the first argument:
```bash
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
```
### 2. Edit the inference example for Wan 2.1 14B
Open `examples/inference/optimizations/attn_qat_inference_example.py` and
update these two values:
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
2. Replace the placeholder
`init_weights_from_safetensors="safetensors_path"` with the directory that
contains the downloaded `.safetensors` files.
Example:
```python
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
num_gpus=1,
use_fsdp_inference=True,
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False,
init_weights_from_safetensors="checkpoints/14B_qat_400",
)
```
The script already sets:
```python
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
```
### 3. Run the example
```bash
python examples/inference/optimizations/attn_qat_inference_example.py
```
The generated videos are written to `video_samples/` by default.
### Notes
- `ATTN_QAT_INFER` requires the in-repo `fastvideo-kernel` build to expose the
`attn_qat_infer` package.
- If you have not built the kernel yet, run `cd fastvideo-kernel && ./build.sh`
first.
- If you keep the example on the `1.3B` base model while loading the 14B QAT
weights, the model/config will not match.
@@ -1,54 +0,0 @@
from fastvideo import VideoGenerator
import os
from pathlib import Path
# from fastvideo.configs.sample import SamplingParam
OUTPUT_PATH = "video_samples"
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
CHECKPOINT_PATH = Path(__file__).parent.parent.parent
def main():
# FastVideo will automatically use the optimal default arguments for the
# model.
# If a local path is provided, FastVideo will make a best effort
# attempt to identify the optimal arguments.
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
# FastVideo will automatically handle distributed setup
num_gpus=1,
use_fsdp_inference=True,
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
# image_encoder_cpu_offload=False,
# Load custom weights from checkpoint
init_weights_from_safetensors="safetensors_path"
)
# sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
# sampling_param.num_frames = 45
# sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
# Generate videos with the same simple API, regardless of GPU count
prompt = (
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
"wide with interest. The playful yet serene atmosphere is complemented by soft "
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
)
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True)
# video = generator.generate_video(prompt, sampling_param=sampling_param, output_path="wan_t2v_videos/")
# Generate another video with a different prompt, without reloading the
# model!
prompt2 = (
"A majestic lion strides across the golden savanna, its powerful frame "
"glistening under the warm afternoon sun. The tall grass ripples gently in "
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
"cinematic.")
video2 = generator.generate_video(prompt2, output_path=OUTPUT_PATH, save_video=True)
if __name__ == "__main__":
main()
@@ -1,58 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." && pwd)"
HF_REPO_ID="${HF_REPO_ID:-FastVideo/14B_qat_400}"
HF_REVISION="${HF_REVISION:-main}"
LOCAL_DIR="${1:-${REPO_ROOT}/checkpoints/14B_qat_400}"
PYTHON_BIN="${PYTHON:-python}"
if ! command -v "${PYTHON_BIN}" >/dev/null 2>&1; then
echo "Python executable not found: ${PYTHON_BIN}" >&2
exit 1
fi
if ! "${PYTHON_BIN}" -c "import huggingface_hub" >/dev/null 2>&1; then
echo "Missing dependency: huggingface_hub" >&2
echo "Install it with: uv pip install huggingface_hub" >&2
exit 1
fi
mkdir -p "${LOCAL_DIR}"
echo "Downloading ${HF_REPO_ID}@${HF_REVISION}"
echo "Local directory: ${LOCAL_DIR}"
"${PYTHON_BIN}" -c '
import argparse
from huggingface_hub import snapshot_download
parser = argparse.ArgumentParser()
parser.add_argument("--repo-id", required=True)
parser.add_argument("--revision", required=True)
parser.add_argument("--local-dir", required=True)
args = parser.parse_args()
snapshot_download(
repo_id=args.repo_id,
revision=args.revision,
repo_type="model",
local_dir=args.local_dir,
local_dir_use_symlinks=False,
resume_download=True,
)
' \
--repo-id "${HF_REPO_ID}" \
--revision "${HF_REVISION}" \
--local-dir "${LOCAL_DIR}"
echo
echo "Download complete."
echo "Use this in your inference script:"
echo "init_weights_from_safetensors=\"${LOCAL_DIR}\""
echo
echo "If the repo is private or gated, make sure you are logged in with:"
echo "huggingface-cli login"
@@ -1,88 +0,0 @@
import torch
from fastvideo import VideoGenerator
from fastvideo.configs.pipelines.base import PipelineConfig
OUTPUT_PATH = "video_samples"
def main():
print("=== FP4 Quantization Video Generation Example ===")
if not torch.cuda.is_available():
print("Warning: CUDA not available. FP4 quantization requires GPU.")
return
gpu_capability = torch.cuda.get_device_capability()
if gpu_capability[0] < 9: # H100 and newer
print(f"Warning: GPU capability {gpu_capability} may not support FP4. Recommended: 9.0+")
print(f"GPU: {torch.cuda.get_device_name()}")
print(f"GPU Capability: {gpu_capability}")
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
# model_id = "Wan-AI/Wan2.1-T2V-14B-Diffusers"
pipeline_config = PipelineConfig.from_pretrained(model_id)
pipeline_config.dit_precision = "bf16"
print("\nLoading model with FP4 quantization...")
generator = VideoGenerator.from_pretrained(
model_id,
pipeline_config=pipeline_config,
num_gpus=1,
use_fsdp_inference=True,
transformer_quant="fp4",
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False,
)
print("FP4 configuration applied. Generating videos...")
print("\n=== Generating Video with FP4 Quantization ===")
prompt1 = (
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
"wide with interest. The playful yet serene atmosphere is complemented by soft "
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
)
print(f"Prompt: {prompt1}")
print("Generating video...")
try:
video1 = generator.generate_video(
prompt1,
output_path=OUTPUT_PATH,
save_video=True,
)
print("✓ First video generated successfully with FP4 quantization!")
# # Generate a second video to show the model can be reused
prompt2 = (
"A majestic lion strides across the golden savanna, its powerful frame "
"glistening under the warm afternoon sun. The tall grass ripples gently in "
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
"cinematic."
)
print(f"\nGenerating second video...")
print(f"Prompt: {prompt2}")
video2 = generator.generate_video(
prompt2,
output_path=OUTPUT_PATH,
save_video=True,
)
print("✓ Second video generated successfully with FP4 quantization!")
except Exception as e:
print(f"Error during video generation: {e}")
return
print(f"Videos saved to: {OUTPUT_PATH}")
if __name__ == "__main__":
main()
@@ -1,47 +1,27 @@
#!/bin/bash
#SBATCH --job-name=wan_t2v_1.3B_finetune
#SBATCH --partition=all
#SBATCH --nodes=1
#SBATCH --gres=gpu:4
#SBATCH --ntasks-per-node=1
#SBATCH --output=logs/wan_t2v_1.3B_finetune.out
#SBATCH --error=logs/wan_t2v_1.3B_finetune.err
source .venv/bin/activate
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
export TOKENIZERS_PARALLELISM=false
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
export WANDB_API_KEY=YOUR_WANDB_API_KEY
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
DATA_DIR=data/Wan-Syn_77x448x832_600k
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
NUM_GPUS=1
DATA_DIR="data/crush-smol_processed_t2v/combined_parquet_dataset/"
VALIDATION_DATASET_FILE="$(dirname "$0")/validation.json"
NUM_GPUS=4
# export CUDA_VISIBLE_DEVICES=4,5
set -euo pipefail
# ---- torchrun rendezvous (multi-node) ----
# Launch ONE torchrun per node (via srun) and let torchrun spawn 4 workers per node.
MASTER_ADDR="$(scontrol show hostnames "$SLURM_JOB_NODELIST" | head -n 1)"
MASTER_PORT="${MASTER_PORT:-29500}"
export MASTER_ADDR MASTER_PORT
# Training arguments
training_args=(
--tracker_project_name "wan_t2v_finetune_qat"
--output_dir "checkpoints/wan_t2v_finetune_1.3B_77"
--max_train_steps 4000
--tracker_project_name "wan_t2v_finetune"
--output_dir "checkpoints/wan_t2v_finetune"
--max_train_steps 5000
--train_batch_size 1
--train_sp_batch_size 1
--gradient_accumulation_steps 1
--gradient_accumulation_steps 8
--num_latent_t 20
--num_height 448
--num_height 480
--num_width 832
--num_frames 77
--enable_gradient_checkpointing_type "full"
@@ -50,7 +30,7 @@ training_args=(
# Parallel arguments
parallel_args=(
--num_gpus $NUM_GPUS
--sp_size 1
--sp_size $NUM_GPUS
--tp_size 1
--hsdp_replicate_dim 1
--hsdp_shard_dim $NUM_GPUS
@@ -65,7 +45,7 @@ model_args=(
# Dataset arguments
dataset_args=(
--data_path $DATA_DIR
--dataloader_num_workers 4
--dataloader_num_workers 1
)
# Validation arguments
@@ -74,16 +54,16 @@ validation_args=(
--validation_dataset_file $VALIDATION_DATASET_FILE
--validation_steps 200
--validation_sampling_steps "50"
--validation_guidance_scale "5.0"
--validation_guidance_scale "3.0"
)
# Optimizer arguments
optimizer_args=(
--learning_rate 1e-6
--learning_rate 5e-5
--mixed_precision "bf16"
--weight_only_checkpointing_steps 1000
--training_state_checkpointing_steps 1000
--weight_decay 0.01
--weight_decay 1e-4
--max_grad_norm 1.0
)
@@ -92,24 +72,23 @@ miscellaneous_args=(
--inference_mode False
--checkpoints_total_limit 3
--training_cfg_rate 0.1
--multi_phased_distill_schedule "4000-1"
--not_apply_cfg_solver
--dit_precision "fp32"
--num_euler_timesteps 50
--ema_start_step 0
--flow_shift 5
--seed 1000
--enable_gradient_checkpointing_type "full"
# --resume_from_checkpoint "checkpoints/wan_t2v_finetune/checkpoint-2500"
)
srun --nodes="$SLURM_NNODES" --ntasks="$SLURM_NNODES" --ntasks-per-node=1 \
torchrun \
--nnodes "$SLURM_NNODES" \
--nproc_per_node 4 \
--rdzv_backend c10d \
--rdzv_endpoint "${MASTER_ADDR}:${MASTER_PORT}" \
--rdzv_id "$SLURM_JOB_ID" \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
torchrun \
--nnodes 1 \
--nproc_per_node $NUM_GPUS \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
@@ -1,124 +0,0 @@
#!/bin/bash
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
#SBATCH --partition=all
#SBATCH --nodes=4
#SBATCH --gres=gpu:4
#SBATCH --ntasks-per-node=1
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
source .venv/bin/activate
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
export TOKENIZERS_PARALLELISM=false
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
export WANDB_API_KEY=YOUR_WANDB_API_KEY
# Use node-local Triton cache to avoid stale file handle errors on shared filesystems
export TRITON_CACHE_DIR="/tmp/triton_cache_${SLURM_JOB_ID}_${SLURM_NODEID}"
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
DATA_DIR=YOUR_DATA_DIR
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
NUM_GPUS=16
# export CUDA_VISIBLE_DEVICES=4,5
set -euo pipefail
# ---- torchrun rendezvous (multi-node) ----
# 1. Get the hostname of the first node (Master)
nodes=( $( scontrol show hostnames $SLURM_JOB_NODELIST ) )
nodes_array=($nodes)
head_node=${nodes_array[0]}
MASTER_ADDR=$(srun --nodes=1 --ntasks=1 -w "$head_node" hostname --ip-address)
MASTER_PORT=29500
# 2. Get the node count automatically
NNODES=$SLURM_NNODES
GPUS_PER_NODE=$SLURM_GPUS_ON_NODE
NUM_GPUS=$((NNODES * GPUS_PER_NODE))
echo "MASTER_ADDR=$MASTER_ADDR MASTER_PORT=$MASTER_PORT NNODES=$NNODES"
# Training arguments
training_args=(
--tracker_project_name "wan_t2v_finetune_qat"
--output_dir "checkpoints/wan_1.3B_t2v_finetune_qat"
--max_train_steps 4000
--train_batch_size 1
--train_sp_batch_size 1
--gradient_accumulation_steps 1
--num_latent_t 20
--num_height 448
--num_width 832
--num_frames 77
--enable_gradient_checkpointing_type "full" # if OOM enable this
)
# Parallel arguments
parallel_args=(
--num_gpus $NUM_GPUS
--sp_size 1
--tp_size 1
--hsdp_replicate_dim $NUM_GPUS
--hsdp_shard_dim 1
)
# Model arguments
model_args=(
--model_path $MODEL_PATH
--pretrained_model_name_or_path $MODEL_PATH
)
# Dataset arguments
dataset_args=(
--data_path "$DATA_DIR"
--dataloader_num_workers 4
)
# Validation arguments
validation_args=(
--log_validation
--validation_dataset_file $VALIDATION_DATASET_FILE
--validation_steps 200
--validation_sampling_steps "50"
--validation_guidance_scale "5.0"
)
# Optimizer arguments
optimizer_args=(
--learning_rate 1e-6
--mixed_precision "bf16"
--weight_only_checkpointing_steps 200
--training_state_checkpointing_steps 200
--weight_decay 0.01
--max_grad_norm 1.0
)
# Miscellaneous arguments
miscellaneous_args=(
--inference_mode False
--checkpoints_total_limit 3
--training_cfg_rate 0.1
--dit_precision "fp32"
--ema_start_step 0
--flow_shift 1
--seed 1000
)
srun torchrun \
--nnodes $NNODES \
--nproc_per_node $GPUS_PER_NODE \
--node_rank $SLURM_PROCID \
--rdzv_backend c10d \
--rdzv_endpoint $MASTER_ADDR:$MASTER_PORT \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
@@ -1,131 +1,31 @@
{
"data": [
{
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"caption": "A large metal cylinder is seen pressing down on a pile of Oreo cookies, flattening them as if they were under a hydraulic press.",
"image_path": null,
"video_path": null,
"num_inference_steps": 50,
"height": 480,
"width": 832,
"num_frames": 77
},
{
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"caption": "A large metal cylinder is seen compressing colorful clay into a compact shape, demonstrating the power of a hydraulic press.",
"image_path": null,
"video_path": null,
"num_inference_steps": 50,
"height": 480,
"width": 832,
"num_frames": 77
},
{
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"caption": "A large metal cylinder is seen pressing down on a pile of colorful candies, flattening them as if they were under a hydraulic press. The candies are crushed and broken into small pieces, creating a mess on the table.",
"image_path": null,
"video_path": null,
"num_inference_steps": 50,
"height": 480,
"width": 832,
"num_frames": 77
},
{
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
} ]
}
]
}
@@ -1,123 +0,0 @@
#!/bin/bash
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
#SBATCH --partition=main
#SBATCH --nodes=4
#SBATCH --ntasks-per-node=1
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=128
#SBATCH --mem=1440G
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
#SBATCH --exclusive
source ~/conda/miniconda/bin/activate
conda activate matthew-fv
# Basic Info
export WANDB_MODE="online"
export NCCL_P2P_DISABLE=1
export TORCH_NCCL_ENABLE_MONITORING=0
# different cache dir for different processes
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
export MASTER_PORT=29500
export NODE_RANK=$SLURM_PROCID
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
export MASTER_ADDR=${nodes[0]}
export CUDA_VISIBLE_DEVICES=$SLURM_LOCALID
export TOKENIZERS_PARALLELISM=false
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
echo "MASTER_ADDR: $MASTER_ADDR"
echo "NODE_RANK: $NODE_RANK"
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
export WANDB_API_KEY=YOUR_WANDB_API_KEY
MODEL_PATH="Wan-AI/Wan2.1-T2V-14B-Diffusers"
DATA_DIR=YOUR_DATA_DIR
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
NUM_GPUS_PER_NODE=8
TOTAL_GPUS=$((NUM_GPUS_PER_NODE * SLURM_JOB_NUM_NODES))
# export CUDA_VISIBLE_DEVICES=4,5
# Training arguments
training_args=(
--tracker_project_name "wan_t2v_finetune_qat"
--output_dir "checkpoints/wan_14B_t2v_finetune_qat"
--max_train_steps 4000
--train_batch_size 1
--train_sp_batch_size 1
--gradient_accumulation_steps 1
--num_latent_t 20
--num_height 768
--num_width 1280
--num_frames 77
--enable_gradient_checkpointing_type "full" # if OOM enable this
)
# Parallel arguments
parallel_args=(
--num_gpus $TOTAL_GPUS
--sp_size 4
--tp_size 1
--hsdp_replicate_dim 4
--hsdp_shard_dim 8
)
# Model arguments
model_args=(
--model_path $MODEL_PATH
--pretrained_model_name_or_path $MODEL_PATH
)
# Dataset arguments
dataset_args=(
--data_path "$DATA_DIR"
--dataloader_num_workers 4
)
# Validation arguments
validation_args=(
# --log_validation
--validation_dataset_file $VALIDATION_DATASET_FILE
--validation_steps 200
--validation_sampling_steps "50"
--validation_guidance_scale "5.0"
)
# Optimizer arguments
optimizer_args=(
--learning_rate 1e-6
--mixed_precision "bf16"
--weight_only_checkpointing_steps 200
--training_state_checkpointing_steps 200
--weight_decay 0.01
--max_grad_norm 1.0
)
# Miscellaneous arguments
miscellaneous_args=(
--inference_mode False
--checkpoints_total_limit 3
--training_cfg_rate 0.1
--dit_precision "fp32"
--ema_start_step 0
--flow_shift 5
--seed 1000
)
srun torchrun \
--nnodes $SLURM_JOB_NUM_NODES \
--nproc_per_node $NUM_GPUS_PER_NODE \
--node_rank $SLURM_PROCID \
--rdzv_backend=c10d \
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
@@ -1,131 +0,0 @@
{
"data": [
{
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
} ]
}
+9 -191
View File
@@ -12,18 +12,6 @@ else()
enable_language(CUDA)
# Ensure CUDA toolkit targets (CUDA::cudart, CUDA::cuda_driver, etc.) are available.
find_package(CUDAToolkit REQUIRED)
if(NOT DEFINED CUDA_TOOLKIT_ROOT_DIR)
if(DEFINED CUDAToolkit_ROOT)
set(CUDA_TOOLKIT_ROOT_DIR "${CUDAToolkit_ROOT}" CACHE PATH
"CUDA toolkit root directory" FORCE)
elseif(DEFINED ENV{CUDAToolkit_ROOT})
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDAToolkit_ROOT}" CACHE PATH
"CUDA toolkit root directory" FORCE)
elseif(DEFINED ENV{CUDA_HOME})
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDA_HOME}" CACHE PATH
"CUDA toolkit root directory" FORCE)
endif()
endif()
endif()
# Import common utils if needed, but we keep it simple for now
@@ -31,46 +19,13 @@ endif()
# Find Python and Torch
find_package(Python COMPONENTS Interpreter Development.Module REQUIRED)
# Locate the installed torch package without importing it. This keeps CMake
# configure working even on nodes where CUDA runtime libraries are not yet on
# the dynamic loader path.
# Robustly find Torch include paths using Python
execute_process(
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('platlib'))"
OUTPUT_VARIABLE PYTHON_PLATLIB
COMMAND "${Python_EXECUTABLE}" -c "import torch; from torch.utils.cpp_extension import include_paths; print(';'.join(include_paths()))"
OUTPUT_VARIABLE TORCH_INCLUDE_PATHS
OUTPUT_STRIP_TRAILING_WHITESPACE
)
execute_process(
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('purelib'))"
OUTPUT_VARIABLE PYTHON_PURELIB
OUTPUT_STRIP_TRAILING_WHITESPACE
)
set(TORCH_PYTHON_PACKAGE_DIR "")
foreach(_candidate
"${PYTHON_PLATLIB}/torch"
"${PYTHON_PURELIB}/torch"
)
if(EXISTS "${_candidate}")
set(TORCH_PYTHON_PACKAGE_DIR "${_candidate}")
break()
endif()
endforeach()
if(NOT TORCH_PYTHON_PACKAGE_DIR)
message(FATAL_ERROR "Could not locate the installed torch Python package.")
endif()
list(APPEND TORCH_INCLUDE_DIRS
"${TORCH_PYTHON_PACKAGE_DIR}/include"
"${TORCH_PYTHON_PACKAGE_DIR}/include/torch/csrc/api/include"
)
if(NOT Torch_DIR)
set(_TORCH_CONFIG_DIR "${TORCH_PYTHON_PACKAGE_DIR}/share/cmake/Torch")
if(EXISTS "${_TORCH_CONFIG_DIR}/TorchConfig.cmake")
set(Torch_DIR "${_TORCH_CONFIG_DIR}" CACHE PATH "Path to Torch CMake config" FORCE)
endif()
endif()
list(APPEND TORCH_INCLUDE_DIRS ${TORCH_INCLUDE_PATHS})
# Find Torch package (still useful for libraries)
find_package(Torch REQUIRED)
@@ -95,21 +50,6 @@ include_directories(
set(FASTVIDEO_KERNEL_BUILD_TK "AUTO" CACHE STRING "Build ThunderKittens kernels: AUTO/ON/OFF")
set_property(CACHE FASTVIDEO_KERNEL_BUILD_TK PROPERTY STRINGS AUTO ON OFF)
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "AUTO")
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 AND NOT DEFINED CACHE{FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER})
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "${FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3}")
endif()
set(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER "${_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT}" CACHE STRING
"Build attn_qat_infer Blackwell inference kernels: AUTO/ON/OFF")
set_property(CACHE FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER PROPERTY STRINGS AUTO ON OFF)
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3)
message(DEPRECATION
"FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 is deprecated. "
"Use FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER instead.")
endif()
# Prefer environment variable (used by CI) if CMake var is not explicitly set.
if(NOT DEFINED TORCH_CUDA_ARCH_LIST AND DEFINED ENV{TORCH_CUDA_ARCH_LIST})
set(TORCH_CUDA_ARCH_LIST "$ENV{TORCH_CUDA_ARCH_LIST}")
@@ -117,7 +57,6 @@ endif()
message(STATUS "TORCH_CUDA_ARCH_LIST (cmake/env): ${TORCH_CUDA_ARCH_LIST}")
message(STATUS "FASTVIDEO_KERNEL_BUILD_TK: ${FASTVIDEO_KERNEL_BUILD_TK}")
message(STATUS "FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER: ${FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER}")
set(ENABLE_TK_KERNELS OFF)
if(FASTVIDEO_KERNEL_BUILD_TK STREQUAL "ON")
@@ -152,54 +91,6 @@ else()
message(STATUS "ThunderKittens kernels: DISABLED (will use Triton fallbacks at runtime)")
endif()
set(ENABLE_ATTN_QAT_INFER OFF)
if(GPU_BACKEND STREQUAL "ROCM")
message(STATUS "attn_qat_infer kernels: DISABLED (ROCm build)")
else()
set(_WANTS_ATTN_QAT_INFER OFF)
if(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "ON")
set(_WANTS_ATTN_QAT_INFER ON)
elseif(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "AUTO")
if(TORCH_CUDA_ARCH_LIST)
string(REGEX MATCH
"(^|[; ,])((12\\.0a)|(120a)|(sm_120a))([; ,]|$)"
_HAS_120A "${TORCH_CUDA_ARCH_LIST}")
if(_HAS_120A)
set(_WANTS_ATTN_QAT_INFER ON)
endif()
else()
execute_process(
COMMAND "${Python_EXECUTABLE}" -c
"import torch; print('1' if (torch.cuda.is_available() and torch.version.cuda and torch.cuda.get_device_capability()[0] >= 12) else '0')"
OUTPUT_VARIABLE _LOCAL_HAS_BLACKWELL
OUTPUT_STRIP_TRAILING_WHITESPACE
ERROR_QUIET
)
if(_LOCAL_HAS_BLACKWELL STREQUAL "1")
set(_WANTS_ATTN_QAT_INFER ON)
endif()
endif()
endif()
if(_WANTS_ATTN_QAT_INFER)
if(CUDAToolkit_VERSION VERSION_LESS 12.8)
message(WARNING
"attn_qat_infer kernels require CUDA Toolkit 12.8+. "
"Skipping because CUDAToolkit_VERSION=${CUDAToolkit_VERSION}.")
else()
set(ENABLE_ATTN_QAT_INFER ON)
endif()
endif()
if(ENABLE_ATTN_QAT_INFER)
message(STATUS "attn_qat_infer kernels: ENABLED")
else()
message(STATUS
"attn_qat_infer kernels: DISABLED "
"(requires CUDA 12.8+ and Blackwell sm_120a)")
endif()
endif()
# Always try to build the extension if CUDA is available, but conditionally add sources/flags
set(BUILD_CXX_KERNELS ON)
@@ -270,15 +161,12 @@ if(BUILD_CXX_KERNELS)
# Also link against libtorch_python to satisfy Python-binding symbols
# (e.g., torch::PyWarningHandler) required by torch/extension.h.
file(GLOB TORCH_PYTHON_LIBRARY_CANDIDATES
"${TORCH_PYTHON_PACKAGE_DIR}/lib/libtorch_python*"
execute_process(
COMMAND "${Python_EXECUTABLE}" -c "import torch; from pathlib import Path; p=Path(torch.__file__).parent/'lib'; m=sorted(p.glob('libtorch_python*')); print(str(m[0]) if m else '')"
OUTPUT_VARIABLE TORCH_PYTHON_LIBRARY_PATH
OUTPUT_STRIP_TRAILING_WHITESPACE
ERROR_QUIET
)
list(LENGTH TORCH_PYTHON_LIBRARY_CANDIDATES _TORCH_PYTHON_LIBRARY_COUNT)
if(_TORCH_PYTHON_LIBRARY_COUNT GREATER 0)
list(GET TORCH_PYTHON_LIBRARY_CANDIDATES 0 TORCH_PYTHON_LIBRARY_PATH)
else()
set(TORCH_PYTHON_LIBRARY_PATH "")
endif()
if(TORCH_PYTHON_LIBRARY_PATH)
message(STATUS "TORCH_PYTHON_LIBRARY_PATH: ${TORCH_PYTHON_LIBRARY_PATH}")
target_link_libraries(fastvideo_kernel_ops PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
@@ -295,73 +183,3 @@ if(BUILD_CXX_KERNELS)
install(TARGETS fastvideo_kernel_ops LIBRARY DESTINATION fastvideo_kernel/_C)
endif()
if(ENABLE_ATTN_QAT_INFER)
set(ATTN_QAT_INFER_DIR ${CMAKE_SOURCE_DIR}/attn_qat_infer)
set(ATTN_QAT_INFER_INCLUDE_DIRS
${ATTN_QAT_INFER_DIR}
${CMAKE_SOURCE_DIR}/include/cutlass/include
${CMAKE_SOURCE_DIR}/include/cutlass/tools/util/include
${TORCH_INCLUDE_DIRS}
)
set(ATTN_QAT_INFER_CUDA_FLAGS
"-O3"
"-std=c++17"
"-U__CUDA_NO_HALF_OPERATORS__"
"-U__CUDA_NO_HALF_CONVERSIONS__"
"-U__CUDA_NO_BFLOAT16_OPERATORS__"
"-U__CUDA_NO_BFLOAT16_CONVERSIONS__"
"-U__CUDA_NO_BFLOAT162_OPERATORS__"
"-U__CUDA_NO_BFLOAT162_CONVERSIONS__"
"--expt-relaxed-constexpr"
"--expt-extended-lambda"
"--use_fast_math"
"--ptxas-options=--verbose,--warn-on-local-memory-usage"
"-lineinfo"
"-DCUTLASS_DEBUG_TRACE_LEVEL=0"
"-DNDEBUG"
"-DQBLKSIZE=128"
"-DKBLKSIZE=128"
"-DCTA256"
"-DDQINRMEM"
)
Python_add_library(fp4attn_cuda MODULE WITH_SOABI
attn_qat_infer/blackwell/api.cu
)
target_include_directories(fp4attn_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
target_compile_definitions(fp4attn_cuda PRIVATE TORCH_EXTENSION_NAME=fp4attn_cuda)
target_compile_options(fp4attn_cuda PRIVATE
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
)
set_target_properties(fp4attn_cuda PROPERTIES
CUDA_ARCHITECTURES "120a"
CXX_STANDARD 17
CUDA_STANDARD 17
)
target_link_libraries(fp4attn_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
Python_add_library(fp4quant_cuda MODULE WITH_SOABI
attn_qat_infer/quantization/fp4_quantization_4d.cu
)
target_include_directories(fp4quant_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
target_compile_definitions(fp4quant_cuda PRIVATE TORCH_EXTENSION_NAME=fp4quant_cuda)
target_compile_options(fp4quant_cuda PRIVATE
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
)
set_target_properties(fp4quant_cuda PROPERTIES
CUDA_ARCHITECTURES "120a"
CXX_STANDARD 17
CUDA_STANDARD 17
)
target_link_libraries(fp4quant_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
if(TORCH_PYTHON_LIBRARY_PATH)
target_link_libraries(fp4attn_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
target_link_libraries(fp4quant_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
endif()
install(TARGETS fp4attn_cuda LIBRARY DESTINATION .)
install(TARGETS fp4quant_cuda LIBRARY DESTINATION .)
endif()
-1
View File
@@ -2,6 +2,5 @@ include LICENSE
include README.md
include pyproject.toml
recursive-include python/fastvideo_kernel *.py
recursive-include attn_qat_infer *.py *.cu *.cuh *.cpp *.h
recursive-include csrc *.cu *.cuh *.cpp *.h
recursive-include include/tk *.cu *.cuh *.cpp *.h *.src
-17
View File
@@ -20,11 +20,6 @@ cd fastvideo-kernel
./build.sh
```
On supported Blackwell environments, the same install also packages
`attn_qat_infer` and builds its `fp4attn_cuda` / `fp4quant_cuda`
extensions directly from `fastvideo-kernel/attn_qat_infer/`. This path
requires CUDA Toolkit 12.8+ and targets `sm_120a`.
### Rocm Build
If you are in a rocm environment without the compilation toolchaine of CUDA.
@@ -63,18 +58,6 @@ cd fastvideo-kernel
python benchmarks/bench_vsa.py --batch_size 1 --num_heads 16 --head_dim 128 --q_seq_lens 49152 --topk 64
```
### Attn QAT Attention Benchmarks
The Attn QAT microbenchmarks now live alongside the kernel package:
```bash
cd fastvideo-kernel
python benchmarks/benchmark_flashattn2.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
python benchmarks/benchmark_sageattn3.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
python benchmarks/benchmark_blockscaled_fp4_attn.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
python benchmarks/benchmark_combined.py --output benchmark_attention.png
```
### TurboDiffusion Kernels
This package also includes kernels from [TurboDiffusion](https://github.com/thu-ml/TurboDiffusion), including INT8 GEMM, Quantization, RMSNorm and LayerNorm.
@@ -1,16 +0,0 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
from .api import sageattn_blackwell
-185
View File
@@ -1,185 +0,0 @@
# Modified from the original SageATtention3 code
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import triton
import triton.language as tl
import torch.nn.functional as F
from typing import Tuple
from torch.nn.functional import scaled_dot_product_attention as sdpa
import fp4attn_cuda
import fp4quant_cuda
# Centralized block size configuration for sageattn_blackwell kernels
# These should match the values in fastvideo/attention/backends/sageattn/blackwell/block_config.h
BLOCK_M = 128 # Block size for M dimension (query sequence length)
BLOCK_N = 128 # Block size for N dimension (key/value sequence length)
@triton.jit
def group_mean_kernel(
q_ptr,
q_out_ptr,
qm_out_ptr,
B, H, L, D: tl.constexpr,
stride_qb, stride_qh, stride_ql, stride_qd,
stride_qmb, stride_qmh, stride_qml, stride_qmd,
GROUP_SIZE: tl.constexpr
):
pid_b = tl.program_id(0)
pid_h = tl.program_id(1)
pid_group = tl.program_id(2)
group_start = pid_group * GROUP_SIZE
offsets = group_start + tl.arange(0, GROUP_SIZE)
q_offsets = pid_b * stride_qb + pid_h * stride_qh + offsets[:, None] * stride_ql + tl.arange(0, D)[None, :] * stride_qd
q_group = tl.load(q_ptr + q_offsets)
qm_group = tl.sum(q_group, axis=0) / GROUP_SIZE
q_group = q_group - qm_group
tl.store(q_out_ptr + q_offsets, q_group)
qm_offset = pid_b * stride_qmb + pid_h * stride_qmh + pid_group * stride_qml + tl.arange(0, D) * stride_qmd
tl.store(qm_out_ptr + qm_offset, qm_group)
def triton_group_mean(q: torch.Tensor):
B, H, L, D = q.shape
GROUP_SIZE = BLOCK_M
num_groups = L // GROUP_SIZE
q_out = torch.empty_like(q) # [B, H, L, D]
qm = torch.empty(B, H, num_groups, D, device=q.device, dtype=q.dtype)
grid = (B, H, num_groups)
group_mean_kernel[grid](
q, q_out, qm,
B, H, L, D,
q.stride(0), q.stride(1), q.stride(2), q.stride(3),
qm.stride(0), qm.stride(1), qm.stride(2), qm.stride(3),
GROUP_SIZE=GROUP_SIZE
)
return q_out, qm
def preprocess_qkv(q: torch.Tensor, k: torch.Tensor, v: torch.Tensor, per_block_mean: bool = True, enable_smoothing_q: bool = False, enable_smoothing_k: bool = False):
def pad_to_block_size(x):
L = x.size(2)
pad_len = (BLOCK_M - L % BLOCK_M) % BLOCK_M
if pad_len == 0:
return x.contiguous()
return F.pad(x, (0, 0, 0, pad_len), value=0).contiguous()
if enable_smoothing_k:
k -= k.mean(dim=-2, keepdim=True)
q, k, v = map(lambda x: pad_to_block_size(x), [q, k, v])
if per_block_mean and enable_smoothing_q:
q, qm = triton_group_mean(q)
elif enable_smoothing_q:
qm = q.mean(dim=-2, keepdim=True)
q = q - qm
if enable_smoothing_q:
delta_s = torch.matmul(qm, k.transpose(-2, -1)).to(torch.float32).contiguous()
else: # used to disable q smoothing
B, H, L, D = q.shape
delta_s = torch.zeros((B, H, L // BLOCK_M, k.shape[2]), device=q.device, dtype=torch.float32)
return q, k, v, delta_s
def scale_and_quant_fp4(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
assert x.ndim == 4
B, H, N, D = x.shape
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
fp4quant_cuda.scaled_fp4_quant(x, packed_fp4, fp8_scale, 1)
return packed_fp4, fp8_scale
def scale_and_quant_fp4_permute(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
assert x.ndim == 4
B, H, N, D = x.shape
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
fp4quant_cuda.scaled_fp4_quant_permute(x, packed_fp4, fp8_scale, 1)
return packed_fp4, fp8_scale
def scale_and_quant_fp4_transpose(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
assert x.ndim == 4
B, H, N, D = x.shape
packed_fp4 = torch.empty((B, H, D, N // 2), device=x.device, dtype=torch.uint8)
fp8_scale = torch.empty((B, H, D, N // 16), device=x.device, dtype=torch.float8_e4m3fn)
fp4quant_cuda.scaled_fp4_quant_trans(x, packed_fp4, fp8_scale, 1)
return packed_fp4, fp8_scale
def blockscaled_fp4_attn(qlist: Tuple,
klist: Tuple,
vlist: Tuple,
delta_s: torch.Tensor,
KL: int,
is_causal: bool = False,
per_block_mean: bool = True,
is_bf16: bool = True,
single_level_p_quant: bool = False
):
softmax_scale = (qlist[0].shape[-1] * 2) ** (-0.5)
return fp4attn_cuda.fwd(qlist[0], klist[0], vlist[0], qlist[1], klist[1], vlist[1], delta_s, KL, None, softmax_scale, is_causal, per_block_mean, is_bf16, single_level_p_quant)
def sageattn_blackwell(q, k, v, attn_mask = None, is_causal = False, per_block_mean = True, single_level_p_quant = True, **kwargs):
"""
SageAttention3 Blackwell kernel for FP4 attention.
Args:
q: Query tensor [B, H, L, D]
k: Key tensor [B, H, L, D]
v: Value tensor [B, H, L, D]
attn_mask: Attention mask (not used)
is_causal: Whether to use causal masking
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly
(standard per-block FP4 quantization like V, no s_P1).
If False (default), use two-level quantization:
s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1).
**kwargs: Additional arguments (ignored)
Returns:
Output tensor [B, H, L, D]
"""
if q.size(-1) >= 256:
print(f"Unsupported Headdim {q.size(-1)}")
return sdpa(q, k, v, is_causal = is_causal)
QL = q.size(2)
KL = k.size(2)
is_bf16 = q.dtype == torch.bfloat16
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean)
qlist_from_cuda = scale_and_quant_fp4(q)
klist_from_cuda = scale_and_quant_fp4_permute(k)
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
o_fp4 = blockscaled_fp4_attn(
qlist_from_cuda,
klist_from_cuda,
vlist_from_cuda,
delta_s,
KL,
is_causal,
per_block_mean,
is_bf16,
single_level_p_quant
)[0][:, :, :QL, :].contiguous()
return o_fp4
@@ -1 +0,0 @@
__version__ = "3.0.0.b1"
@@ -1,346 +0,0 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
// Include these 2 headers instead of torch/extension.h since we don't need all of the torch headers.
#include <torch/python.h>
#include <torch/nn/functional.h>
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAGuard.h>
#include <cutlass/numeric_types.h>
#include "params.h"
#include "launch.h"
#include "static_switch.h"
#include "block_config.h"
#define CHECK_DEVICE(x) TORCH_CHECK(x.is_cuda(), #x " must be on CUDA")
#define CHECK_SHAPE(x, ...) TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), #x " must have shape (" #__VA_ARGS__ ")")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
void set_params_fprop(Flash_fwd_params &params,
// sizes
const size_t b,
const size_t seqlen_q,
const size_t seqlen_k,
const size_t unpadded_seqlen_k,
const size_t seqlen_q_rounded,
const size_t seqlen_k_rounded,
const size_t h,
const size_t h_k,
const size_t d,
const size_t d_rounded,
// device pointers
const at::Tensor q,
const at::Tensor k,
const at::Tensor v,
const at::Tensor delta_s,
at::Tensor out,
const at::Tensor sfq,
const at::Tensor sfk,
const at::Tensor sfv,
void *cu_seqlens_q_d,
void *cu_seqlens_k_d,
void *seqused_k,
void *p_d,
void *softmax_lse_d,
float p_dropout,
float softmax_scale,
int window_size_left,
int window_size_right,
bool per_block_mean,
bool is_bf16,
bool single_level_p_quant=false,
bool seqlenq_ngroups_swapped=false) {
// Reset the parameters
params = {};
// Set the pointers and strides.
params.q_ptr = q.data_ptr();
params.k_ptr = k.data_ptr();
params.v_ptr = v.data_ptr();
params.delta_s_ptr = delta_s.data_ptr();
params.sfq_ptr = sfq.data_ptr();
params.sfk_ptr = sfk.data_ptr();
params.sfv_ptr = sfv.data_ptr();
// All stride are in elements, not bytes.
params.q_row_stride = q.stride(-2) * 2;
params.k_row_stride = k.stride(-2) * 2;
params.v_row_stride = v.stride(-2) * 2;;
params.q_head_stride = q.stride(-3) * 2;
params.k_head_stride = k.stride(-3) * 2;
params.v_head_stride = v.stride(-3) * 2; // for packed q k v
params.ds_row_stride = delta_s.stride(-2);
params.ds_head_stride = delta_s.stride(-3);
params.sfq_row_stride = sfq.stride(-2);
params.sfk_row_stride = sfk.stride(-2);
params.sfv_row_stride = sfv.stride(-2);
params.sfq_head_stride = sfq.stride(-3);
params.sfk_head_stride = sfk.stride(-3);
params.sfv_head_stride = sfv.stride(-3);
params.o_ptr = out.data_ptr();
params.o_row_stride = out.stride(-2);
params.o_head_stride = out.stride(-3);
if (cu_seqlens_q_d == nullptr) {
params.q_batch_stride = q.stride(0) * 2;
params.k_batch_stride = k.stride(0) * 2;
params.v_batch_stride = v.stride(0) * 2;
params.ds_batch_stride = delta_s.stride(0);
params.sfq_batch_stride = sfq.stride(0);
params.sfk_batch_stride = sfk.stride(0);
params.sfv_batch_stride = sfv.stride(0);
params.o_batch_stride = out.stride(0);
if (seqlenq_ngroups_swapped) {
params.q_batch_stride *= seqlen_q;
params.o_batch_stride *= seqlen_q;
}
}
params.cu_seqlens_q = static_cast<int *>(cu_seqlens_q_d);
params.cu_seqlens_k = static_cast<int *>(cu_seqlens_k_d);
params.seqused_k = static_cast<int *>(seqused_k);
// P = softmax(QK^T)
params.p_ptr = p_d;
// Softmax sum
params.softmax_lse_ptr = softmax_lse_d;
// Set the dimensions.
params.b = b;
params.h = h;
params.h_k = h_k;
params.h_h_k_ratio = h / h_k;
params.seqlen_q = seqlen_q;
params.seqlen_k = seqlen_k;
params.unpadded_seqlen_k = unpadded_seqlen_k;
params.seqlen_q_rounded = seqlen_q_rounded;
params.seqlen_k_rounded = seqlen_k_rounded;
params.d = d;
params.d_rounded = d_rounded;
params.head_divmod = cutlass::FastDivmod(int(h));
// Set the different scale values.
params.scale_softmax = softmax_scale;
params.scale_softmax_log2 = softmax_scale * M_LOG2E;
__half scale_softmax_log2_half = __float2half(params.scale_softmax_log2);
__half2 scale_softmax_log2_half2 = __half2(scale_softmax_log2_half, scale_softmax_log2_half);
params.scale_softmax_log2_half2 = reinterpret_cast<uint32_t&>(scale_softmax_log2_half2);
// Set this to probability of keeping an element to simplify things.
params.p_dropout = 1.f - p_dropout;
// Convert p from float to int so we don't have to convert the random uint to float to compare.
// [Minor] We want to round down since when we do the comparison we use <= instead of <
// params.p_dropout_in_uint = uint32_t(std::floor(params.p_dropout * 4294967295.0));
// params.p_dropout_in_uint16_t = uint16_t(std::floor(params.p_dropout * 65535.0));
params.p_dropout_in_uint8_t = uint8_t(std::floor(params.p_dropout * 255.0));
params.rp_dropout = 1.f / params.p_dropout;
params.scale_softmax_rp_dropout = params.rp_dropout * params.scale_softmax;
TORCH_CHECK(p_dropout < 1.f);
#ifdef FLASHATTENTION_DISABLE_DROPOUT
TORCH_CHECK(p_dropout == 0.0f, "This flash attention build does not support dropout.");
#endif
// Causal is the special case where window_size_right == 0 and window_size_left < 0.
// Local is the more general case where window_size_right >= 0 or window_size_left >= 0.
params.is_causal = window_size_left < 0 && window_size_right == 0;
params.per_block_mean = per_block_mean;
if (per_block_mean) {
params.seqlen_s = seqlen_q;
} else {
params.seqlen_s = flash::BLOCK_M; // size of BLOCK_M
}
if (window_size_left < 0 && window_size_right >= 0) { window_size_left = seqlen_k; }
if (window_size_left >= 0 && window_size_right < 0) { window_size_right = seqlen_k; }
params.window_size_left = window_size_left;
params.window_size_right = window_size_right;
#ifdef FLASHATTENTION_DISABLE_LOCAL
TORCH_CHECK(params.is_causal || (window_size_left < 0 && window_size_right < 0),
"This flash attention build does not support local attention.");
#endif
params.is_seqlens_k_cumulative = true;
params.is_bf16 = is_bf16;
params.single_level_p_quant = single_level_p_quant;
#ifdef FLASHATTENTION_DISABLE_UNEVEN_K
TORCH_CHECK(d == d_rounded, "This flash attention build does not support headdim not being a multiple of 32.");
#endif
}
template<bool IsBF16>
void run_mha_fwd_dispatch_dtype(Flash_fwd_params &params, cudaStream_t stream) {
using OType = std::conditional_t<IsBF16, cutlass::bfloat16_t, cutlass::half_t>;
if (params.d == 64) {
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 64, OType>(params, stream);
} else if (params.d == 128) {
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 128, OType>(params, stream);
}
}
void run_mha_fwd(Flash_fwd_params &params, cudaStream_t stream, bool force_split_kernel = false) {
BOOL_SWITCH(params.is_bf16, IsBF16, ([&] {
run_mha_fwd_dispatch_dtype<IsBF16>(params, stream);
}));
}
std::vector<at::Tensor>
mha_fwd(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head_size // 2)
const at::Tensor &k, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
const at::Tensor &v, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
const at::Tensor &sfq,
const at::Tensor &sfk,
const at::Tensor &sfv,
const at::Tensor &delta_s,
int unpadded_k,
c10::optional<at::Tensor> &out_, // batch_size x seqlen_q x num_heads x head_size
const float softmax_scale,
bool is_causal,
bool per_block_mean,
bool is_bf16,
bool single_level_p_quant=false // If true, use only per-row scale s_P2 (no per-block s_P1)
) {
auto dprops = at::cuda::getCurrentDeviceProperties();
bool is_sm120 = dprops->major == 12 && dprops->minor == 0;
TORCH_CHECK(is_sm120, "only supports Blackwell GPUs or newer.");
auto q_dtype = q.dtype();
auto sfq_dtype = sfq.dtype();
TORCH_CHECK(q_dtype == torch::kUInt8, "q dtype must be uint8");
TORCH_CHECK(k.dtype() == q_dtype, "query and key must have the same dtype");
TORCH_CHECK(v.dtype() == q_dtype, "query and value must have the same dtype");
CHECK_DEVICE(q); CHECK_DEVICE(k); CHECK_DEVICE(v);
TORCH_CHECK(sfq_dtype == torch::kFloat8_e4m3fn, "q dtype must be uint8");
TORCH_CHECK(sfk.dtype() == sfq_dtype, "query and key must have the same dtype");
TORCH_CHECK(sfv.dtype() == sfq_dtype, "query and value must have the same dtype");
CHECK_DEVICE(sfq); CHECK_DEVICE(sfk); CHECK_DEVICE(sfv);
TORCH_CHECK(q.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(k.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(v.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(delta_s.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(q.is_contiguous(), "Input tensor must be contiguous");
TORCH_CHECK(k.is_contiguous(), "Input tensor must be contiguous");
TORCH_CHECK(v.is_contiguous(), "Input tensor must be contiguous");
const auto sizes = q.sizes();
auto opts = q.options();
const int batch_size = sizes[0];
int seqlen_q = sizes[2];
int num_heads = sizes[1];
const int head_size_og = sizes[3];
const int unpacked_head_size = head_size_og * 2;
const int seqlen_k = k.size(2);
const int num_heads_k = k.size(1);
TORCH_CHECK(batch_size > 0, "batch size must be postive");
TORCH_CHECK(unpacked_head_size <= 256, "FlashAttention forward only supports head dimension at most 256");
TORCH_CHECK(num_heads % num_heads_k == 0, "Number of heads in key/value must divide number of heads in query");
TORCH_CHECK(num_heads == num_heads_k, "We do not support MQA/GQA yet");
TORCH_CHECK(unpacked_head_size == 64 || unpacked_head_size == 128 || unpacked_head_size == 256, "Only support head size 64, 128, and 256 for now");
CHECK_SHAPE(q, batch_size, num_heads, seqlen_q, head_size_og);
CHECK_SHAPE(k, batch_size, num_heads_k, seqlen_k, head_size_og);
CHECK_SHAPE(v, batch_size, num_heads_k, unpacked_head_size, seqlen_k/2);
// CHECK_SHAPE(delta_s, batch_size, num_heads, seqlen_q / 128, seqlen_k);
// CHECK_SHAPE(sfq, batch_size, seqlen_q, num_heads, unpacked_head_size);
// CHECK_SHAPE(sfk, batch_size, seqlen_k, num_heads_k, unpacked_head_size);
// CHECK_SHAPE(sfv, batch_size, unpacked_head_size, num_heads_k, seqlen_k);
TORCH_CHECK(unpacked_head_size % 8 == 0, "head_size must be a multiple of 8");
auto dtype = is_bf16 ? at::ScalarType::BFloat16 : at::ScalarType::Half;
at::Tensor out = torch::empty({batch_size, num_heads, seqlen_q, unpacked_head_size}, opts.dtype(dtype));
auto round_multiple = [](int x, int m) { return (x + m - 1) / m * m; };
// const int head_size = round_multiple(head_size_og, 8);
// const int head_size_rounded = round_multiple(head_size, 32);
const int seqlen_q_rounded = round_multiple(seqlen_q, flash::BLOCK_M);
const int seqlen_k_rounded = round_multiple(seqlen_k, flash::BLOCK_N);
// Otherwise the kernel will be launched from cuda:0 device
// Cast to char to avoid compiler warning about narrowing
at::cuda::CUDAGuard device_guard{(char)q.get_device()};
auto softmax_lse = torch::empty({batch_size, num_heads, seqlen_q}, opts.dtype(at::kFloat));
at::Tensor p;
Flash_fwd_params params;
set_params_fprop(params,
batch_size,
seqlen_q, seqlen_k, unpadded_k,
seqlen_q_rounded, seqlen_k_rounded,
num_heads, num_heads_k,
unpacked_head_size, unpacked_head_size,
q, k, v, delta_s, out,
sfq, sfk, sfv,
/*cu_seqlens_q_d=*/nullptr,
/*cu_seqlens_k_d=*/nullptr,
/*seqused_k=*/nullptr,
nullptr,
softmax_lse.data_ptr(),
/*p_dropout=*/0.f,
softmax_scale,
/*window_size_left=*/-1,
/*window_size_right=*/is_causal ? 0 : -1,
per_block_mean,
is_bf16,
single_level_p_quant
);
// TODO: 132 sm count?
auto tile_count_semaphore = is_causal ? torch::full({1}, 132, opts.dtype(torch::kInt32)) : torch::empty({1}, opts.dtype(torch::kInt32));
params.tile_count_semaphore = tile_count_semaphore.data_ptr<int>();
if (seqlen_k > 0) {
auto stream = at::cuda::getCurrentCUDAStream().stream();
run_mha_fwd(params, stream);
} else {
// If seqlen_k == 0, then we have an empty tensor. We need to set the output to 0.
out.zero_();
softmax_lse.fill_(std::numeric_limits<float>::infinity());
}
// at::Tensor out_padded = out;
// if (head_size_og % 8 != 0) {
// out = out.index({"...", torch::indexing::Slice(torch::indexing::None, head_size_og)});
// if (out_.has_value()) { out_.value().copy_(out); }
// }
// return {out, q_padded, k_padded, v_padded, out_padded, softmax_lse, p};
// cudaDeviceSynchronize();
// auto err = cudaGetLastError();
// printf("%s\n", cudaGetErrorString(err));
return {out, softmax_lse};
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.doc() = "FlashAttention";
m.def("fwd", &mha_fwd, "Forward pass");
}
@@ -1,28 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
// Centralized block size configuration for sageattn_blackwell kernels
// Block sizes for M and N dimensions
namespace flash {
// Block size for M dimension (query sequence length)
static constexpr int BLOCK_M = 128;
// Block size for N dimension (key/value sequence length)
static constexpr int BLOCK_N = 128;
}
@@ -1,60 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
namespace flash {
////////////////////////////////////////////////////////////////////////////////////////////////////
template<bool Varlen=true>
struct BlockInfo {
template<typename Params>
__device__ BlockInfo(const Params &params, const int bidb)
: sum_s_q(!Varlen || params.cu_seqlens_q == nullptr ? -1 : params.cu_seqlens_q[bidb])
, sum_s_k(!Varlen || params.cu_seqlens_k == nullptr || !params.is_seqlens_k_cumulative ? -1 : params.cu_seqlens_k[bidb])
, actual_seqlen_q(!Varlen || params.cu_seqlens_q == nullptr ? params.seqlen_q : params.cu_seqlens_q[bidb + 1] - sum_s_q)
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
, seqlen_k_cache(!Varlen || params.cu_seqlens_k == nullptr ? params.seqlen_k : (params.is_seqlens_k_cumulative ? params.cu_seqlens_k[bidb + 1] - sum_s_k : params.cu_seqlens_k[bidb]))
, actual_seqlen_k(params.seqused_k ? params.seqused_k[bidb] : seqlen_k_cache + (params.knew_ptr == nullptr ? 0 : params.seqlen_knew))
{
}
template <typename index_t>
__forceinline__ __device__ index_t q_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
return sum_s_q == -1 ? bidb * batch_stride : uint32_t(sum_s_q) * row_stride;
}
template <typename index_t>
__forceinline__ __device__ index_t k_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
return sum_s_k == -1 ? bidb * batch_stride : uint32_t(sum_s_k) * row_stride;
}
const int sum_s_q;
const int sum_s_k;
const int actual_seqlen_q;
// We have to have seqlen_k_cache declared before actual_seqlen_k, otherwise actual_seqlen_k is set to 0.
const int seqlen_k_cache;
const int actual_seqlen_k;
};
////////////////////////////////////////////////////////////////////////////////////////////////////
} // namespace flash
@@ -1,149 +0,0 @@
/***************************************************************************************************
* Copyright (c) 2023 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: BSD-3-Clause
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are met:
*
* 1. Redistributions of source code must retain the above copyright notice, this
* list of conditions and the following disclaimer.
*
* 2. Redistributions in binary form must reproduce the above copyright notice,
* this list of conditions and the following disclaimer in the documentation
* and/or other materials provided with the distribution.
*
* 3. Neither the name of the copyright holder nor the names of its
* contributors may be used to endorse or promote products derived from
* this software without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*
**************************************************************************************************/
/*! \file
\brief Blocked Scale configs specific for SM100 BlockScaled MMA
*/
#pragma once
#include "cutlass/layout/matrix.h"
#include "cute/int_tuple.hpp"
#include "cute/atom/mma_traits_sm100.hpp"
namespace flash {
/////////////////////////////////////////////////////////////////////////////////////////////////
using namespace cute;
template<int SFVecSize, UMMA::Major major = UMMA::Major::K>
struct BlockScaledBasicChunk {
using Blk_MN = _64;
using Blk_SF = _4;
using SfAtom = Layout< Shape< Shape<_16,_4>, Shape<Int<SFVecSize>, _4>>,
Stride<Stride<_16,_4>, Stride< _0, _1>>>;
};
template<int SFVecSize_>
struct BlockScaledConfig {
// We are creating the SFA and SFB tensors' layouts in the collective since they always have the same layout.
// k-major order
static constexpr int SFVecSize = SFVecSize_;
static constexpr int MMA_NSF = 4; // SFVecSize, MMA_NSF
using BlkScaledChunk = BlockScaledBasicChunk<SFVecSize>;
using Blk_MN = _64;
using Blk_SF = _4;
using mnBasicBlockShape = Shape<_16,_4>;
using mnBasicBlockStride = Stride<_16,_4>;
using kBasicBlockShape = Shape<Int<SFVecSize>, Int<MMA_NSF>>; // SFVecSize, MMA_NSF
using kBasicBlockStride = Stride<_0, _1>;
using SfAtom = Layout< Shape< mnBasicBlockShape, kBasicBlockShape>,
Stride<mnBasicBlockStride, kBasicBlockStride>>;
using LayoutSF = decltype(blocked_product(SfAtom{},
make_layout(
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))));
// A single indivisible block will hold 4 scale factors of 64 rows/columns (A/B matrix).
// 4 is chosen to make consecutive 32bits of data to have scale factors for only a single row (col). 32bits corresponds to the TMEM word size
using Blk_Elems = decltype(Blk_MN{} * Blk_SF{});
using sSF_strideMN = decltype(prepend(Blk_Elems{}, mnBasicBlockStride{}));
// The following function is provided for user fill dynamic problem size to the layout_SFA.
template < class ProblemShape>
CUTE_HOST_DEVICE
static constexpr auto
tile_atom_to_shape_SFQKV(ProblemShape problem_shape) {
auto [Seqlen, Dim, HeadNum, Batch] = problem_shape;
return tile_to_shape(SfAtom{}, make_shape(Seqlen, Dim, HeadNum, Batch), Step<_2,_1,_3,_4>{});
}
// The following function is provided for user fill dynamic problem size to the layout_SFB.
template <class ProblemShape>
CUTE_HOST_DEVICE
static constexpr auto
tile_atom_to_shape_SFVt(ProblemShape problem_shape) {
auto [Dim, Seqlen, HeadNum, Batch] = problem_shape;
return tile_to_shape(SfAtom{}, make_shape(Dim, Seqlen, HeadNum, Batch), Step<_2,_1,_3,_4>{});
}
template<class TiledMma, class TileShape_MNK>
CUTE_HOST_DEVICE
static constexpr auto
deduce_smem_layoutSFQ(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
using sSFQ_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
using sSFQ_shapeM = decltype(prepend(size<0>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
using sSFQ_strideM = sSF_strideMN;
using sSFQ_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<0>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
using sSFQ_shape = decltype(make_shape(sSFQ_shapeM{}, sSFQ_shapeK{}));
using sSFQ_stride = decltype(make_stride(sSFQ_strideM{}, sSFQ_strideK{}));
using SmemLayoutAtomSFQ = decltype(make_layout(sSFQ_shape{}, sSFQ_stride{}));
return SmemLayoutAtomSFQ{};
}
template<class TiledMma, class TileShape_MNK>
CUTE_HOST_DEVICE
static constexpr auto
deduce_smem_layoutSFKV(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
using sSFK_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
using sSFK_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
using sSFK_strideN = sSF_strideMN;
using sSFK_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
using sSFK_shape = decltype(make_shape(sSFK_shapeN{}, sSFK_shapeK{}));
using sSFK_stride = decltype(make_stride(sSFK_strideN{}, sSFK_strideK{}));
using SmemLayoutAtomSFK = decltype(make_layout(sSFK_shape{}, sSFK_stride{}));
return SmemLayoutAtomSFK{};
}
template<class TiledMma, class TileShape_MNK>
CUTE_HOST_DEVICE
static constexpr auto
deduce_smem_layoutSFVt(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
using sSFVt_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
using sSFVt_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
using sSFVt_strideN = sSF_strideMN;
using sSFVt_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
using sSFVt_shape = decltype(make_shape(sSFVt_shapeN{}, sSFVt_shapeK{}));
using sSFVt_stride = decltype(make_stride(sSFVt_strideN{}, sSFVt_strideK{}));
using SmemLayoutAtomSFVt = decltype(make_layout(sSFVt_shape{}, sSFVt_stride{}));
return SmemLayoutAtomSFVt{};
}
};
} // namespace flash
@@ -1,327 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "cute/arch/mma_sm120.hpp"
#include "cute/atom/mma_traits_sm120.hpp"
#include "cute/atom/mma_atom.hpp"
#include "cutlass/cutlass.h"
#include "cutlass/float8.h"
#include "cutlass/float_subbyte.h"
namespace cute::SM120::BLOCKSCALED {
using cutlass::float_e2m1_t;
using cutlass::float_ue4m3_t;
// MMA.SF 16x32x64 TN E2M1 x E2M1 with SF E4M3
struct SM120_16x32x64_TN_VS_NVFP4 {
using DRegisters = float[16];
using ARegisters = uint32_t[4];
using BRegisters = uint32_t[8];
using CRegisters = float[16];
static constexpr int SFBits = 32;
using RegTypeSF = cute::uint_bit_t<SFBits>;
using SFARegisters = RegTypeSF[1];
using SFBRegisters = RegTypeSF[1];
CUTE_HOST_DEVICE static void
fma(float & d0 , float & d1 , float & d2 , float & d3 ,
float & d4 , float & d5 , float & d6 , float & d7 ,
float & d8 , float & d9 , float & d10, float & d11,
float & d12, float & d13, float & d14, float & d15,
uint32_t const& a0 , uint32_t const& a1 , uint32_t const& a2 , uint32_t const& a3 ,
uint32_t const& b0 , uint32_t const& b1 , uint32_t const& b2 , uint32_t const& b3 ,
uint32_t const& b4 , uint32_t const& b5 , uint32_t const& b6 , uint32_t const& b7 ,
float const & c0 , float const & c1 , float const & c2 , float const & c3 ,
float const & c4 , float const & c5 , float const & c6 , float const & c7 ,
float const & c8 , float const & c9 , float const & c10 , float const & c11,
float const & c12, float const & c13, float const & c14, float const & c15,
RegTypeSF const& sfa0,
RegTypeSF const& sfb0)
{
static constexpr uint16_t tidA = 0;
static constexpr uint16_t bidA = 0;
static constexpr uint16_t bidB = 0;
static constexpr uint16_t tidB0 = 0;
static constexpr uint16_t tidB1 = 1;
static constexpr uint16_t tidB2 = 2;
static constexpr uint16_t tidB3 = 3;
#if defined(CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED)
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d0), "=f"(d1), "=f"(d8), "=f"(d9)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b0), "r"(b1),
"f"(c0), "f"(c1), "f"(c8), "f"(c9),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB0));
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d2), "=f"(d3), "=f"(d10), "=f"(d11)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b2), "r"(b3),
"f"(c2), "f"(c3), "f"(c10), "f"(c11),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB1));
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d4), "=f"(d5), "=f"(d12), "=f"(d13)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b4), "r"(b5),
"f"(c4), "f"(c5), "f"(c12), "f"(c13),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB2));
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d6), "=f"(d7), "=f"(d14), "=f"(d15)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b6), "r"(b7),
"f"(c6), "f"(c7), "f"(c14), "f"(c15),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB3));
#else
CUTE_INVALID_CONTROL_PATH("Attempting to use SM120::BLOCKSCALED::SM120_16x8x64_TN_VS without CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED");
#endif
}
};
} // namespace cute::SM120::BLOCKSCALED
namespace cute {
// MMA NVFP4 16x32x64 TN
template <>
struct MMA_Traits<SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4>
{
// The MMA accepts 4-bit inputs regardless of the types for A and B
using ValTypeA = uint4_t;
using ValTypeB = uint4_t;
using ValTypeD = float;
using ValTypeC = float;
using ValTypeSF = cutlass::float_ue4m3_t;
constexpr static int SFVecSize = 16;
using Shape_MNK = Shape<_16,_32,_64>;
using ThrID = Layout<_32>;
// (T32,V32) -> (M16,K64)
using ALayout = Layout<Shape <Shape < _4,_8>,Shape < _8,_2, _2>>,
Stride<Stride<_128,_1>,Stride<_16,_8,_512>>>;
// (T32,V64) -> (N32,K64)
using BLayout = Layout<Shape <Shape < _4,_8>,Shape <_8, _2, _4>>,
Stride<Stride<_256,_1>,Stride<_32,_1024, _8>>>;
// (T32,V64) -> (M16,K64)
using SFALayout = Layout<Shape <Shape <_2,_2,_8>,_64>,
Stride<Stride<_8,_0,_1>,_16>>;
// (T32,V64) -> (N32,K64)
using SFBLayout = Layout<Shape <Shape <_4,_8>,_64>,
Stride<Stride<_8,_1>, _32>>;
// (T32,V16) -> (M16,N32)
using CLayout = Layout<Shape <Shape < _4,_8>,Shape < Shape<_2, _4>,_2>>,
Stride<Stride<_32,_1>,Stride<Stride<_16, _128>,_8>>>;
};
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<0>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<1>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
return thr_tensor;
}
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<1>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<2>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
return thr_tensor;
}
template <class SFATensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
return thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
}
template <class SFATensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
return make_fragment_like<ValTypeSF>(partition_SFA(sfatensor, thread_mma));
}
template <class SFBTensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
return thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
}
template <class SFBTensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
return make_fragment_like<ValTypeSF>(partition_SFB(sfbtensor, thread_mma));
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFA_TV(TiledMma& mma)
{
// (M,K) -> (M,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto atile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<1>{} , Int<0>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFB_TV(TiledMma& mma)
{
// (N,K) -> (N,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto btile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<0>{} , Int<1>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
}
} // namespace cute
@@ -1,222 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cutlass/cutlass.h>
#include "cute/tensor.hpp"
#include "cutlass/gemm/collective/collective_builder.hpp"
#include "named_barrier.h"
#include "utils.h"
namespace flash {
using namespace cute;
template <typename Ktraits>
struct CollectiveEpilogueFwd{
using Element = typename Ktraits::ElementOut;
static constexpr int kBlockM = Ktraits::kBlockM;
static constexpr int kBlockN = Ktraits::kBlockN;
static constexpr int kHeadDim = Ktraits::kHeadDim;
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
static constexpr int kNWarps = Ktraits::kNWarps;
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
static constexpr int NumMmaThreads = kNThreads - cutlass::NumThreadsPerWarpGroup;
using GmemTiledCopyOTMA = cute::SM90_TMA_STORE;
// These are for storing the output tensor without TMA (e.g., for setting output to zero)
static constexpr int kGmemElemsPerLoad = sizeof(cute::uint128_t) / sizeof(Element);
static_assert(kHeadDim % kGmemElemsPerLoad == 0, "kHeadDim must be a multiple of kGmemElemsPerLoad");
static constexpr int kGmemThreadsPerRow = kHeadDim / kGmemElemsPerLoad;
static_assert(NumMmaThreads % kGmemThreadsPerRow == 0, "NumMmaThreads must be a multiple of kGmemThreadsPerRow");
using GmemLayoutAtom = Layout<Shape <Int<NumMmaThreads / kGmemThreadsPerRow>, Int<kGmemThreadsPerRow>>,
Stride<Int<kGmemThreadsPerRow>, _1>>;
using GmemTiledCopyO = decltype(
make_tiled_copy(Copy_Atom<DefaultCopy, Element>{},
GmemLayoutAtom{},
Layout<Shape<_1, Int<kGmemElemsPerLoad>>>{})); // Val layout, 8 or 16 vals per store
using SmemLayoutO = typename Ktraits::SmemLayoutO;
using SmemCopyAtomO = Copy_Atom<SM90_U32x2_STSM_N, Element>;
using SharedStorage = cute::array_aligned<Element, cute::cosize_v<SmemLayoutO>>;
using ShapeO = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen_q, d, head, batch)
using StrideO = cute::Stride<int64_t, _1, int64_t, int64_t>;
using StrideLSE = cute::Stride<_1, int64_t, int64_t>; // (seqlen_q, head, batch)
using TMA_O = decltype(make_tma_copy(
GmemTiledCopyOTMA{},
make_tensor(make_gmem_ptr(static_cast<Element*>(nullptr)), repeat_like(StrideO{}, int32_t(0)), StrideO{}),
SmemLayoutO{},
select<0, 2>(TileShape_MNK{}),
_1{})); // no mcast for O
// Host side kernel arguments
struct Arguments {
Element* ptr_O;
ShapeO const shape_O;
StrideO const stride_O;
float* ptr_LSE;
StrideLSE const stride_LSE;
};
// Device side kernel params
struct Params {
Element* ptr_O;
ShapeO const shape_O;
StrideO const stride_O;
float* ptr_LSE;
StrideLSE const stride_LSE;
TMA_O tma_store_O;
};
static Params
to_underlying_arguments(Arguments const& args) {
Tensor mO = make_tensor(make_gmem_ptr(args.ptr_O), args.shape_O, args.stride_O);
TMA_O tma_store_O = make_tma_copy(
GmemTiledCopyOTMA{},
mO,
SmemLayoutO{},
select<0, 2>(TileShape_MNK{}),
_1{}); // no mcast for O
return {args.ptr_O, args.shape_O, args.stride_O, args.ptr_LSE, args.stride_LSE, tma_store_O};
}
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
CUTLASS_DEVICE
static void prefetch_tma_descriptors(Params const& epilogue_params) {
cute::prefetch_tma_descriptor(epilogue_params.tma_store_O.get_tma_descriptor());
}
template <typename SharedStorage, typename FrgTensorO, typename TiledMma>
CUTLASS_DEVICE void
mma_store(
SharedStorage& shared_storage,
TiledMma tiled_mma,
FrgTensorO const& tOrO,
int thread_idx
){
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
auto smem_tiled_copy_O = make_tiled_copy_C(SmemCopyAtomO{}, tiled_mma);
auto smem_thr_copy_O = smem_tiled_copy_O.get_thread_slice(thread_idx);
constexpr int numel = decltype(size(tOrO))::value;
cutlass::NumericArrayConverter<Element, float, numel> convert_op;
// HACK: this requires tensor to be "contiguous"
auto frag = convert_op(*reinterpret_cast<const cutlass::Array<float, numel> *>(tOrO.data()));
auto tOrO_out = make_tensor(make_rmem_ptr<Element>(&frag), tOrO.layout());
Tensor taccOrO = smem_thr_copy_O.retile_S(tOrO_out); // ((Atom,AtomNum), MMA_M, MMA_N)
Tensor taccOsO = smem_thr_copy_O.partition_D(sO); // ((Atom,AtomNum),PIPE_M,PIPE_N)
cute::copy(smem_tiled_copy_O, taccOrO, taccOsO);
cutlass::arch::fence_view_async_shared(); // ensure smem writes are visible to TMA
}
template<typename SharedStorage, typename Params, typename WorkTileInfo, typename SchedulerParams>
CUTLASS_DEVICE void
tma_store(
SharedStorage& shared_storage,
Params const& epilogue_params,
WorkTileInfo work_tile_info,
SchedulerParams const& scheduler_params,
int thread_idx
) {
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
Tensor mO = epilogue_params.tma_store_O.get_tma_tensor(epilogue_params.shape_O);
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
auto block_tma_O = epilogue_params.tma_store_O.get_slice(_0{});
Tensor tOgO = block_tma_O.partition_D(gO); // (TMA, TMA_M, TMA_K)
Tensor tOsO = block_tma_O.partition_S(sO); // (TMA, TMA_M, TMA_K)
// auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
// Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
// Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
// Tensor caccO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{}));
// auto thread_mma = tiled_mma.get_thread_slice(thread_idx);
// Tensor taccOcO = thread_mma.partition_C(caccO); // (MMA,MMA_M,MMA_K)
// static_assert(decltype(size<0, 0>(taccOcO))::value == 2);
// static_assert(decltype(size<0, 1>(taccOcO))::value == 2);
// // // // taccOcO has shape ((2, 2, V), MMA_M, MMA_K), we only take only the row indices.
// Tensor taccOcO_row = taccOcO(make_coord(_0{}, _), _, _0{});
// CUTE_STATIC_ASSERT_V(size(lse) == size(taccOcO_row)); // MMA_M
// if (get<1>(taccOcO_row(_0{})) == 0) {
// #pragma unroll
// for (int mi = 0; mi < size(lse); ++mi) {
// const int row = get<0>(taccOcO_row(mi));
// if (row < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(row) = lse(mi); }
// }
// }
// if (cutlass::canonical_warp_idx_sync() == kNWarps - 1) {
// cutlass::arch::NamedBarrier::sync(NumMmaThreads + cutlass::NumThreadsPerWarp,
// static_cast<uint32_t>(FP4NamedBarriers::EpilogueBarrier));
// int const lane_predicate = cute::elect_one_sync();
// if (lane_predicate) {
// cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
// tma_store_arrive();
// }
// }
cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
tma_store_arrive();
}
CUTLASS_DEVICE void
store_tail() {
tma_store_wait<0>();
}
// Write 0 to output and -inf to LSE
CUTLASS_DEVICE void
store_zero(
Params const& epilogue_params,
int thread_idx,
cute::tuple<int32_t, int32_t, int32_t> const& block_coord
) {
auto [m_block, bidh, bidb] = block_coord;
Tensor mO = make_tensor(make_gmem_ptr(epilogue_params.ptr_O), epilogue_params.shape_O, epilogue_params.stride_O);
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
GmemTiledCopyO gmem_tiled_copy_O;
auto gmem_thr_copy_O = gmem_tiled_copy_O.get_thread_slice(thread_idx);
Tensor tOgO = gmem_thr_copy_O.partition_D(gO);
Tensor tOrO = make_fragment_like(tOgO);
clear(tOrO);
// Construct identity layout for sO
Tensor cO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{})); // (BLK_M,BLK_K) -> (blk_m,blk_k)
// Repeat the partitioning with identity layouts
Tensor tOcO = gmem_thr_copy_O.partition_D(cO);
Tensor tOpO = make_tensor<bool>(make_shape(size<2>(tOgO)));
#pragma unroll
for (int k = 0; k < size(tOpO); ++k) { tOpO(k) = get<1>(tOcO(_0{}, _0{}, k)) < get<1>(epilogue_params.shape_O); }
// Clear_OOB_K must be false since we don't want to write zeros to gmem
flash::copy</*Is_even_MN=*/false, /*Is_even_K=*/false, /*Clear_OOB_MN=*/false, /*Clear_OOB_K=*/false>(
gmem_tiled_copy_O, tOrO, tOgO, tOcO, tOpO, get<0>(epilogue_params.shape_O) - m_block * kBlockM
);
static_assert(kBlockM <= NumMmaThreads);
if (thread_idx < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(thread_idx) = INFINITY; }
}
};
} // namespace flash
@@ -1,202 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cute/algorithm/copy.hpp"
#include "cute/atom/mma_atom.hpp"
#include "cutlass/gemm/collective/collective_builder.hpp"
#include "cute/tensor.hpp"
#include "cutlass/cutlass.h"
#include "cutlass/layout/layout.h"
#include "cutlass/numeric_types.h"
#include "cutlass/pipeline/pipeline.hpp"
#include "blockscaled_layout.h"
#include "cute_extension.h"
#include "named_barrier.h"
using namespace cute;
template <
int kStages,
int EpiStages,
typename Element,
typename ElementSF,
typename OutputType,
typename SmemLayoutQ,
typename SmemLayoutK,
typename SmemLayoutV,
typename SmemLayoutDS,
typename SmemLayoutO,
typename SmemLayoutSFQ,
typename SmemLayoutSFK,
typename SmemLayoutSFV
>
struct SharedStorageQKVOwithSF : cute::aligned_struct<128, _0>{
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutQ>> smem_q;
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutK>> smem_k;
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFQ>> smem_SFQ;
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFK>> smem_SFK;
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFV>> smem_SFV;
alignas(1024) cute::ArrayEngine<float, cute::cosize_v<SmemLayoutDS>> smem_ds;
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutV>> smem_v;
alignas(1024) cute::ArrayEngine<OutputType, cute::cosize_v<SmemLayoutO>> smem_o;
struct {
alignas(16) typename cutlass::PipelineTmaAsync<1>::SharedStorage pipeline_q;
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_k;
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_v;
alignas(16) typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>::SharedStorage barrier_o;
int tile_count_semaphore;
};
};
template <
int kHeadDim_,
int kBlockM_,
int kBlockN_,
int kStages_,
int kClusterM_,
bool BlockMean_,
typename ElementPairType_ = cutlass::nv_float4_t<cutlass::float_e2m1_t>,
typename ElementOut_ = cutlass::bfloat16_t
>
struct Flash_fwd_kernel_traits {
static constexpr int kBlockM = kBlockM_;
static constexpr int kBlockN = kBlockN_;
static constexpr int kHeadDim = kHeadDim_;
static constexpr bool BlockMean = BlockMean_;
static constexpr bool SmoothQ = true;
static_assert(kHeadDim % 32 == 0);
static_assert(kBlockM == 64 || kBlockM == 128);
static constexpr int kNWarps = kBlockM == 128 ? 12 : 8;
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
static constexpr int kClusterM = kClusterM_;
static constexpr int kStages = kStages_;
static constexpr int EpiStages = 1;
static constexpr int NumSFQK = kHeadDim / 16;
static constexpr int NumSFPV = kBlockN / 16;
using ElementSF = cutlass::float_ue4m3_t;
using Element = cutlass::float_e2m1_t;
using ElementAccum = float;
using ElementOut = ElementOut_;
using index_t = int64_t;
static constexpr auto SFVectorSize = 16;
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
using ClusterShape_MNK = Shape<_1, _1, _1>;
using PermTileM = decltype(cute::min(size<0>(TileShape_MNK{}), _128{}));
using PermTileN = _32;
using PermTileK = Int<kHeadDim>;
using ElementQMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
using ElementKMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
using AtomLayoutMNK = std::conditional_t<kBlockM == 128,
Layout<Shape<_8, _1, _1>>,
Layout<Shape<_4, _1, _1>>
>;
using TiledMmaQK = decltype(cute::make_tiled_mma(
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
AtomLayoutMNK{},
Tile<PermTileM, PermTileN, PermTileK>{}
));
using TiledMmaPV = decltype(cute::make_tiled_mma(
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
AtomLayoutMNK{},
Tile<PermTileM, _32, PermTileK>{}
));
static constexpr int MMA_NSF = size<2>(typename TiledMmaQK::AtomShape_MNK{}) / SFVectorSize;
using GmemTiledCopy = SM90_TMA_LOAD;
using GmemTiledCopySF = SM90_TMA_LOAD;
using SmemLayoutAtomQ = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
using SmemLayoutAtomK = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
using SmemLayoutAtomV = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
using SmemLayoutAtomVt = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<1>(TileShape_MNK{}))>());
using SmemLayoutQ = decltype(tile_to_shape(SmemLayoutAtomQ{}, select<0, 2>(TileShape_MNK{})));
using SmemLayoutK =
decltype(tile_to_shape(SmemLayoutAtomK{},
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
using SmemLayoutV =
decltype(tile_to_shape(SmemLayoutAtomV{},
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
using SmemLayoutVt =
decltype(tile_to_shape(SmemLayoutAtomVt{},
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
using SmemLayoutAtomDS = Layout<Shape<Int<kBlockM>, Int<kBlockN>>, Stride<_0, _1>>;
using SmemLayoutDS =
decltype(tile_to_shape(SmemLayoutAtomDS{},
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
using SmemCopyAtomQ = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
using SmemCopyAtomKV = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
using SmemCopyAtomSF = Copy_Atom<UniversalCopy<ElementSF>, ElementSF>;
using SmemCopyAtomDS = Copy_Atom<UniversalCopy<float>, float>;
using BlkScaledConfig = flash::BlockScaledConfig<SFVectorSize>;
using LayoutSF = typename BlkScaledConfig::LayoutSF;
using SfAtom = typename BlkScaledConfig::SfAtom;
using SmemLayoutAtomSFQ = decltype(BlkScaledConfig::deduce_smem_layoutSFQ(TiledMmaQK{}, TileShape_MNK{}));
using SmemLayoutAtomSFK = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaQK{}, TileShape_MNK{}));
using SmemLayoutAtomSFV = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaPV{}, TileShape_MNK{}));
using SmemLayoutAtomSFVt = decltype(BlkScaledConfig::deduce_smem_layoutSFVt(TiledMmaPV{}, Shape<Int<kBlockM>, Int<kHeadDim>, Int<kBlockN>>{}));
using LayoutSFP = decltype(
make_layout(
make_shape(make_shape(_16{}, _4{}), _1{}, Int<kBlockN / 64>{}),
make_stride(make_stride(_0{}, _1{}), _0{}, _4{})
)
);
using LayoutP = decltype(
make_layout(
make_shape(make_shape(_8{}, _2{}, _2{}), _1{}, Int<kBlockN / 64>{}),
make_stride(make_stride(_1{}, _8{}, _16{}), _0{}, _32{})
)
);
using SmemLayoutSFQ = decltype(make_layout(
shape(SmemLayoutAtomSFQ{}),
stride(SmemLayoutAtomSFQ{})
));
using SmemLayoutSFK = decltype(make_layout(
append(shape(SmemLayoutAtomSFK{}), Int<kStages>{}),
append(stride(SmemLayoutAtomSFK{}), size(filter_zeros(SmemLayoutAtomSFK{})))
));
using SmemLayoutSFV = decltype(make_layout(
append(shape(SmemLayoutAtomSFV{}), Int<kStages>{}),
append(stride(SmemLayoutAtomSFV{}), size(filter_zeros(SmemLayoutAtomSFV{})))
));
using SmemLayoutSFVt = decltype(make_layout(
append(shape(SmemLayoutAtomSFVt{}), Int<kStages>{}),
append(stride(SmemLayoutAtomSFVt{}), size(filter_zeros(SmemLayoutAtomSFVt{})))
));
using SmemLayoutAtomO = decltype(cutlass::gemm::collective::detail::ss_smem_selector<GMMA::Major::K, ElementOut,
decltype(cute::get<0>(TileShape_MNK{})), decltype(cute::get<2>(TileShape_MNK{}))>());
using SmemLayoutO = decltype(tile_to_shape(SmemLayoutAtomO{}, select<0, 2>(TileShape_MNK{}), Step<_1, _2>{}));
using SharedStorage = SharedStorageQKVOwithSF<kStages, EpiStages, Element, ElementSF, ElementOut,
SmemLayoutQ, SmemLayoutK, SmemLayoutV, SmemLayoutDS,
SmemLayoutO, SmemLayoutSFQ, SmemLayoutSFK, SmemLayoutSFVt>;
using MainloopPipeline = typename cutlass::PipelineTmaAsync<kStages>;
using PipelineState = typename cutlass::PipelineState<kStages>;
using MainloopPipelineQ = cutlass::PipelineTmaAsync<1>;
using PipelineParamsQ = typename MainloopPipelineQ::Params;
using PipelineStateQ = typename cutlass::PipelineState<1>;
using EpilogueBarrier = typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>;
};
@@ -1,204 +0,0 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cute/tensor.hpp"
#include <cutlass/cutlass.h>
#include <cutlass/arch/reg_reconfig.h>
#include <cutlass/array.h>
#include <cutlass/numeric_types.h>
#include <cutlass/numeric_conversion.h>
#include "cutlass/pipeline/pipeline.hpp"
#include "params.h"
#include "utils.h"
#include "tile_scheduler.h"
#include "mainloop_tma_ws.h"
#include "epilogue_tma_ws.h"
#include "named_barrier.h"
#include "softmax_fused.h"
namespace flash {
using namespace cute;
template <typename Ktraits, bool Is_causal, typename TileScheduler>
__global__ void __launch_bounds__(Ktraits::kNWarps * cutlass::NumThreadsPerWarp, 1)
compute_attn_ws(CUTE_GRID_CONSTANT Flash_fwd_params const params,
CUTE_GRID_CONSTANT typename CollectiveMainloopFwd<Ktraits, Is_causal>::Params const mainloop_params,
CUTE_GRID_CONSTANT typename CollectiveEpilogueFwd<Ktraits>::Params const epilogue_params,
CUTE_GRID_CONSTANT typename TileScheduler::Params const scheduler_params
) {
using Element = typename Ktraits::Element;
using ElementAccum = typename Ktraits::ElementAccum;
using SoftType = ElementAccum;
using TileShape_MNK = typename Ktraits::TileShape_MNK;
using ClusterShape = typename Ktraits::ClusterShape_MNK;
static constexpr int NumMmaThreads = size(typename Ktraits::TiledMmaQK{});
static constexpr int NumCopyThreads = cutlass::NumThreadsPerWarpGroup;
static constexpr int kBlockM = Ktraits::kBlockM;
using CollectiveMainloop = CollectiveMainloopFwd<Ktraits, Is_causal>;
using CollectiveEpilogue = CollectiveEpilogueFwd<Ktraits>;
using MainloopPipeline = typename Ktraits::MainloopPipeline;
using PipelineParams = typename MainloopPipeline::Params;
using PipelineState = typename MainloopPipeline::PipelineState;
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
using PipelineStateQ = typename Ktraits::PipelineStateQ;
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
enum class WarpGroupRole {
Producer = 0,
Consumer0 = 1,
Consumer1 = 2
};
enum class ProducerWarpRole {
Mainloop = 0,
Epilogue = 1,
Warp2 = 2,
Warp3 = 3
};
extern __shared__ char shared_memory[];
auto &shared_storage = *reinterpret_cast<typename Ktraits::SharedStorage*>(shared_memory);
int const lane_predicate = cute::elect_one_sync();
int const warp_idx = cutlass::canonical_warp_idx_sync();
int warp_group_idx = cutlass::canonical_warp_group_idx();
int const warp_group_thread_idx = threadIdx.x % cutlass::NumThreadsPerWarpGroup;
int warp_idx_in_warp_group = warp_idx % cutlass::NumWarpsPerWarpGroup;
auto warp_group_role = WarpGroupRole(warp_group_idx);
auto producer_warp_role = ProducerWarpRole(warp_idx_in_warp_group);
// Issue Tma Descriptor Prefetch from a single thread
if (warp_idx == 0 && lane_predicate) {
CollectiveMainloop::prefetch_tma_descriptors(mainloop_params);
CollectiveEpilogue::prefetch_tma_descriptors(epilogue_params);
}
// Obtain warp index
PipelineParams pipeline_params_v;
pipeline_params_v.transaction_bytes = CollectiveMainloop::TmaTransactionBytesV;
pipeline_params_v.role = warp_group_role == WarpGroupRole::Producer
? MainloopPipeline::ThreadCategory::Producer
: MainloopPipeline::ThreadCategory::Consumer;
pipeline_params_v.is_leader = warp_group_thread_idx == 0;
pipeline_params_v.num_consumers = NumMmaThreads;
PipelineParams pipeline_params_k;
pipeline_params_k.transaction_bytes = CollectiveMainloop::TmaTransactionBytesK;
pipeline_params_k.role = warp_group_role == WarpGroupRole::Producer
? MainloopPipeline::ThreadCategory::Producer
: MainloopPipeline::ThreadCategory::Consumer;
pipeline_params_k.is_leader = warp_group_thread_idx == 0;
pipeline_params_k.num_consumers = NumMmaThreads;
PipelineParamsQ pipeline_params_q;
pipeline_params_q.transaction_bytes = CollectiveMainloop::TmaTransactionBytesQ;
pipeline_params_q.role = warp_group_role == WarpGroupRole::Producer
? MainloopPipelineQ::ThreadCategory::Producer
: MainloopPipelineQ::ThreadCategory::Consumer;
pipeline_params_q.is_leader = warp_group_thread_idx == 0;
pipeline_params_q.num_consumers = NumMmaThreads;
// We're counting on pipeline_k to call cutlass::arch::fence_barrier_init();
MainloopPipelineQ pipeline_q(shared_storage.pipeline_q, pipeline_params_q, ClusterShape{});
MainloopPipeline pipeline_k(shared_storage.pipeline_k, pipeline_params_k, ClusterShape{});
MainloopPipeline pipeline_v(shared_storage.pipeline_v, pipeline_params_v, ClusterShape{});
uint32_t epilogue_barrier_group_size_list[2] = {cutlass::NumThreadsPerWarp, NumMmaThreads};
typename EpilogueBarrier::Params params_epilogue_barrier;
params_epilogue_barrier.group_id = (warp_group_role == WarpGroupRole::Producer);
params_epilogue_barrier.group_size_list = epilogue_barrier_group_size_list;
EpilogueBarrier barrier_o(shared_storage.barrier_o, params_epilogue_barrier);
CollectiveMainloop collective_mainloop;
CollectiveEpilogue collective_epilogue;
__syncthreads();
if (warp_group_role == WarpGroupRole::Producer) {
cutlass::arch::warpgroup_reg_dealloc<24>();
TileScheduler scheduler;
if (producer_warp_role == ProducerWarpRole::Mainloop) { // Load Q, K, V
PipelineStateQ smem_pipe_write_q = cutlass::make_producer_start_state<MainloopPipelineQ>();
PipelineState smem_pipe_write_k = cutlass::make_producer_start_state<MainloopPipeline>();
PipelineState smem_pipe_write_v = cutlass::make_producer_start_state<MainloopPipeline>();
int work_idx = 0;
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
int tile_count_semaphore = 0;
collective_mainloop.load(mainloop_params, scheduler_params,
pipeline_q, pipeline_k, pipeline_v,
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v,
shared_storage, work_tile_info, work_idx, tile_count_semaphore);
}
collective_mainloop.load_tail(pipeline_q, pipeline_k, pipeline_v,
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v);
} else if (producer_warp_role == ProducerWarpRole::Epilogue) {
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
barrier_o.wait();
collective_epilogue.tma_store(shared_storage, epilogue_params, work_tile_info, scheduler_params, threadIdx.x);
collective_epilogue.store_tail();
barrier_o.arrive();
}
}
} else if (warp_group_role == WarpGroupRole::Consumer0 || warp_group_role == WarpGroupRole::Consumer1) {
cutlass::arch::warpgroup_reg_alloc<232>();
typename Ktraits::TiledMmaPV tiled_mma_pv;
TileScheduler scheduler{};
PipelineState smem_pipe_read_k, smem_pipe_read_v;
PipelineStateQ smem_pipe_read_q;
int work_idx = 0;
CUTLASS_PRAGMA_NO_UNROLL
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
// Attention output (GEMM-II) accumulator.
Tensor tOrO = partition_fragment_C(tiled_mma_pv, select<0, 2>(TileShape_MNK{}));
// flash::Softmax<2 * (2 * kBlockM / NumMmaThreads)> softmax;
// Pass single_level_p_quant flag to control P quantization mode
flash::SoftmaxFused<2 * (2 * kBlockM / NumMmaThreads)> softmax_fused(params.single_level_p_quant);
auto block_coord = work_tile_info.get_block_coord(scheduler_params);
auto [m_block, bidh, bidb] = block_coord;
int n_block_max = collective_mainloop.get_n_block_max(mainloop_params, m_block);
if (Is_causal && n_block_max <= 0) { // We exit early and write 0 to gO and -inf to gLSE.
collective_epilogue.store_zero(epilogue_params, threadIdx.x - NumCopyThreads, block_coord);
continue;
}
collective_mainloop.mma(mainloop_params, pipeline_q, pipeline_k, pipeline_v, smem_pipe_read_q, smem_pipe_read_k, smem_pipe_read_v,
tOrO, softmax_fused, n_block_max, threadIdx.x - NumCopyThreads, work_idx, m_block, shared_storage);
barrier_o.wait();
collective_epilogue.mma_store(shared_storage, tiled_mma_pv, tOrO, threadIdx.x - NumCopyThreads);
barrier_o.arrive();
++work_idx;
}
}
}
} // namespace flash
@@ -1,114 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <ATen/cuda/CUDAContext.h>
#include "cute/tensor.hpp"
#include "cutlass/cluster_launch.hpp"
#include "static_switch.h"
#include "params.h"
#include "tile_scheduler.h"
#include "kernel_ws.h"
#include "kernel_traits.h"
#include "block_config.h"
template<typename Kernel_traits, bool Is_causal>
void run_flash_fwd(Flash_fwd_params &params, cudaStream_t stream) {
using Element = typename Kernel_traits::Element;
using ElementSF = typename Kernel_traits::ElementSF;
using ElementOut = typename Kernel_traits::ElementOut;
using TileShape_MNK = typename Kernel_traits::TileShape_MNK;
using ClusterShape = typename Kernel_traits::ClusterShape_MNK;
using CollectiveMainloop = flash::CollectiveMainloopFwd<Kernel_traits, Is_causal>;
using CollectiveEpilogue = flash::CollectiveEpilogueFwd<Kernel_traits>;
// using Scheduler = flash::SingleTileScheduler;
using Scheduler = flash::StaticPersistentTileScheduler;
typename CollectiveMainloop::Params mainloop_params =
CollectiveMainloop::to_underlying_arguments({
static_cast<Element const*>(params.q_ptr),
{params.seqlen_q, params.d, params.h, params.b}, // shape_Q
{params.q_row_stride, _1{}, params.q_head_stride, params.q_batch_stride}, // stride_Q
static_cast<Element const*>(params.k_ptr),
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_K
{params.k_row_stride, _1{}, params.k_head_stride, params.k_batch_stride}, // stride_K
{params.unpadded_seqlen_k, params.d, params.h_k, params.b}, // shape_K
static_cast<Element const*>(params.v_ptr),
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_Vt
{params.v_row_stride, _1{}, params.v_head_stride, params.v_batch_stride}, // stride_Vt
static_cast<ElementSF const*>(params.sfq_ptr),
{params.seqlen_q, params.d, params.h, params.b}, // shape_SFQ
static_cast<ElementSF const*>(params.sfk_ptr),
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_SFK
static_cast<ElementSF const*>(params.sfv_ptr),
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_SFVt
static_cast<float const*>(params.delta_s_ptr),
{params.seqlen_s, params.seqlen_k, params.h_k, params.b},
{params.ds_row_stride, _1{}, params.ds_head_stride, params.ds_batch_stride},
params.scale_softmax_log2
});
typename CollectiveEpilogue::Params epilogue_params =
CollectiveEpilogue::to_underlying_arguments({
static_cast<ElementOut*>(params.o_ptr),
{params.seqlen_q, params.d, params.h, params.b}, // shape_O
{params.o_row_stride, _1{}, params.o_head_stride, params.o_batch_stride}, // stride_O
static_cast<float*>(params.softmax_lse_ptr),
{_1{}, params.seqlen_q, params.h * params.seqlen_q}, // stride_LSE
});
int num_blocks_m = cutlass::ceil_div(params.seqlen_q, Kernel_traits::kBlockM);
num_blocks_m = cutlass::ceil_div(num_blocks_m, size<0>(ClusterShape{})) * size<0>(ClusterShape{});
typename Scheduler::Arguments scheduler_args = {num_blocks_m, params.h, params.b};
typename Scheduler::Params scheduler_params = Scheduler::to_underlying_arguments(scheduler_args);
// Get the ptr to kernel function.
void *kernel;
kernel = (void *)flash::compute_attn_ws<Kernel_traits, Is_causal, Scheduler>;
int smem_size = sizeof(typename Kernel_traits::SharedStorage);
if (smem_size >= 48 * 1024) {
C10_CUDA_CHECK(cudaFuncSetAttribute(kernel, cudaFuncAttributeMaxDynamicSharedMemorySize, smem_size));
}
static constexpr int ctaSize = Kernel_traits::kNWarps * 32;
params.m_block_divmod = cutlass::FastDivmod(num_blocks_m);
params.total_blocks = num_blocks_m * params.h * params.b;
dim3 grid_dims = Scheduler::get_grid_dim(scheduler_args, 170);
dim3 block_dims(ctaSize);
dim3 cluster_dims(size<0>(ClusterShape{}), size<1>(ClusterShape{}), size<2>(ClusterShape{}));
cutlass::ClusterLaunchParams launch_params{grid_dims, block_dims, cluster_dims, smem_size, stream};
cutlass::launch_kernel_on_cluster(launch_params, kernel, params, mainloop_params, epilogue_params, scheduler_params);
C10_CUDA_KERNEL_LAUNCH_CHECK();
}
template<typename T, int Headdim, typename O = cutlass::bfloat16_t>
void run_mha_fwd_(Flash_fwd_params &params, cudaStream_t stream) {
BOOL_SWITCH(params.is_causal, Is_causal, [&] {
BOOL_SWITCH(params.per_block_mean, per_block, [&] {
if constexpr (Headdim == 64 || Headdim == 128) {
run_flash_fwd<
Flash_fwd_kernel_traits<Headdim, flash::BLOCK_M, flash::BLOCK_N, 3, 1, per_block, T, O>,
Is_causal
>(params, stream);
} else {
static_assert(Headdim == 64 || Headdim == 128, "Unsupported Headdim");
}
});
});
}
@@ -1,908 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cutlass/cutlass.h>
#include <cutlass/array.h>
#include <cutlass/numeric_types.h>
#include <cutlass/numeric_conversion.h>
#include "cutlass/pipeline/pipeline.hpp"
#include "cute/tensor.hpp"
#include "cutlass/gemm/collective/collective_builder.hpp"
#include "utils.h"
#include "named_barrier.h"
namespace flash {
using namespace cute;
template <typename Ktraits, bool Is_causal>
struct CollectiveMainloopFwd {
using Element = typename Ktraits::Element;
using ElementSF = typename Ktraits::ElementSF;
// using TMAElement = Element;
// using TMAElementSF = typename Ktraits::ElementSF;
using TileShape_MNK = typename Ktraits::TileShape_MNK;
using ClusterShape = typename Ktraits::ClusterShape_MNK;
static constexpr int kStages = Ktraits::kStages;
static constexpr int kHeadDim = Ktraits::kHeadDim;
static constexpr int BlockMean = Ktraits::BlockMean;
using GmemTiledCopy = typename Ktraits::GmemTiledCopy;
using SmemLayoutQ = typename Ktraits::SmemLayoutQ;
using SmemLayoutK = typename Ktraits::SmemLayoutK;
using SmemLayoutV = typename Ktraits::SmemLayoutV;
using SmemLayoutVt = typename Ktraits::SmemLayoutVt;
using SmemLayoutDS = typename Ktraits::SmemLayoutDS;
using SmemLayoutAtomDS = typename Ktraits::SmemLayoutAtomDS;
using LayoutDS = decltype(
blocked_product(
SmemLayoutAtomDS{},
make_layout(
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))
)
);
using ShapeQKV = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d, head, batch)
using StrideQKV = cute::Stride<int64_t, _1, int64_t, int64_t>;
using ShapeSF = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d // 16, head, batch)
using LayoutSF = typename Ktraits::LayoutSF;
using LayoutP = typename Ktraits::LayoutP;
using LayoutSFP = typename Ktraits::LayoutSFP;
using SfAtom = typename Ktraits::SfAtom;
using TMA_Q = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
SmemLayoutQ{},
select<0, 2>(TileShape_MNK{}),
_1{}));
using TMA_KV = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
take<0, 2>(SmemLayoutK{}),
select<1, 2>(TileShape_MNK{}),
_1{}));
using TMA_Vt = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
take<0, 2>(SmemLayoutVt{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}));
using TMA_DS = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<float const*>(nullptr)), LayoutDS{}),
take<0, 2>(SmemLayoutDS{}),
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}));
using BlkScaledConfig = typename Ktraits::BlkScaledConfig;
using GmemTiledCopySF = typename Ktraits::GmemTiledCopySF;
using SmemLayoutSFQ = typename Ktraits::SmemLayoutSFQ;
using SmemLayoutSFK = typename Ktraits::SmemLayoutSFK;
using SmemLayoutSFV = typename Ktraits::SmemLayoutSFV;
using SmemLayoutSFVt = typename Ktraits::SmemLayoutSFVt;
using TMA_SFQ = decltype(make_tma_copy<uint16_t>(
GmemTiledCopySF{},
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
SmemLayoutSFQ{},
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{})); // No programmatic multicast
using TMA_SFKV = decltype(make_tma_copy<uint16_t>(
GmemTiledCopySF{},
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
SmemLayoutSFK{}(_,_,cute::Int<0>{}),
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{}));
using TMA_SFVt = decltype(make_tma_copy<uint16_t>(
GmemTiledCopySF{},
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
SmemLayoutSFVt{}(_,_,cute::Int<0>{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}));
using SmemCopyAtomQ = typename Ktraits::SmemCopyAtomQ;
using SmemCopyAtomKV = typename Ktraits::SmemCopyAtomKV;
using SmemCopyAtomSF = typename Ktraits::SmemCopyAtomSF;
using TiledMmaQK = typename Ktraits::TiledMmaQK;
using TiledMmaPV = typename Ktraits::TiledMmaPV;
static constexpr int NumMmaThreads = size(TiledMmaQK{});
using MainloopPipeline = typename Ktraits::MainloopPipeline;
using PipelineParams = typename MainloopPipeline::Params;
using PipelineState = typename MainloopPipeline::PipelineState;
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
using PipelineStateQ = typename Ktraits::PipelineStateQ;
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
// Set the bytes transferred in this TMA transaction (may involve multiple issues)
static constexpr uint32_t TmaTransactionBytesQ = static_cast<uint32_t>(
cutlass::bits_to_bytes(cosize((SmemLayoutSFQ{})) * cute::sizeof_bits_v<ElementSF>) +
cutlass::bits_to_bytes(size((SmemLayoutQ{})) * sizeof_bits<Element>::value));
static constexpr uint32_t TmaTransactionBytesK = static_cast<uint32_t>(
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFK{})) * cute::sizeof_bits_v<ElementSF>) +
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutDS{})) * cute::sizeof_bits_v<float>) +
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutK{})) * sizeof_bits<Element>::value));
static constexpr uint32_t TmaTransactionBytesV = static_cast<uint32_t>(
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFVt{})) * cute::sizeof_bits_v<ElementSF>) +
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutVt{})) * sizeof_bits<Element>::value));
// Host side kernel arguments
struct Arguments {
Element const* ptr_Q;
ShapeQKV const shape_Q;
StrideQKV const stride_Q;
Element const* ptr_K;
ShapeQKV const shape_K;
StrideQKV const stride_K;
ShapeQKV const unpadded_shape_K;
Element const* ptr_Vt;
ShapeQKV const shape_Vt;
StrideQKV const stride_Vt;
ElementSF const* ptr_SFQ{nullptr};
ShapeSF const shape_SFQ{};
ElementSF const* ptr_SFK{nullptr};
ShapeSF const shape_SFK{};
ElementSF const* ptr_SFVt{nullptr};
ShapeSF const shape_SFVt{};
float const* ptr_ds;
ShapeQKV const shape_ds;
StrideQKV const stride_ds;
float const softmax_scale_log2;
};
// Device side kernel params
struct Params {
ShapeQKV const shape_Q;
LayoutSF const layout_SFQ;
ShapeQKV const shape_K;
ShapeQKV const unpadded_shape_K;
LayoutSF const layout_SFK;
ShapeQKV const shape_Vt;
LayoutSF const layout_SFVt;
LayoutDS const layout_DS;
TMA_Q tma_load_Q;
TMA_SFQ tma_load_SFQ;
TMA_KV tma_load_K;
TMA_SFKV tma_load_SFK;
TMA_Vt tma_load_Vt;
TMA_SFVt tma_load_SFVt;
TMA_DS tma_load_DS;
float const softmax_scale_log2;
};
static Params
to_underlying_arguments(Arguments const& args) {
Tensor mQ = make_tensor(make_gmem_ptr(args.ptr_Q), args.shape_Q, args.stride_Q);
TMA_Q tma_load_Q = make_tma_copy(
GmemTiledCopy{},
mQ,
SmemLayoutQ{},
select<0, 2>(TileShape_MNK{}),
_1{}); // no mcast for Q
Tensor mK = make_tensor(make_gmem_ptr(args.ptr_K), args.shape_K, args.stride_K);
TMA_KV tma_load_K = make_tma_copy(
GmemTiledCopy{},
mK,
SmemLayoutK{}(_, _, _0{}),
select<1, 2>(TileShape_MNK{}),
_1{}); // mcast along M mode for this N load, if any
Tensor mVt = make_tensor(make_gmem_ptr(args.ptr_Vt), args.shape_Vt, args.stride_Vt);
TMA_Vt tma_load_Vt = make_tma_copy(
GmemTiledCopy{},
mVt,
SmemLayoutVt{}(_, _, _0{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}); // mcast along M mode for this N load, if any
auto [Seqlen_Q, Seqlen_K, HeadNum, Batch] = args.shape_ds;
LayoutDS layout_ds = tile_to_shape(SmemLayoutAtomDS{}, make_shape(Seqlen_Q, Seqlen_K, HeadNum, Batch), Step<_2,_1,_3,_4>{});
Tensor mDS = make_tensor(make_gmem_ptr(args.ptr_ds), layout_ds);
TMA_DS tma_load_ds = make_tma_copy (
GmemTiledCopy{},
mDS,
SmemLayoutDS{}(_, _, _0{}),
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{});
LayoutSF layout_sfq = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFQ);
Tensor mSFQ = make_tensor(make_gmem_ptr(args.ptr_SFQ), layout_sfq);
TMA_SFQ tma_load_sfq = make_tma_copy<uint16_t>(
GmemTiledCopySF{},
mSFQ,
SmemLayoutSFQ{},
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{});
LayoutSF layout_sfk = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFK);
Tensor mSFK = make_tensor(make_gmem_ptr(args.ptr_SFK), layout_sfk);
TMA_SFKV tma_load_sfk = make_tma_copy<uint16_t>(
GmemTiledCopySF{},
mSFK,
SmemLayoutSFK{}(_, _, _0{}),
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{});
LayoutSF layout_sfvt = BlkScaledConfig::tile_atom_to_shape_SFVt(args.shape_SFVt);
Tensor mSFVt = make_tensor(make_gmem_ptr(args.ptr_SFVt), layout_sfvt);
TMA_SFVt tma_load_sfvt = make_tma_copy<uint16_t>(
GmemTiledCopySF{},
mSFVt,
SmemLayoutSFVt{}(_, _, _0{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{});
return {args.shape_Q, layout_sfq,
args.shape_K, args.unpadded_shape_K, layout_sfk,
args.shape_Vt, layout_sfvt,
layout_ds,
tma_load_Q, tma_load_sfq,
tma_load_K, tma_load_sfk,
tma_load_Vt, tma_load_sfvt,
tma_load_ds,
args.softmax_scale_log2};
}
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
CUTLASS_DEVICE
static void prefetch_tma_descriptors(Params const& mainloop_params) {
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Q.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_K.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Vt.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFQ.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFK.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFVt.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_DS.get_tma_descriptor());
}
CUTLASS_DEVICE
int get_n_block_max(Params const& mainloop_params, int m_block) {
static constexpr int kBlockM = get<0>(TileShape_MNK{});
static constexpr int kBlockN = get<1>(TileShape_MNK{});
int const seqlen_q = get<0>(mainloop_params.shape_Q);
int const seqlen_k = get<0>(mainloop_params.shape_K);
int n_block_max = cute::ceil_div(seqlen_k, kBlockN);
if constexpr (Is_causal) {
n_block_max = std::min(n_block_max,
cute::ceil_div((m_block + 1) * kBlockM + seqlen_k - seqlen_q, kBlockN));
}
return n_block_max;
}
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<0>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<1>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
return thr_tensor;
}
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<1>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<2>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
return thr_tensor;
}
template <class SFATensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma)
{
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
auto partition_SFA = thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
return make_fragment_like<ValTypeSF>(partition_SFA);
}
template <class SFBTensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma)
{
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
auto partition_SFB = thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
return make_fragment_like<ValTypeSF>(partition_SFB);
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFA_TV(TiledMma& mma)
{
// (M,K) -> (M,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto atile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<1>{} , Int<0>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFB_TV(TiledMma& mma)
{
// (N,K) -> (N,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto btile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<0>{} , Int<1>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
}
template <typename SchedulerParams, typename SharedStorage, typename WorkTileInfo>
CUTLASS_DEVICE void
load(Params const& mainloop_params,
SchedulerParams const& scheduler_params,
MainloopPipelineQ pipeline_q,
MainloopPipeline pipeline_k,
MainloopPipeline pipeline_v,
PipelineStateQ& smem_pipe_write_q,
PipelineState& smem_pipe_write_k,
PipelineState& smem_pipe_write_v,
SharedStorage &shared_storage,
WorkTileInfo work_tile_info,
int& work_idx,
int& tile_count_semaphore
) {
static constexpr int kBlockM = get<0>(TileShape_MNK{});
static constexpr int kBlockN = get<1>(TileShape_MNK{});
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
int n_block_max = get_n_block_max(mainloop_params, m_block);
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
Tensor mQ = mainloop_params.tma_load_Q.get_tma_tensor(mainloop_params.shape_Q);
Tensor mK = mainloop_params.tma_load_K.get_tma_tensor(mainloop_params.shape_K);
Tensor mVt = mainloop_params.tma_load_Vt.get_tma_tensor(mainloop_params.shape_Vt);
Tensor mDS = mainloop_params.tma_load_DS.get_tma_tensor(shape(mainloop_params.layout_DS));
Tensor mSFQ = mainloop_params.tma_load_SFQ.get_tma_tensor(shape(mainloop_params.layout_SFQ));
Tensor mSFK = mainloop_params.tma_load_SFK.get_tma_tensor(shape(mainloop_params.layout_SFK));
Tensor mSFVt = mainloop_params.tma_load_SFVt.get_tma_tensor(shape(mainloop_params.layout_SFVt));
uint32_t block_rank_in_cluster = cute::block_rank_in_cluster();
constexpr uint32_t cluster_shape_x = get<0>(ClusterShape());
uint2 cluster_local_block_id = {block_rank_in_cluster % cluster_shape_x, block_rank_in_cluster / cluster_shape_x};
Tensor gQ = local_tile(mQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
Tensor gK = local_tile(mK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{})); // (N, K, _)
Tensor gVt = local_tile(mVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _)); // (N, K, _)
Tensor gDS = [&] {
if constexpr (BlockMean) {
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(m_block, _));
} else {
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(_0{}, _));
}
}();
Tensor gSFQ = local_tile(mSFQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{}));
Tensor gSFK = local_tile(mSFK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{}));
Tensor gSFVt = local_tile(mSFVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _));
auto block_tma_q = mainloop_params.tma_load_Q.get_slice(_0{});
Tensor tQgQ = block_tma_q.partition_S(gQ);
Tensor tQsQ = block_tma_q.partition_D(sQ);
auto block_tma_sfq = mainloop_params.tma_load_SFQ.get_slice(_0{});
Tensor tQgSFQ = block_tma_sfq.partition_S(gSFQ);
Tensor tQsSFQ = block_tma_sfq.partition_D(sSFQ);
auto block_tma_k = mainloop_params.tma_load_K.get_slice(cluster_local_block_id.x);
Tensor tKgK = group_modes<0, 3>(block_tma_k.partition_S(gK));
Tensor tKsK = group_modes<0, 3>(block_tma_k.partition_D(sK));
auto block_tma_sfk = mainloop_params.tma_load_SFK.get_slice(cluster_local_block_id.x);
Tensor tKgSFK = group_modes<0, 3>(block_tma_sfk.partition_S(gSFK));
Tensor tKsSFK = group_modes<0, 3>(block_tma_sfk.partition_D(sSFK));
auto block_tma_vt = mainloop_params.tma_load_Vt.get_slice(cluster_local_block_id.x);
Tensor tVgVt = group_modes<0, 3>(block_tma_vt.partition_S(gVt));
Tensor tVsVt = group_modes<0, 3>(block_tma_vt.partition_D(sVt));
auto block_tma_sfvt = mainloop_params.tma_load_SFVt.get_slice(cluster_local_block_id.x);
Tensor tVgSFVt = group_modes<0, 3>(block_tma_sfvt.partition_S(gSFVt));
Tensor tVsSFVt = group_modes<0, 3>(block_tma_sfvt.partition_D(sSFVt));
auto block_tma_ds = mainloop_params.tma_load_DS.get_slice(cluster_local_block_id.x);
Tensor tDSgDS = group_modes<0, 3>(block_tma_ds.partition_S(gDS));
Tensor tDSsDS = group_modes<0, 3>(block_tma_ds.partition_D(sDS));
uint16_t mcast_mask_kv = 0;
int n_block = n_block_max - 1;
int lane_predicate = cute::elect_one_sync();
if (lane_predicate) {
pipeline_q.producer_acquire(smem_pipe_write_q);
copy(mainloop_params.tma_load_Q.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgQ, tQsQ);
copy(mainloop_params.tma_load_SFQ.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgSFQ, tQsSFQ);
++smem_pipe_write_q;
pipeline_k.producer_acquire(smem_pipe_write_k);
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
++smem_pipe_write_k;
pipeline_v.producer_acquire(smem_pipe_write_v);
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
++smem_pipe_write_v;
}
n_block--;
if (lane_predicate) {
// CUTLASS_PRAGMA_NO_UNROLL
#pragma unroll 2
for (; n_block >= 0; --n_block) {
pipeline_k.producer_acquire(smem_pipe_write_k);
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
++smem_pipe_write_k;
pipeline_v.producer_acquire(smem_pipe_write_v);
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
++smem_pipe_write_v;
}
}
++work_idx;
}
/// Perform a Producer Epilogue to prevent early exit of blocks in a Cluster
CUTLASS_DEVICE void
load_tail(MainloopPipelineQ pipeline_q,
MainloopPipeline pipeline_k,
MainloopPipeline pipeline_v,
PipelineStateQ& smem_pipe_write_q,
PipelineState& smem_pipe_write_k,
PipelineState& smem_pipe_write_v) {
int lane_predicate = cute::elect_one_sync();
// Issue the epilogue waits
if (lane_predicate) {
pipeline_q.producer_tail(smem_pipe_write_q);
pipeline_k.producer_tail(smem_pipe_write_k);
pipeline_v.producer_tail(smem_pipe_write_v);
}
}
template <typename SharedStorage, typename FrgTensorO, typename SoftmaxFused>
CUTLASS_DEVICE void
mma(Params const& mainloop_params,
MainloopPipelineQ pipeline_q,
MainloopPipeline pipeline_k,
MainloopPipeline pipeline_v,
PipelineStateQ& smem_pipe_read_q,
PipelineState& smem_pipe_read_k,
PipelineState& smem_pipe_read_v,
FrgTensorO& tOrO_store,
SoftmaxFused& softmax_fused,
int n_block_count,
int thread_idx,
int work_idx,
int m_block,
SharedStorage& shared_storage
) {
static_assert(is_rmem<FrgTensorO>::value, "O tensor must be rmem resident.");
static constexpr int kBlockM = get<0>(TileShape_MNK{});
static constexpr int kBlockN = get<1>(TileShape_MNK{});
static constexpr int kBlockK = get<2>(TileShape_MNK{});
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
Tensor cQ = make_identity_tensor(make_shape(size<0>(sQ), size<1>(sQ)));
Tensor cKV = make_identity_tensor(make_shape(size<0>(sK), size<1>(sK)));
TiledMmaQK tiled_mma_qk;
TiledMmaPV tiled_mma_pv;
auto thread_mma_qk = tiled_mma_qk.get_thread_slice(thread_idx);
auto thread_mma_pv = tiled_mma_pv.get_thread_slice(thread_idx);
Tensor tSrQ = thread_mma_qk.partition_fragment_A(sQ);
Tensor tSrK = thread_mma_qk.partition_fragment_B(sK(_,_,Int<0>{}));
Tensor tOrVt = thread_mma_pv.partition_fragment_B(sVt(_,_,Int<0>{}));
Tensor tOrP = make_tensor_like<Element>(LayoutP{});
Tensor tSrSFQ = partition_fragment_SFA(sSFQ, thread_mma_qk);
Tensor tSrSFK = partition_fragment_SFB(sSFK(_,_,Int<0>{}), thread_mma_qk);
Tensor tOrSFVt = partition_fragment_SFB(sSFVt(_,_,Int<0>{}), thread_mma_pv);
Tensor tOrSFP = make_tensor<ElementSF>(LayoutSFP{});
Tensor tOrSFP_flt = filter_zeros(tOrSFP);
Tensor tSrDS = make_tensor<float>(make_shape(_8{}, _4{}), make_stride(_1{}, _8{}));
// copy qk and sf from smem to rmem
auto smem_tiled_copy_Q = make_tiled_copy_A(SmemCopyAtomQ{}, tiled_mma_qk);
auto smem_thr_copy_Q = smem_tiled_copy_Q.get_thread_slice(thread_idx);
Tensor tSsQ = smem_thr_copy_Q.partition_S(as_position_independent_swizzle_tensor(sQ));
Tensor tSrQ_copy_view = smem_thr_copy_Q.retile_D(tSrQ);
auto smem_tiled_copy_K = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_qk);
auto smem_thr_copy_K = smem_tiled_copy_K.get_thread_slice(thread_idx);
Tensor tSsK = smem_thr_copy_K.partition_S(as_position_independent_swizzle_tensor(sK));
Tensor tSrK_copy_view = smem_thr_copy_K.retile_D(tSrK);
auto smem_tiled_copy_V = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_pv);
auto smem_thr_copy_V = smem_tiled_copy_V.get_thread_slice(thread_idx);
Tensor tOsVt = smem_thr_copy_V.partition_S(as_position_independent_swizzle_tensor(sVt));
Tensor tOrVt_copy_view = smem_thr_copy_V.retile_D(tOrVt);
auto tile_shape_mnk = tile_shape(tiled_mma_qk);
auto smem_tiled_copy_SFQ = make_tiled_copy_impl(SmemCopyAtomSF{},
get_layoutSFA_TV(tiled_mma_qk),
make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk))
);
auto smem_thr_copy_SFQ = smem_tiled_copy_SFQ.get_thread_slice(thread_idx);
Tensor tSsSFQ = smem_thr_copy_SFQ.partition_S(as_position_independent_swizzle_tensor(sSFQ));
Tensor tSrSFQ_copy_view = smem_thr_copy_SFQ.retile_D(tSrSFQ);
auto smem_tiled_copy_SFK = make_tiled_copy_impl(SmemCopyAtomSF{},
get_layoutSFB_TV(tiled_mma_qk),
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
);
auto smem_thr_copy_SFK = smem_tiled_copy_SFK.get_thread_slice(thread_idx);
Tensor tSsSFK = smem_thr_copy_SFK.partition_S(as_position_independent_swizzle_tensor(sSFK));
Tensor tSrSFK_copy_view = smem_thr_copy_SFK.retile_D(tSrSFK);
auto smem_tiled_copy_SFV = make_tiled_copy_impl(SmemCopyAtomSF{},
get_layoutSFB_TV(tiled_mma_pv),
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
);
auto smem_thr_copy_SFV = smem_tiled_copy_SFV.get_thread_slice(thread_idx);
Tensor tOsSFVt = smem_thr_copy_SFV.partition_S(as_position_independent_swizzle_tensor(sSFVt));
Tensor tOrSFVt_copy_view = smem_thr_copy_SFV.retile_D(tOrSFVt);
auto consumer_wait = [](auto& pipeline, auto& smem_pipe_read) {
auto barrier_token = pipeline.consumer_try_wait(smem_pipe_read);
pipeline.consumer_wait(smem_pipe_read, barrier_token);
};
int const seqlen_q = get<0>(mainloop_params.shape_Q);
int const seqlen_k = get<0>(mainloop_params.shape_K);
int const unpadded_seqlen_k = get<0>(mainloop_params.unpadded_shape_K);
int n_block = n_block_count - 1;
auto copy_k_block = [&](auto block_id) {
auto tSsK_stage = tSsK(_, _, _, smem_pipe_read_k.index());
auto tSsSFK_stage = tSsSFK(_, _, _, smem_pipe_read_k.index());
copy(smem_tiled_copy_K, tSsK_stage(_, _, block_id), tSrK_copy_view(_, _, block_id));
copy(smem_tiled_copy_SFK, tSsSFK_stage(_, _, block_id), tSrSFK_copy_view(_, _, block_id));
};
auto copy_v_block = [&](auto block_id) {
auto tOsVt_stage = tOsVt(_, _, _, smem_pipe_read_v.index());
auto tOsSFVt_stage = tOsSFVt(_, _, _, smem_pipe_read_v.index());
copy(smem_tiled_copy_V, tOsVt_stage(_, _, block_id), tOrVt_copy_view(_, _, block_id));
copy(smem_tiled_copy_SFV, tOsSFVt_stage(_, _, block_id), tOrSFVt_copy_view(_, _, block_id));
};
// auto gemm_qk = [&](auto block_id) {
// cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, block_id), tSrSFQ(_, _, block_id)), make_zip_tensor(tSrK(_, _, block_id), tSrSFK(_, _, block_id)), tSrS);
// };
// auto gemm_pv = [&](auto block_id) {
// cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, block_id), tOrSFP(_, _, block_id)), make_zip_tensor(tOrVt(_, _, block_id), tOrSFVt(_, _, block_id)), tOrO);
// };
auto add_delta_s = [&](auto& acc) {
auto tSsDS_stage = recast<float4>(sDS(_, _, smem_pipe_read_k.index()));
auto acc_float4 = recast<float4>(acc);
int quad_id = (threadIdx.x % 4) * 2;
for (int i = 0; i < 4; i++) {
auto num = quad_id + i * 8;
float4 delta_s_0 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num, _0{}));
float4 delta_s_1 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num + 1, _0{}));
acc_float4(make_coord(make_coord(_0{}, _0{}), _0{}), _0{}, i) = delta_s_0;
acc_float4(make_coord(make_coord(_0{}, _0{}), _1{}), _0{}, i) = delta_s_0;
acc_float4(make_coord(make_coord(_0{}, _1{}), _0{}), _0{}, i) = delta_s_1;
acc_float4(make_coord(make_coord(_0{}, _1{}), _1{}), _0{}, i) = delta_s_1;
}
};
consumer_wait(pipeline_q, smem_pipe_read_q);
copy(smem_tiled_copy_Q, tSsQ, tSrQ_copy_view);
copy(smem_tiled_copy_SFQ, tSsSFQ, tSrSFQ_copy_view);
pipeline_q.consumer_release(smem_pipe_read_q);
++smem_pipe_read_q;
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
Tensor AbsMaxP = make_tensor_like<float>(
make_layout(shape(group<1, 4>(flatten(tSrS_converion_view.layout()(make_coord(_0{}, _), _, _)))))
);
consumer_wait(pipeline_k, smem_pipe_read_k);
copy_k_block(_0{});
add_delta_s(tSrS);
CUTLASS_PRAGMA_UNROLL
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
if (k_block < size<2>(tSrQ) - 1) {
copy_k_block(k_block + 1);
} else {
pipeline_k.consumer_release(smem_pipe_read_k);
++smem_pipe_read_k;
}
}
auto col_limit_causal = [&](int row, int n_block) {
return row + 1 + seqlen_k - n_block * kBlockN - seqlen_q + m_block * kBlockM;
};
{
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
Tensor tScS = thread_mma_qk.partition_C(cS);
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < size(tSrS); ++i) {
if constexpr (!Is_causal) { // Just masking based on col
if (int(get<1>(tScS(i))) >= int(unpadded_seqlen_k - n_block * kBlockN)) { tSrS(i) = -INFINITY; }
} else {
if (int(get<1>(tScS(i))) >= std::min(seqlen_k - n_block * kBlockN,
col_limit_causal(int(get<0>(tScS(i))), n_block))) {
tSrS(i) = -INFINITY;
}
}
}
}
auto quantize = [&](auto mma_k, auto acc_conversion_view) {
Tensor AbsMaxP_stagek = AbsMaxP(_, make_coord(_, _, mma_k));
Tensor acc_conversion_stagek = acc_conversion_view(_, _, mma_k);
Tensor SFP = make_tensor_like<cutlass::float_ue4m3_t>(AbsMaxP_stagek.layout());
Tensor SFP_uint32_view = recast<uint32_t>(SFP);
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < size(AbsMaxP_stagek); i += 4) {
uint32_t& tmp = SFP_uint32_view(i / 4);
flash::packed_float_to_ue4m3(
AbsMaxP_stagek(i),
AbsMaxP_stagek(i + 1),
AbsMaxP_stagek(i + 2),
AbsMaxP_stagek(i + 3),
tmp
);
}
int const quad_id = threadIdx.x & 3;
uint32_t MASK = (0xFF00FF) << ((quad_id & 1) * 8);
Tensor tOrSFP_uint32_view = recast<uint32_t>(tOrSFP(_, _, mma_k));
Tensor tOrP_uint32_view = recast<uint32_t>(tOrP(_, _, mma_k));
CUTLASS_PRAGMA_UNROLL
for (int mma_m = 0; mma_m < size<1>(tOrP); ++mma_m) {
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < 4; ++i) {
flash::packed_float_to_e2m1(
acc_conversion_stagek(make_coord(_0{}, i), mma_m),
acc_conversion_stagek(make_coord(_1{}, i), mma_m),
acc_conversion_stagek(make_coord(_2{}, i), mma_m),
acc_conversion_stagek(make_coord(_3{}, i), mma_m),
acc_conversion_stagek(make_coord(_4{}, i), mma_m),
acc_conversion_stagek(make_coord(_5{}, i), mma_m),
acc_conversion_stagek(make_coord(_6{}, i), mma_m),
acc_conversion_stagek(make_coord(_7{}, i), mma_m),
tOrP_uint32_view(i, mma_m)
);
}
uint32_t local_sfp = SFP_uint32_view(_0{}, _0{}, mma_m);
uint32_t peer_sfp = __shfl_xor_sync(int32_t(-1), local_sfp, 2);
if ((quad_id & 1) == 0) {
uint32_t sfp = (local_sfp & MASK) | ((peer_sfp & MASK) << 8);
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
} else {
uint32_t sfp = (peer_sfp & MASK) | ((local_sfp & MASK) >> 8);
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
}
}
};
softmax_fused.template online_softmax_with_quant</*Is_first=*/true>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
consumer_wait(pipeline_v, smem_pipe_read_v);
copy_v_block(_0{});
quantize(_0{}, tSrS_converion_view);
CUTLASS_PRAGMA_UNROLL
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO_store);
if (v_block < size<2>(tOrP) - 1) {
copy_v_block(v_block + 1);
quantize(v_block + 1, tSrS_converion_view);
} else {
pipeline_v.consumer_release(smem_pipe_read_v);
++smem_pipe_read_v;
}
}
n_block--;
constexpr int n_masking_steps = !Is_causal ? 1 : cute::ceil_div(kBlockM, kBlockN) + 1;
// // Only go through these if Is_causal, since n_masking_steps = 1 when !Is_causal
CUTLASS_PRAGMA_UNROLL
for (int masking_step = 0; masking_step < n_masking_steps - 1 && n_block >= 0; ++masking_step, --n_block) {
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
consumer_wait(pipeline_k, smem_pipe_read_k);
copy_k_block(_0{});
add_delta_s(tSrS);
CUTLASS_PRAGMA_UNROLL
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
if (k_block < size<2>(tSrQ) - 1) {
copy_k_block(k_block + 1);
}
}
pipeline_k.consumer_release(smem_pipe_read_k); // release K
++smem_pipe_read_k;
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
Tensor tScS = thread_mma_qk.partition_C(cS);
#pragma unroll
for (int i = 0; i < size(tSrS); ++i) {
if (int(get<1>(tScS(i))) >= col_limit_causal(int(get<0>(tScS(i))), n_block - 1)) {
tSrS(i) = -INFINITY;
}
}
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
Tensor tOrO = make_fragment_like(tOrO_store);
consumer_wait(pipeline_v, smem_pipe_read_v);
copy_v_block(_0{});
quantize(_0{}, tSrS_converion_view);
CUTLASS_PRAGMA_UNROLL
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
if (v_block < size<2>(tOrP) - 1) {
copy_v_block(v_block + 1);
quantize(v_block + 1, tSrS_converion_view);
}
}
pipeline_v.consumer_release(smem_pipe_read_v);
++smem_pipe_read_v;
if (masking_step > 0) { softmax_fused.rescale_o(tOrO_store, tOrO); }
}
#pragma unroll 1
for (; n_block >= 0; --n_block) {
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
consumer_wait(pipeline_k, smem_pipe_read_k);
copy_k_block(_0{});
add_delta_s(tSrS);
CUTLASS_PRAGMA_UNROLL
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
if (k_block < size<2>(tSrQ) - 1) {
copy_k_block(k_block + 1);
} else {
pipeline_k.consumer_release(smem_pipe_read_k);
++smem_pipe_read_k;
}
}
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
Tensor tOrO = make_fragment_like(tOrO_store);
consumer_wait(pipeline_v, smem_pipe_read_v);
copy_v_block(_0{});
quantize(_0{}, tSrS_converion_view);
CUTLASS_PRAGMA_UNROLL
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
if (v_block < size<2>(tOrP) - 1) {
copy_v_block(v_block + 1);
quantize(v_block + 1, tSrS_converion_view);
} else {
pipeline_v.consumer_release(smem_pipe_read_v);
++smem_pipe_read_v;
}
}
softmax_fused.rescale_o(tOrO_store, tOrO);
}
softmax_fused.finalize(tOrO_store);
return;
}
};
} // namespace flash
@@ -1,119 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cutlass/arch/barrier.h"
#include "cutlass/pipeline/sm90_pipeline.hpp"
namespace flash {
enum class FP4NamedBarriers {
QueryEmpty = 1,
WarpSpecializedConsumer = 2,
WarpSpecializedPingPongConsumer1 = 3,
WarpSpecializedPingPongConsumer2 = 4,
ProducerEnd = 5,
ConsumerEnd = 6,
EpilogueBarrier = 7
};
template<int SequenceDepth, int SequenceLength>
struct OrderedSequenceBarrierVarGroupSizeSharedStorage {
using Barrier = cutlass::arch::ClusterBarrier;
Barrier barrier_[SequenceDepth][SequenceLength];
};
template<int SequenceDepth_, int SequenceLength_>
class OrderedSequenceBarrierVarGroupSize {
public:
static constexpr int SequenceDepth = SequenceDepth_;
static constexpr int SequenceLength = SequenceLength_;
using Barrier = cutlass::arch::ClusterBarrier;
using SharedStorage = flash::OrderedSequenceBarrierVarGroupSizeSharedStorage<SequenceDepth, SequenceLength>;
struct Params {
uint32_t group_id;
uint32_t* group_size_list;
};
private :
// In future this Params object can be replaced easily with a CG object
Params params_;
Barrier *barrier_ptr_;
cutlass::PipelineState<SequenceDepth> stage_;
static constexpr int Depth = SequenceDepth;
static constexpr int Length = SequenceLength;
public:
OrderedSequenceBarrierVarGroupSize() = delete;
OrderedSequenceBarrierVarGroupSize(const OrderedSequenceBarrierVarGroupSize&) = delete;
OrderedSequenceBarrierVarGroupSize(OrderedSequenceBarrierVarGroupSize&&) = delete;
OrderedSequenceBarrierVarGroupSize& operator=(const OrderedSequenceBarrierVarGroupSize&) = delete;
OrderedSequenceBarrierVarGroupSize& operator=(OrderedSequenceBarrierVarGroupSize&&) = delete;
~OrderedSequenceBarrierVarGroupSize() = default;
CUTLASS_DEVICE
OrderedSequenceBarrierVarGroupSize(SharedStorage& storage, Params const& params) :
params_(params),
barrier_ptr_(&storage.barrier_[0][0]),
// Group 0 - starts with an opposite phase
stage_({0, params.group_id == 0, 0}) {
int warp_idx = cutlass::canonical_warp_idx_sync();
int lane_predicate = cute::elect_one_sync();
// Barrier FULL, EMPTY init
// Init is done only by the one elected thread of the block
if (warp_idx == 0 && lane_predicate) {
for (int d = 0; d < Depth; ++d) {
for (int l = 0; l < Length; ++l) {
barrier_ptr_[d * Length + l].init(*(params.group_size_list + l));
}
}
}
cutlass::arch::fence_barrier_init();
}
// Wait on a stage to be unlocked
CUTLASS_DEVICE
void wait() {
get_barrier_for_current_stage(params_.group_id).wait(stage_.phase());
}
// Signal completion of Stage and move to the next stage
// (group_id) signals to (group_id+1)
CUTLASS_DEVICE
void arrive() {
int signalling_id = (params_.group_id + 1) % Length;
get_barrier_for_current_stage(signalling_id).arrive();
++stage_;
}
CUTLASS_DEVICE
void advance() {
++stage_;
}
private:
CUTLASS_DEVICE
Barrier& get_barrier_for_current_stage(int group_id) {
return barrier_ptr_[stage_.index() * Length + group_id];
}
};
} // flash
@@ -1,180 +0,0 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cuda.h>
#include <vector>
#ifdef OLD_GENERATOR_PATH
#include <ATen/CUDAGeneratorImpl.h>
#else
#include <ATen/cuda/CUDAGeneratorImpl.h>
#endif
#include <ATen/cuda/CUDAGraphsUtils.cuh> // For at::cuda::philox::unpack
#include "cutlass/fast_math.h" // For cutlass::FastDivmod
////////////////////////////////////////////////////////////////////////////////////////////////////
struct Qkv_params {
using index_t = int64_t;
// The QKV matrices.
void *__restrict__ q_ptr;
void *__restrict__ k_ptr;
void *__restrict__ v_ptr;
void *__restrict__ delta_s_ptr;
// The QKV scale factor matrices.
void *__restrict__ sfq_ptr;
void *__restrict__ sfk_ptr;
void *__restrict__ sfv_ptr;
// The stride between rows of the Q, K and V matrices.
index_t q_batch_stride;
index_t k_batch_stride;
index_t v_batch_stride;
index_t q_row_stride;
index_t k_row_stride;
index_t v_row_stride;
index_t q_head_stride;
index_t k_head_stride;
index_t v_head_stride;
index_t ds_batch_stride;
index_t ds_row_stride;
index_t ds_head_stride;
// The stride of the Q, K and V scale factor matrices.
index_t sfq_batch_stride;
index_t sfk_batch_stride;
index_t sfv_batch_stride;
index_t sfq_row_stride;
index_t sfk_row_stride;
index_t sfv_row_stride;
index_t sfq_head_stride;
index_t sfk_head_stride;
index_t sfv_head_stride;
// The number of heads.
int h, h_k;
// In the case of multi-query and grouped-query attention (MQA/GQA), nheads_k could be
// different from nheads (query).
int h_h_k_ratio; // precompute h / h_k,
};
////////////////////////////////////////////////////////////////////////////////////////////////////
struct Flash_fwd_params : public Qkv_params {
// The O matrix (output).
void * __restrict__ o_ptr;
void * __restrict__ oaccum_ptr;
void * __restrict__ s_ptr;
// The stride between rows of O.
index_t o_batch_stride;
index_t o_row_stride;
index_t o_head_stride;
// The pointer to the P matrix.
void * __restrict__ p_ptr;
// The pointer to the softmax sum.
void * __restrict__ softmax_lse_ptr;
void * __restrict__ softmax_lseaccum_ptr;
// The dimensions.
int b, seqlen_q, seqlen_k, seqlen_knew, d, seqlen_q_rounded, seqlen_k_rounded, d_rounded, rotary_dim, unpadded_seqlen_k;
cutlass::FastDivmod head_divmod, m_block_divmod;
int total_blocks;
int seqlen_s;
// The scaling factors for the kernel.
float scale_softmax;
float scale_softmax_log2;
uint32_t scale_softmax_log2_half2;
// array of length b+1 holding starting offset of each sequence.
int * __restrict__ cu_seqlens_q;
int * __restrict__ cu_seqlens_k;
// If provided, the actual length of each k sequence.
int * __restrict__ seqused_k;
int *__restrict__ blockmask;
// The K_new and V_new matrices.
void * __restrict__ knew_ptr;
void * __restrict__ vnew_ptr;
// The stride between rows of the Q, K and V matrices.
index_t knew_batch_stride;
index_t vnew_batch_stride;
index_t knew_row_stride;
index_t vnew_row_stride;
index_t knew_head_stride;
index_t vnew_head_stride;
// The cos and sin matrices for rotary embedding.
void * __restrict__ rotary_cos_ptr;
void * __restrict__ rotary_sin_ptr;
// The indices to index into the KV cache.
int * __restrict__ cache_batch_idx;
// Paged KV cache
int * __restrict__ block_table;
index_t block_table_batch_stride;
int page_block_size;
// The dropout probability (probability of keeping an activation).
float p_dropout;
// uint32_t p_dropout_in_uint;
// uint16_t p_dropout_in_uint16_t;
uint8_t p_dropout_in_uint8_t;
// Scale factor of 1 / (1 - p_dropout).
float rp_dropout;
float scale_softmax_rp_dropout;
// Local window size
int window_size_left, window_size_right;
// Random state.
at::PhiloxCudaState philox_args;
// Pointer to the RNG seed (idx 0) and offset (idx 1).
uint64_t * rng_state;
bool is_bf16;
bool is_e4m3;
bool is_causal;
bool per_block_mean;
bool single_level_p_quant; // If true, use single-level 1x16 block scale quantization for P (like V), instead of two-level quantization
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
bool is_seqlens_k_cumulative;
bool is_rotary_interleaved;
int num_splits; // For split-KV version
void * __restrict__ alibi_slopes_ptr;
index_t alibi_slopes_batch_stride;
int * __restrict__ tile_count_semaphore;
};
////////////////////////////////////////////////////////////////////////////////////////////////////
@@ -1,190 +0,0 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cmath>
#include "cute/tensor.hpp"
#include "cutlass/numeric_types.h"
#include "utils.h"
namespace flash {
using namespace cute;
template <int Rows>
struct SoftmaxFused{
using TensorT = decltype(make_fragment_like<float>(Shape<Int<Rows>>{}));
TensorT row_sum, row_max, scores_scale;
static constexpr float fp8_scalexfp4_scale = 1.f / (448 * 6);
static constexpr float fp8_scalexfp4_scale_log2 = -11.392317422778762f; //log2f(fp8_scalexfp4_scale)
static constexpr float fp4_scale_log2 = -2.584962500721156f; // log2f(fp4_scale)
static constexpr int RowReductionThr = 4;
// If true, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly (standard per-block FP4 quantization like V)
// If false (default), use two-level quantization: s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1)
bool single_level_p_quant;
CUTLASS_DEVICE SoftmaxFused(bool single_level = false) : single_level_p_quant(single_level) {};
template<bool FirstTile, bool InfCheck = false, typename TensorAcc, typename TensorMax>
CUTLASS_DEVICE auto online_softmax_with_quant(
TensorAcc& acc,
TensorMax& AbsMaxP,
const float softmax_scale_log2
) {
Tensor acc_reduction_view = make_tensor(acc.data(), flash::convert_to_reduction_layout(acc.layout()));
Tensor acc_conversion_view = make_tensor(acc.data(), flash::convert_to_conversion_layout(acc.layout()));
Tensor acc_conversion_flatten = group_modes<1, 5>(group_modes<0, 2>(flatten(acc_conversion_view)));
if constexpr (FirstTile) {
fill(row_max, -INFINITY);
clear(row_sum);
fill(scores_scale, 1.f);
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
CUTLASS_PRAGMA_UNROLL
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), acc_reduction_view(mi, make_coord(ei, ni)));
}
float max_recv = __shfl_xor_sync(int32_t(-1), AbsMaxP(mi, ni), 1); // exchange max with neighbour thread of 8 elements
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), max_recv);
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
}
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
row_max(mi) = fmaxf(row_max(mi), max_recv);
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
// - Pre-scales P to [0, 448×6] range before φ, output scaled by s_P1
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
// - No s_P1, just standard per-block FP4 quantization φ
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
const float max_scaled = InfCheck
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
}
// s_P2 = max(P_block)/6 — per-block scale factor from φ function (same formula for both modes)
// The difference is in max_scaled: two-level includes 448×6 pre-scaling, single-level doesn't
CUTLASS_PRAGMA_UNROLL
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
}
}
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
row_sum(mi) += acc_reduction_view(mi, ni);
}
}
}
else {
Tensor scores_max_prev = make_fragment_like(row_max);
cute::copy(row_max, scores_max_prev);
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
float local_max = -INFINITY;
CUTLASS_PRAGMA_UNROLL
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
local_max = fmaxf(local_max, acc_reduction_view(mi, make_coord(ei, ni)));
}
float max_recv = __shfl_xor_sync(int32_t(-1), local_max, 1); // exchange max with neighbour thread of 8 elements
AbsMaxP(mi, ni) = fmaxf(local_max, max_recv);
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
}
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
row_max(mi) = fmaxf(row_max(mi), max_recv);
float scores_max_cur = !InfCheck
? row_max(mi)
: (row_max(mi) == -INFINITY ? 0.0f : row_max(mi));
scores_scale(mi) = flash::ptx_exp2((scores_max_prev(mi) - scores_max_cur) * softmax_scale_log2);
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
const float max_scaled = InfCheck
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
row_sum(mi) = row_sum(mi) * scores_scale(mi);
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
row_sum(mi) += acc_reduction_view(mi, ni);
}
// s_P2 = max(P_block)/6 — per-block scale factor from φ function
CUTLASS_PRAGMA_UNROLL
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
}
// scores_scale(mi) = max_scaled;
}
}
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < size(AbsMaxP); ++i) {
CUTLASS_PRAGMA_UNROLL
for (int j = 0; j < size<0>(acc_conversion_flatten); ++j)
acc_conversion_flatten(j, i) /= AbsMaxP(i);
}
}
template<typename TensorAcc>
CUTLASS_DEVICE void finalize(TensorAcc& o_store) {
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size(row_max); ++mi) {
CUTLASS_PRAGMA_UNROLL
for (int i = 1; i < RowReductionThr; i <<= 1) {
float sum_recv = __shfl_xor_sync(int32_t(-1), row_sum(mi), i);
row_sum(mi) += sum_recv;
}
float sum = row_sum(mi);
float inv_sum = (sum == 0.f || sum != sum) ? 0.f : 1 / sum;
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
o_store_reduction_view(mi, ni) *= inv_sum;
}
}
}
template<typename TensorAcc>
CUTLASS_DEVICE void rescale_o(TensorAcc& o_store, TensorAcc const& o_tmp) {
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
Tensor o_tmp_reduction_view = make_tensor(o_tmp.data(), flash::convert_to_reduction_layout(o_tmp.layout()));
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size(row_max); ++mi) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
o_store_reduction_view(mi, ni) = o_store_reduction_view(mi, ni) * scores_scale(mi) + o_tmp_reduction_view(mi, ni);
}
}
}
};
} // namespace flash
@@ -1,83 +0,0 @@
// Inspired by
// https://github.com/NVIDIA/DALI/blob/main/include/dali/core/static_switch.h
// and https://github.com/pytorch/pytorch/blob/master/aten/src/ATen/Dispatch.h
#pragma once
/// @param COND - a boolean expression to switch by
/// @param CONST_NAME - a name given for the constexpr bool variable.
/// @param ... - code to execute for true and false
///
/// Usage:
/// ```
/// BOOL_SWITCH(flag, BoolConst, [&] {
/// some_function<BoolConst>(...);
/// });
/// ```
//
#define BOOL_SWITCH(COND, CONST_NAME, ...) \
[&] { \
if (COND) { \
constexpr static bool CONST_NAME = true; \
return __VA_ARGS__(); \
} else { \
constexpr static bool CONST_NAME = false; \
return __VA_ARGS__(); \
} \
}()
#define PREC_SWITCH(PRECTYPE, ...) \
[&] { \
if (PRECTYPE == 1) { \
using kPrecType = cutlass::half_t; \
constexpr static bool kSoftFp16 = false; \
constexpr static bool kHybrid = false; \
return __VA_ARGS__(); \
} else if (PRECTYPE == 2) { \
using kPrecType = cutlass::float_e4m3_t; \
constexpr static bool kSoftFp16 = false; \
constexpr static bool kHybrid = false; \
return __VA_ARGS__(); \
} else if (PRECTYPE == 3) { \
using kPrecType = cutlass::float_e4m3_t; \
constexpr static bool kSoftFp16 = false; \
constexpr static bool kHybrid = true; \
return __VA_ARGS__(); \
} else if (PRECTYPE == 4) { \
using kPrecType = cutlass::float_e4m3_t; \
constexpr static bool kSoftFp16 = true; \
constexpr static bool kHybrid = false; \
return __VA_ARGS__(); \
} \
}()
#define HEADDIM_SWITCH(HEADDIM, ...) \
[&] { \
if (HEADDIM == 64) { \
constexpr static int kHeadSize = 64; \
return __VA_ARGS__(); \
} else if (HEADDIM == 128) { \
constexpr static int kHeadSize = 128; \
return __VA_ARGS__(); \
} else if (HEADDIM == 256) { \
constexpr static int kHeadSize = 256; \
return __VA_ARGS__(); \
} \
}()
#define SEQLEN_SWITCH(USE_VAR_SEQ_LEN, SEQ_LEN_OUT_OF_BOUND_CHECK, ...) \
[&] { \
if (!USE_VAR_SEQ_LEN) { \
if (SEQ_LEN_OUT_OF_BOUND_CHECK) { \
using kSeqLenTraitsType = FixedSeqLenTraits<true>; \
return __VA_ARGS__(); \
} else { \
using kSeqLenTraitsType = FixedSeqLenTraits<false>; \
return __VA_ARGS__(); \
} \
} else { \
using kSeqLenTraitsType = VarSeqLenTraits; \
return __VA_ARGS__(); \
} \
}()
@@ -1,304 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cutlass/fast_math.h"
namespace flash {
///////////////////////////////////////////////////////////////////////////////
class StaticPersistentTileSchedulerOld {
//
// Data members
//
private:
int current_work_linear_idx_;
cutlass::FastDivmod const &m_block_divmod, &head_divmod;
int const total_blocks;
public:
struct WorkTileInfo {
int M_idx = 0;
int H_idx = 0;
int B_idx = 0;
bool is_valid_tile = false;
CUTLASS_HOST_DEVICE
bool
is_valid() const {
return is_valid_tile;
}
CUTLASS_HOST_DEVICE
static WorkTileInfo
invalid_work_tile() {
return {-1, -1, -1, false};
}
};
public:
CUTLASS_DEVICE explicit StaticPersistentTileSchedulerOld(cutlass::FastDivmod const &m_block_divmod_,
cutlass::FastDivmod const &head_divmod_,
int const total_blocks_) :
m_block_divmod(m_block_divmod_), head_divmod(head_divmod_), total_blocks(total_blocks_) {
// MSVC requires protecting use of CUDA-specific nonstandard syntax,
// like blockIdx and gridDim, with __CUDA_ARCH__.
#if defined(__CUDA_ARCH__)
// current_work_linear_idx_ = blockIdx.x + blockIdx.y * gridDim.x + blockIdx.z * gridDim.x * gridDim.y;
current_work_linear_idx_ = blockIdx.x;
#else
CUTLASS_ASSERT(false && "This line should never be reached");
#endif
}
CUTLASS_DEVICE
WorkTileInfo
get_current_work() const {
return get_current_work_for_linear_idx(current_work_linear_idx_);
}
CUTLASS_DEVICE
WorkTileInfo
get_current_work_for_linear_idx(int linear_idx) const {
if (linear_idx >= total_blocks) {
return WorkTileInfo::invalid_work_tile();
}
// Map worker's linear index into the CTA tiled problem shape to the corresponding MHB indices
int M_idx, H_idx, B_idx;
int quotient = m_block_divmod.divmod(M_idx, linear_idx);
B_idx = head_divmod.divmod(H_idx, quotient);
return {M_idx, H_idx, B_idx, true};
}
CUTLASS_DEVICE
void
// advance_to_next_work(int advance_count = 1) {
advance_to_next_work() {
// current_work_linear_idx_ += int(gridDim.x * gridDim.y * gridDim.z);
current_work_linear_idx_ += int(gridDim.x);
}
CUTLASS_DEVICE
WorkTileInfo
fetch_next_work() {
WorkTileInfo new_work_tile_info;
advance_to_next_work();
new_work_tile_info = get_current_work();
return new_work_tile_info;
}
};
///////////////////////////////////////////////////////////////////////////////
class SingleTileScheduler {
public:
// Host side kernel arguments
struct Arguments {
int const num_blocks_m, num_head, num_batch;
int const* tile_count_semaphore = nullptr;
};
// Device side kernel params
struct Params {};
static Params
to_underlying_arguments(Arguments const& args) {
return {};
}
static dim3
get_grid_dim(Arguments const& args, int num_sm) {
return {uint32_t(args.num_blocks_m), uint32_t(args.num_head), uint32_t(args.num_batch)};
}
struct WorkTileInfo {
int M_idx = 0;
int H_idx = 0;
int B_idx = 0;
bool is_valid_tile = false;
CUTLASS_DEVICE
bool
is_valid(Params const& params) const {
return is_valid_tile;
}
CUTLASS_DEVICE
cute::tuple<int32_t, int32_t, int32_t>
get_block_coord(Params const& params) const {
return {M_idx, H_idx, B_idx};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params) const {
return {-1, -1, -1, false};
}
};
CUTLASS_DEVICE
WorkTileInfo
get_initial_work() const {
return {int(blockIdx.x), int(blockIdx.y), int(blockIdx.z), true};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
return {-1, -1, -1, false};
}
};
///////////////////////////////////////////////////////////////////////////////
class StaticPersistentTileScheduler {
public:
// Host side kernel arguments
struct Arguments {
int const num_blocks_m, num_head, num_batch;
int const* tile_count_semaphore = nullptr;
};
// Device side kernel params
struct Params {
int total_blocks;
cutlass::FastDivmod m_block_divmod, head_divmod;
};
static Params
to_underlying_arguments(Arguments const& args) {
return {args.num_blocks_m * args.num_head * args.num_batch,
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head)};
}
static dim3
get_grid_dim(Arguments const& args, int num_sm) {
return {uint32_t(num_sm)};
}
struct WorkTileInfo {
int tile_idx;
CUTLASS_DEVICE
bool
is_valid(Params const& params) const {
return tile_idx < params.total_blocks;
}
CUTLASS_DEVICE
cute::tuple<int32_t, int32_t, int32_t>
get_block_coord(Params const& params) const {
int m_block, bidh, bidb;
bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
return {m_block, bidh, bidb};
}
};
CUTLASS_DEVICE
WorkTileInfo
get_initial_work() const {
return {int(blockIdx.x)};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
return {current_work.tile_idx + int(gridDim.x)};
}
};
class DynamicPersistentTileScheduler {
public:
// Host side kernel arguments
struct Arguments {
int const num_blocks_m, num_head, num_batch;
int const* tile_count_semaphore;
};
// Device side kernel params
struct Params {
int const total_blocks;
cutlass::FastDivmod const m_block_divmod, head_divmod;
int const* tile_count_semaphore;
};
static Params
to_underlying_arguments(Arguments const& args) {
return {args.num_blocks_m * args.num_head * args.num_batch,
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head),
args.tile_count_semaphore};
}
static dim3
get_grid_dim(Arguments const& args, int num_sm) {
return {uint32_t(num_sm)};
}
using WorkTileInfo = StaticPersistentTileScheduler::WorkTileInfo;
// struct WorkTileInfo {
// int tile_idx;
// CUTLASS_DEVICE
// bool
// is_valid(Params const& params) const {
// return tile_idx < params.total_blocks;
// }
// CUTLASS_DEVICE
// cute::tuple<int32_t, int32_t, int32_t>
// get_block_coord(Params const& params) const {
// int m_block, bidh, bidb;
// bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
// return {m_block, bidh, bidb};
// }
// };
CUTLASS_DEVICE
WorkTileInfo
get_initial_work() const {
return {int(blockIdx.x)};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
return {current_work.tile_idx + int(gridDim.x)};
}
};
} // flash
@@ -1,408 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <assert.h>
#include <stdint.h>
#include <stdlib.h>
#include <cuda_fp16.h>
#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800
#include <cuda_bf16.h>
#endif
#include <cute/tensor.hpp>
#include <cutlass/array.h>
#include <cutlass/cutlass.h>
#include <cutlass/numeric_conversion.h>
#include <cutlass/numeric_types.h>
namespace flash {
using namespace cute;
////////////////////////////////////////////////////////////////////////////////////////////////////
template<typename T>
struct MaxOp {
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x > y ? x : y; }
};
template <>
struct MaxOp<float> {
// This is slightly faster
__device__ __forceinline__ float operator()(float const &x, float const &y) { return max(x, y); }
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<typename T>
struct SumOp {
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x + y; }
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<int THREADS>
struct Allreduce {
static_assert(THREADS == 32 || THREADS == 16 || THREADS == 8 || THREADS == 4);
template<typename T, typename Operator>
static __device__ __forceinline__ T run(T x, Operator &op) {
constexpr int OFFSET = THREADS / 2;
x = op(x, __shfl_xor_sync(uint32_t(-1), x, OFFSET));
return Allreduce<OFFSET>::run(x, op);
}
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<>
struct Allreduce<2> {
template<typename T, typename Operator>
static __device__ __forceinline__ T run(T x, Operator &op) {
x = op(x, __shfl_xor_sync(uint32_t(-1), x, 1));
return x;
}
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
__device__ __forceinline__ void thread_reduce_(Tensor<Engine0, Layout0> const &tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
CUTE_STATIC_ASSERT_V(size<0>(summary) == size<0>(tensor));
#pragma unroll
for (int mi = 0; mi < size<0>(tensor); mi++) {
summary(mi) = zero_init ? tensor(mi, 0) : op(summary(mi), tensor(mi, 0));
#pragma unroll
for (int ni = 1; ni < size<1>(tensor); ni++) {
summary(mi) = op(summary(mi), tensor(mi, ni));
}
}
}
template<typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
__device__ __forceinline__ void quad_allreduce_(Tensor<Engine0, Layout0> &dst, Tensor<Engine1, Layout1> &src, Operator &op) {
CUTE_STATIC_ASSERT_V(size(dst) == size(src));
#pragma unroll
for (int i = 0; i < size(dst); i++){
dst(i) = Allreduce<4>::run(src(i), op);
}
}
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
__device__ __forceinline__ void reduce_(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
thread_reduce_<zero_init>(tensor, summary, op);
quad_allreduce_(summary, summary, op);
}
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__device__ __forceinline__ void reduce_max(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &max){
MaxOp<float> max_op;
reduce_<zero_init>(tensor, max, max_op);
}
template<bool zero_init=true, bool warp_reduce=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__device__ __forceinline__ void reduce_sum(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &sum){
SumOp<float> sum_op;
thread_reduce_<zero_init>(tensor, sum, sum_op);
if constexpr (warp_reduce) { quad_allreduce_(sum, sum, sum_op); }
}
__forceinline__ __device__ __half2 half_exp(__half2 x) {
uint32_t tmp_out, tmp_in;
tmp_in = reinterpret_cast<uint32_t&>(x);
asm ("ex2.approx.f16x2 %0, %1;\n"
: "=r"(tmp_out)
: "r"(tmp_in));
__half2 out = reinterpret_cast<__half2&>(tmp_out);
return out;
}
// Apply the exp to all the elements.
template <bool zero_init=false, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__forceinline__ __device__ void max_scale_exp2_sum(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> &max, Tensor<Engine1, Layout1> &sum, const float scale) {
static_assert(Layout0::rank == 2, "Only support 2D Tensor"); static_assert(Layout1::rank == 1, "Only support 1D Tensor"); CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
#pragma unroll
for (int mi = 0; mi < size<0>(tensor); ++mi) {
MaxOp<float> max_op;
max(mi) = zero_init ? tensor(mi, 0) : max_op(max(mi), tensor(mi, 0));
#pragma unroll
for (int ni = 1; ni < size<1>(tensor); ni++) {
max(mi) = max_op(max(mi), tensor(mi, ni));
}
max(mi) = Allreduce<4>::run(max(mi), max_op);
// If max is -inf, then all elements must have been -inf (possibly due to masking).
// We don't want (-inf - (-inf)) since that would give NaN.
const float max_scaled = max(mi) == -INFINITY ? 0.f : max(mi) * scale;
sum(mi) = 0;
#pragma unroll
for (int ni = 0; ni < size<1>(tensor); ++ni) {
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
// max * log_2(e)) This allows the compiler to use the ffma
// instruction instead of fadd and fmul separately.
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
sum(mi) += tensor(mi, ni);
}
}
}
// Apply the exp to all the elements.
template <bool Scale_max=true, bool Check_inf=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__forceinline__ __device__ void scale_apply_exp2(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> const &max, const float scale) {
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
#pragma unroll
for (int mi = 0; mi < size<0>(tensor); ++mi) {
// If max is -inf, then all elements must have been -inf (possibly due to masking).
// We don't want (-inf - (-inf)) since that would give NaN.
// If we don't have float around M_LOG2E the multiplication is done in fp64.
const float max_scaled = Check_inf
? (max(mi) == -INFINITY ? 0.f : (max(mi) * (Scale_max ? scale : float(M_LOG2E))))
: (max(mi) * (Scale_max ? scale : float(M_LOG2E)));
#pragma unroll
for (int ni = 0; ni < size<1>(tensor); ++ni) {
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
// max * log_2(e)) This allows the compiler to use the ffma
// instruction instead of fadd and fmul separately.
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
}
}
}
////////////////////////////////////////////////////////////////////////////////////////////////////
__forceinline__ __device__ float ptx_exp2(float x) {
float y;
asm volatile("ex2.approx.ftz.f32 %0, %1;" : "=f"(y) : "f"(x));
return y;
}
CUTLASS_DEVICE void
packed_float_to_ue4m3(
float const &f0, float const &f1, float const &f2, float const &f3,
uint32_t &out
) {
asm volatile( \
"{\n" \
".reg .b16 lo;\n" \
".reg .b16 hi;\n" \
"cvt.rn.satfinite.e4m3x2.f32 lo, %2, %1;\n" \
"cvt.rn.satfinite.e4m3x2.f32 hi, %4, %3;\n" \
"mov.b32 %0, {lo, hi};\n" \
"}" \
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3));
}
CUTLASS_DEVICE void
packed_float_to_e2m1(
float const &f0, float const &f1, float const &f2, float const& f3,
float const &f4, float const &f5, float const &f6, float const& f7,
uint32_t &out
) {
asm volatile( \
"{\n" \
".reg .b8 byte0;\n" \
".reg .b8 byte1;\n" \
".reg .b8 byte2;\n" \
".reg .b8 byte3;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n" \
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n" \
"}" \
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3),
"f"(f4), "f"(f5), "f"(f6), "f"(f7));
}
CUTLASS_DEVICE void
add(float2 & c,
float2 const& a,
float2 const& b)
{
asm volatile("add.f32x2 %0, %1, %2;\n"
: "=l"(reinterpret_cast<uint64_t &>(c))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)));
}
CUTLASS_DEVICE void
add_inplace(float2 &a,
float2 const& b)
{
asm volatile("add.f32x2 %0, %0, %1;\n"
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
);
}
CUTLASS_DEVICE void
sub(float2 & c,
float2 const& a,
float2 const& b)
{
asm volatile("sub.f32x2 %0, %1, %2;\n"
: "=l"(reinterpret_cast<uint64_t &>(c))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)));
}
CUTLASS_DEVICE void
sub_inplace(float2 &a,
float2 const& b)
{
asm volatile("sub.f32x2 %0, %0, %1;\n"
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
);
}
CUTLASS_DEVICE void
mul(float2 & c,
float2 const& a,
float2 const& b)
{
asm volatile("mul.f32x2 %0, %1, %2;\n"
: "=l"(reinterpret_cast<uint64_t &>(c))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)));
}
CUTLASS_DEVICE void
fma(float2 & d,
float2 const& a,
float2 const& b,
float2 const& c)
{
asm volatile("fma.rn.f32x2 %0, %1, %2, %3;\n"
: "=l"(reinterpret_cast<uint64_t &>(d))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)),
"l"(reinterpret_cast<uint64_t const&>(c)));
}
CUTLASS_DEVICE void
fma_inplace(float2 &a,
float2 const& b,
float2 const& c)
{
asm volatile("fma.rn.f32x2 %0, %0, %1, %2;\n"
: "+l"(reinterpret_cast<uint64_t &>(a))
: "l"(reinterpret_cast<uint64_t const&>(b)),
"l"(reinterpret_cast<uint64_t const&>(c)));
}
////////////////////////////////////////////////////////////////////////////////////////////////////
template <
class Layout
>
CUTLASS_DEVICE constexpr
auto convert_to_reduction_layout(Layout mma_layout) {
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
return make_layout(
make_layout(get<0,1>(mma_layout), get<1>(mma_layout)),
make_layout(get<0,0>(mma_layout), get<2>(mma_layout))
);
}
template <
class Tensor
>
CUTLASS_DEVICE constexpr
auto convert_to_reduction_tensor(Tensor mma_tensor) {
return make_tensor(mma_tensor.data(), convert_to_reduction_layout(mma_tensor.layout()));
}
template <
class Layout
>
CUTLASS_DEVICE constexpr
auto convert_to_conversion_layout(Layout mma_layout) {
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
constexpr int MmaAtomN = size<0, 0>(mma_layout);
constexpr int MmaAtomM = size<0, 1>(mma_layout);
constexpr int MmaM = size<1>(mma_layout);
constexpr int MmaN = size<2>(mma_layout);
static_assert(MmaAtomN == 8, "MmaAtomN should be 8.");
static_assert(MmaAtomM == 2, "MmaAtomM should be 2.");
static_assert(MmaN % 2 == 0, "MmaN should be multiple of 2.");
auto mma_n_division = zipped_divide(
layout<2>(mma_layout), make_tile(_2{})
);
return make_layout(
make_layout(layout<0,0>(mma_layout), make_layout(layout<0,1>(mma_layout), layout<0>(mma_n_division))),
layout<1>(mma_layout), layout<1>(mma_n_division)
);
}
template <
class Tensor
>
CUTLASS_DEVICE constexpr
auto convert_to_conversion_tensor(Tensor mma_tensor) {
return make_tensor(mma_tensor.data(), convert_to_conversion_layout(mma_tensor.layout()));
}
////////////////////////////////////////////////////////////////////////////////////////////////////
template <bool Is_even_MN=true, bool Is_even_K=true, bool Clear_OOB_MN=false, bool Clear_OOB_K=true,
typename TiledCopy, typename Engine0, typename Layout0, typename Engine1, typename Layout1,
typename Engine2, typename Layout2, typename Engine3, typename Layout3>
CUTLASS_DEVICE void copy(TiledCopy tiled_copy, Tensor<Engine0, Layout0> const &S,
Tensor<Engine1, Layout1> &D, Tensor<Engine2, Layout2> const &identity_MN,
Tensor<Engine3, Layout3> const &predicate_K, const int max_MN=0) {
CUTE_STATIC_ASSERT_V(rank(S) == Int<3>{});
CUTE_STATIC_ASSERT_V(rank(D) == Int<3>{});
CUTE_STATIC_ASSERT_V(size<0>(S) == size<0>(D)); // MMA
CUTE_STATIC_ASSERT_V(size<1>(S) == size<1>(D)); // MMA_M
CUTE_STATIC_ASSERT_V(size<2>(S) == size<2>(D)); // MMA_K
// There's no case where !Clear_OOB_K && Clear_OOB_MN
static_assert(!(Clear_OOB_MN && !Clear_OOB_K));
#pragma unroll
for (int m = 0; m < size<1>(S); ++m) {
if (Is_even_MN || get<0>(identity_MN(0, m, 0)) < max_MN) {
#pragma unroll
for (int k = 0; k < size<2>(S); ++k) {
if (Is_even_K || predicate_K(k)) {
cute::copy(tiled_copy, S(_, m, k), D(_, m, k));
} else if (Clear_OOB_K) {
cute::clear(D(_, m, k));
}
}
} else if (Clear_OOB_MN) {
cute::clear(D(_, m, _));
}
}
}
} // namespace flash
@@ -1 +0,0 @@
__version__ = "3.0.0.b1"
@@ -1,90 +0,0 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import fp4quant
from triton.tools.mxfp import MXFP4Tensor
from bench_utils import bench_kineto
b = 1
h = 32
n = 16384
d = 128
def test():
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
fp4quant.scaled_fp4_quant_permute(q, o, o_s, 1)
test()
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
throughput = IO / t * 1e-9
print(f"Throughput: {throughput:.2f} GB/s")
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
B, H, M, N = x.shape
x = x.view(B, H, M, N // 16, 16)
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
if all_ones:
scales = torch.ones_like(scales)
x_scaled = x / scales
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
permuted_fp8_scale = None
if permuted:
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
b = 2
h = 4
n = 251
n_padded = (n + 127) // 128 * 128
d = 128
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
k_permute = [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
o_permuted = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
o_s_permuted = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant_permute(q, o_permuted, o_s_permuted, 1)
# padding
if n % 128 != 0:
o_permuted_gt = torch.cat([o, torch.zeros((b, h, n_padded - n, d // 2), dtype=torch.uint8, device='cuda')], dim=2)
o_s_permuted_gt = torch.cat([o_s, torch.zeros((b, h, n_padded - n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')], dim=2)
else:
o_permuted_gt = o
o_s_permuted_gt = o_s
# use scale_and_fp4_tensor + torch permutation to get the ground truth
o_permuted_gt = o_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 2)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 2)
o_s_permuted_gt = o_s_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 16)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 16)
assert((o_permuted - o_permuted_gt).abs().max() == 0)
assert((o_s_permuted.float() - o_s_permuted_gt.float()).abs().max() == 0)
print("All tests passed!")
@@ -1,86 +0,0 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import fp4quant
from triton.tools.mxfp import MXFP4Tensor
from bench_utils import bench_kineto
b = 1
h = 32
n = 16384
d = 128
def test():
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
test()
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
throughput = IO / t * 1e-9
print(f"Throughput: {throughput:.2f} GB/s")
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
B, H, M, N = x.shape
x = x.view(B, H, M, N // 16, 16)
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
if all_ones:
scales = torch.ones_like(scales)
x_scaled = x / scales
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
permuted_fp8_scale = None
if permuted:
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
b = 2
h = 4
n = 251
d = 128
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale = scale_and_fp4_tensor(q, packed_dim=3)
assert((fp8_scale.float() - o_s.float()).abs().max() == 0)
o_binary = [
(int(bin_str[:4], 2), int(bin_str[4:], 2))
for bin_str in [format(x.item(), '08b') for x in o.view(-1)]
]
o_binary_gt = [
(int(bin_str[:4], 2), int(bin_str[4:], 2))
for bin_str in [format(x.item(), '08b') for x in packed_fp4.view(-1)]
]
for i in range(len(o_binary)):
# check contiguous 4 bits. Difference should be at most one
assert(abs(o_binary[i][0] - o_binary_gt[i][0]) <= 1)
assert(abs(o_binary[i][1] - o_binary_gt[i][1]) <= 1)
print("All tests passed!")
@@ -1,86 +0,0 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import fp4quant
from triton.tools.mxfp import MXFP4Tensor
from bench_utils import bench_kineto
b = 1
h = 32
n = 16384
d = 128
def test():
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
o = torch.empty((b, h, d, n // 2), device="cuda", dtype=torch.uint8)
o_s = torch.empty((b, h, d, n // 16), device="cuda", dtype=torch.float8_e4m3fn)
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
test()
t = bench_kineto(test, "scaled_fp4_quant_trans_kernel", suppress_kineto_output=True)
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
throughput = IO / t * 1e-9
print(f"Throughput: {throughput:.2f} GB/s")
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
B, H, M, N = x.shape
x = x.view(B, H, M, N // 16, 16)
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
if all_ones:
scales = torch.ones_like(scales)
x_scaled = x / scales
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
permuted_fp8_scale = None
if permuted:
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
b = 2
h = 4
n = 491
n_padded = (n + 127) // 128 * 128
d = 128
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
o = torch.empty((b, h, d, n_padded // 2), dtype=torch.uint8, device='cuda')
o_s = torch.empty((b, h, d, n_padded // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
if n % 128 != 0:
q_padded = torch.cat([q, torch.zeros((b, h, n_padded - n, d), dtype=torch.float16, device='cuda')], dim=2)
else:
q_padded = q
# use torch transpose + scaled_fp4_quant to get the ground truth
q_padded = q_padded.transpose(2, 3).reshape(b, h, n_padded, d).contiguous()
o_gt = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
o_s_gt = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant(q_padded, o_gt, o_s_gt, 1)
o_gt = o_gt.reshape(b, h, d, n_padded // 2).contiguous()
o_s_gt = o_s_gt.reshape(b, h, d, n_padded // 16).contiguous()
assert((o_s_gt.float() - o_s.float()).abs().max() == 0)
assert((o_gt - o).abs().max() == 0)
print("All tests passed!")
@@ -1,169 +0,0 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import os
import sys
import torch
import torch.distributed as dist
def bench(fn, num_warmups: int = 5, num_tests: int = 10,
high_precision: bool = False):
# Flush L2 cache with 256 MB data
torch.cuda.synchronize()
cache = torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda')
cache.zero_()
# Warmup
for _ in range(num_warmups):
fn()
# Add a large kernel to eliminate the CPU launch overhead
if high_precision:
x = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
y = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
x @ y
# Testing
start_event = torch.cuda.Event(enable_timing=True)
end_event = torch.cuda.Event(enable_timing=True)
start_event.record()
for i in range(num_tests):
fn()
end_event.record()
torch.cuda.synchronize()
return start_event.elapsed_time(end_event) / num_tests
class empty_suppress:
def __enter__(self):
return self
def __exit__(self, *_):
pass
class suppress_stdout_stderr:
def __enter__(self):
self.outnull_file = open(os.devnull, 'w')
self.errnull_file = open(os.devnull, 'w')
self.old_stdout_fileno_undup = sys.stdout.fileno()
self.old_stderr_fileno_undup = sys.stderr.fileno()
self.old_stdout_fileno = os.dup(sys.stdout.fileno())
self.old_stderr_fileno = os.dup(sys.stderr.fileno())
self.old_stdout = sys.stdout
self.old_stderr = sys.stderr
os.dup2(self.outnull_file.fileno(), self.old_stdout_fileno_undup)
os.dup2(self.errnull_file.fileno(), self.old_stderr_fileno_undup)
sys.stdout = self.outnull_file
sys.stderr = self.errnull_file
return self
def __exit__(self, *_):
sys.stdout = self.old_stdout
sys.stderr = self.old_stderr
os.dup2(self.old_stdout_fileno, self.old_stdout_fileno_undup)
os.dup2(self.old_stderr_fileno, self.old_stderr_fileno_undup)
os.close(self.old_stdout_fileno)
os.close(self.old_stderr_fileno)
self.outnull_file.close()
self.errnull_file.close()
def bench_kineto(fn, kernel_names, num_tests: int = 30, suppress_kineto_output: bool = False,
trace_path: str = None, barrier_comm_profiling: bool = False, flush_l2: bool = False):
# Conflict with Nsight Systems
using_nsys = os.environ.get('DG_NSYS_PROFILING', False)
# For some auto-tuning kernels with prints
fn()
# Profile
suppress = suppress_stdout_stderr if suppress_kineto_output and not using_nsys else empty_suppress
with suppress():
schedule = torch.profiler.schedule(wait=0, warmup=1, active=1, repeat=1) if not using_nsys else None
profiler = torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CUDA], schedule=schedule) if not using_nsys else empty_suppress()
with profiler:
for i in range(2):
# NOTES: use a large kernel and a barrier to eliminate the unbalanced CPU launch overhead
if barrier_comm_profiling:
lhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
rhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
lhs @ rhs
dist.all_reduce(torch.ones(1, dtype=torch.float, device='cuda'))
for _ in range(num_tests):
if flush_l2:
torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda').zero_()
fn()
if not using_nsys:
profiler.step()
# Return 1 if using Nsight Systems
if using_nsys:
return 1
# Parse the profiling table
assert isinstance(kernel_names, str) or isinstance(kernel_names, tuple)
is_tupled = isinstance(kernel_names, tuple)
prof_lines = profiler.key_averages().table(sort_by='cuda_time_total', max_name_column_width=100).split('\n')
kernel_names = (kernel_names, ) if isinstance(kernel_names, str) else kernel_names
assert all([isinstance(name, str) for name in kernel_names])
for name in kernel_names:
assert sum([name in line for line in prof_lines]) == 1, f'Errors of the kernel {name} in the profiling table'
# Save chrome traces
if trace_path is not None:
profiler.export_chrome_trace(trace_path)
# Return average kernel times
units = {'ms': 1e3, 'us': 1e6}
kernel_times = []
for name in kernel_names:
for line in prof_lines:
if name in line:
time_str = line.split()[-2]
for unit, scale in units.items():
if unit in time_str:
kernel_times.append(float(time_str.replace(unit, '')) / scale)
break
break
return tuple(kernel_times) if is_tupled else kernel_times[0]
def calc_diff(x, y):
x, y = x.double(), y.double()
denominator = (x * x + y * y).sum()
sim = 2 * (x * y).sum() / denominator
return 1 - sim
def count_bytes(tensors):
total = 0
for t in tensors:
if isinstance(t, tuple):
total += count_bytes(t)
else:
total += t.numel() * t.element_size()
return total
@@ -1,52 +0,0 @@
#pragma once
#include <stdio.h>
#if defined(__HIPCC__)
#define HOST_DEVICE_INLINE __host__ __device__
#define DEVICE_INLINE __device__
#define HOST_INLINE __host__
#elif defined(__CUDACC__) || defined(_NVHPC_CUDA)
#define HOST_DEVICE_INLINE __host__ __device__ __forceinline__
#define DEVICE_INLINE __device__ __forceinline__
#define HOST_INLINE __host__ __forceinline__
#else
#define HOST_DEVICE_INLINE inline
#define DEVICE_INLINE inline
#define HOST_INLINE inline
#endif
#define CUDA_CHECK(cmd) \
do { \
cudaError_t e = cmd; \
if (e != cudaSuccess) { \
printf("Failed: Cuda error %s:%d '%s'\n", __FILE__, __LINE__, \
cudaGetErrorString(e)); \
exit(EXIT_FAILURE); \
} \
} while (0)
int64_t get_device_attribute(int64_t attribute, int64_t device_id) {
static int value = [=]() {
int device = static_cast<int>(device_id);
if (device < 0) {
CUDA_CHECK(cudaGetDevice(&device));
}
int value;
CUDA_CHECK(cudaDeviceGetAttribute(
&value, static_cast<cudaDeviceAttr>(attribute), device));
return static_cast<int>(value);
}();
return value;
}
namespace cuda_utils {
template <typename T>
HOST_DEVICE_INLINE constexpr std::enable_if_t<std::is_integral_v<T>, T>
ceil_div(T a, T b) {
return (a + b - 1) / b;
}
}; // namespace cuda_utils
@@ -1,629 +0,0 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include <torch/all.h>
#include <torch/python.h>
#include <torch/nn/functional.h>
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAGuard.h>
#include <cuda_runtime_api.h>
#include <cuda_runtime.h>
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAGuard.h>
#include <cuda_fp8.h>
#include "cuda_utils.h"
#include "../blackwell/block_config.h"
#define DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(pytorch_dtype, c_type, ...) \
if (pytorch_dtype == at::ScalarType::Half) { \
using c_type = half; \
__VA_ARGS__ \
} else if (pytorch_dtype == at::ScalarType::BFloat16) { \
using c_type = nv_bfloat16; \
__VA_ARGS__ \
} else { \
std::ostringstream oss; \
oss << __PRETTY_FUNCTION__ << " failed to dispatch data type " << pytorch_dtype; \
TORCH_CHECK(false, oss.str()); \
}
#define DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, ...) \
if (head_dim == 64) { \
constexpr int HEAD_DIM = 64; \
__VA_ARGS__ \
} else if (head_dim == 128) { \
constexpr int HEAD_DIM = 128; \
__VA_ARGS__ \
} else { \
std::ostringstream err_msg; \
err_msg << "Unsupported head dim: " << int(head_dim); \
throw std::invalid_argument(err_msg.str()); \
}
#define CHECK_CUDA(x) \
TORCH_CHECK(x.is_cuda(), "Tensor " #x " must be on CUDA")
#define CHECK_DTYPE(x, true_dtype) \
TORCH_CHECK(x.dtype() == true_dtype, \
"Tensor " #x " must have dtype (" #true_dtype ")")
#define CHECK_DIMS(x, true_dim) \
TORCH_CHECK(x.dim() == true_dim, \
"Tensor " #x " must have dimension number (" #true_dim ")")
#define CHECK_SHAPE(x, ...) \
TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), \
"Tensor " #x " must have shape (" #__VA_ARGS__ ")")
#define CHECK_CONTIGUOUS(x) \
TORCH_CHECK(x.is_contiguous(), "Tensor " #x " must be contiguous")
#define CHECK_LASTDIM_CONTIGUOUS(x) \
TORCH_CHECK(x.stride(-1) == 1, \
"Tensor " #x " must be contiguous at the last dimension")
constexpr int CVT_FP4_ELTS_PER_THREAD = 16;
// Convert 4 float2 values into 8 e2m1 values (represented as one uint32_t).
inline __device__ uint32_t fp32_vec_to_e2m1(float2 *array) {
#if defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 1000)
uint32_t val;
asm volatile(
"{\n"
".reg .b8 byte0;\n"
".reg .b8 byte1;\n"
".reg .b8 byte2;\n"
".reg .b8 byte3;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n"
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n"
"}"
: "=r"(val)
: "f"(array[0].x), "f"(array[0].y), "f"(array[1].x), "f"(array[1].y),
"f"(array[2].x), "f"(array[2].y), "f"(array[3].x), "f"(array[3].y));
return val;
#else
return 0;
#endif
}
// Get type2 from type or vice versa (applied to half and bfloat16)
template <typename T>
struct TypeConverter {
using Type = half2;
}; // keep for generality
template <>
struct TypeConverter<half2> {
using Type = half;
};
template <>
struct TypeConverter<half> {
using Type = half2;
};
template <>
struct TypeConverter<__nv_bfloat162> {
using Type = __nv_bfloat16;
};
template <>
struct TypeConverter<__nv_bfloat16> {
using Type = __nv_bfloat162;
};
// Define a 32 bytes packed data type.
template <class Type>
struct PackedVec {
typename TypeConverter<Type>::Type elts[8];
};
template <uint32_t head_dim, uint32_t BLOCK_SIZE, bool permute, typename T>
__global__ void scaled_fp4_quant_kernel(
const T* input, uint8_t* output, uint8_t* output_sf,
int batch_size, int num_heads, int num_tokens,
int stride_bz_input, int stride_h_input, int stride_seq_input,
int stride_bz_output, int stride_h_output, int stride_seq_output,
int stride_bz_output_sf, int stride_h_output_sf, int stride_seq_output_sf) {
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
using PackedVec = PackedVec<T>;
const int batch_id = blockIdx.y;
const int head_id = blockIdx.z;
const int token_block_id = blockIdx.x;
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
"Vec size is not matched.");
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
// load input
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
int load_token_id;
if constexpr (!permute) {
load_token_id = token_id;
} else {
int local_token_id = threadIdx.x / NUM_THREADS_PER_TOKEN;
int local_token_id_residue = local_token_id % 32;
// [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
load_token_id = token_block_id * BLOCK_SIZE + (local_token_id / 32) * 32 +
(local_token_id_residue / 8) * 2 +
((local_token_id_residue % 8) / 2) * 8 +
(local_token_id_residue % 8) % 2;
}
PackedVec in_vec;
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
}
if (load_token_id < num_tokens) {
in_vec = reinterpret_cast<PackedVec const*>(input +
batch_id * stride_bz_input + // batch dim
head_id * stride_h_input + // head dim
load_token_id * stride_seq_input + // seq dim
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
}
// calculate max of every consecutive 16 elements
auto localMax = __habs2(in_vec.elts[0]);
#pragma unroll
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
}
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
}
float vecMax = float(__hmax(localMax.x, localMax.y));
// scaling factor
float SFValue = vecMax / 6.0f;
uint8_t SFValueFP8;
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
// convert input to float2 and apply scale
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
if constexpr (std::is_same<T, half>::value) {
fp2Vals[i] = __half22float2(in_vec.elts[i]);
} else {
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
}
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
}
// convert to e2m1
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
}
// save, do not check range
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
reinterpret_cast<uint32_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
token_id * stride_seq_output +
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = e2m1Vals[0];
} else {
reinterpret_cast<uint64_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
token_id * stride_seq_output +
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
}
uint8_t* output_sf_save_base = output_sf + batch_id * stride_bz_output_sf + head_id * stride_h_output_sf + (token_id / 64) * 64 * stride_seq_output_sf;
uint32_t token_id_local = token_id % 64;
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
uint32_t col_id_local = threadIdx.x % NUM_THREADS_PER_TOKEN;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
} else {
if (threadIdx.x % 2 == 0) {
uint32_t col_id_local = (threadIdx.x % NUM_THREADS_PER_TOKEN) / 2;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
}
}
}
template <uint32_t head_dim, uint32_t BLOCK_SIZE, typename T>
__global__ void scaled_fp4_quant_trans_kernel(
const T* input, uint8_t* output, uint8_t* output_sf,
int batch_size, int num_heads, int num_tokens,
int stride_bz_input, int stride_h_input, int stride_seq_input,
int stride_bz_output, int stride_h_output, int stride_d_output,
int stride_bz_output_sf, int stride_h_output_sf, int stride_d_output_sf) {
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
using PackedVec = PackedVec<T>;
const int batch_id = blockIdx.y;
const int head_id = blockIdx.z;
const int token_block_id = blockIdx.x;
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
"Vec size is not matched.");
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
constexpr uint32_t NUM_THREADS_PER_SEQ = BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD;
// load input
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
PackedVec in_vec;
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
}
if (token_id < num_tokens) {
in_vec = reinterpret_cast<PackedVec const*>(input +
batch_id * stride_bz_input + // batch dim
head_id * stride_h_input + // head dim
token_id * stride_seq_input + // seq dim
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
}
// transpose
__shared__ T shared_input[BLOCK_SIZE * head_dim];
reinterpret_cast<PackedVec*>(shared_input)[threadIdx.x] = in_vec;
__syncthreads();
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
in_vec.elts[i].x = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i) * head_dim];
in_vec.elts[i].y = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i + 1) * head_dim];
}
// calculate max of every consecutive 16 elements
auto localMax = __habs2(in_vec.elts[0]);
#pragma unroll
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
}
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
}
float vecMax = float(__hmax(localMax.x, localMax.y));
// scaling factor
float SFValue = vecMax / 6.0f;
uint8_t SFValueFP8;
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
// convert input to float2 and apply scale
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
if constexpr (std::is_same<T, half>::value) {
fp2Vals[i] = __half22float2(in_vec.elts[i]);
} else {
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
}
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
}
// convert to e2m1
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
}
// save
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
reinterpret_cast<uint32_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = e2m1Vals[0];
} else {
reinterpret_cast<uint64_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
}
uint8_t *output_sf_save_base = output_sf +
batch_id * stride_bz_output_sf +
head_id * stride_h_output_sf +
(threadIdx.x / NUM_THREADS_PER_SEQ / 64) * 64 * stride_d_output_sf;
uint32_t row_id_local = (threadIdx.x / NUM_THREADS_PER_SEQ) % 64;
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + threadIdx.x % NUM_THREADS_PER_SEQ;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
} else {
if (threadIdx.x % 2 == 0) {
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + (threadIdx.x % NUM_THREADS_PER_SEQ) / 2;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
}
}
}
void scaled_fp4_quant(torch::Tensor const& input,
torch::Tensor const& output,
torch::Tensor const& output_sf,
int tensor_layout) {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
CHECK_CUDA(input);
CHECK_CUDA(output);
CHECK_CUDA(output_sf);
CHECK_LASTDIM_CONTIGUOUS(input);
CHECK_LASTDIM_CONTIGUOUS(output);
CHECK_LASTDIM_CONTIGUOUS(output_sf);
CHECK_DTYPE(output, at::ScalarType::Byte);
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
CHECK_DIMS(input, 4);
CHECK_DIMS(output, 4);
CHECK_DIMS(output_sf, 4);
const int batch_size = input.size(0);
const int head_dim = input.size(3);
const int stride_bz_input = input.stride(0);
const int stride_bz_output = output.stride(0);
const int stride_bz_output_sf = output_sf.stride(0);
int num_tokens, num_heads;
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
int stride_h_input, stride_h_output, stride_h_output_sf;
if (tensor_layout == 0) {
num_tokens = input.size(1);
num_heads = input.size(2);
stride_seq_input = input.stride(1);
stride_seq_output = output.stride(1);
stride_seq_output_sf = output_sf.stride(1);
stride_h_input = input.stride(2);
stride_h_output = output.stride(2);
stride_h_output_sf = output_sf.stride(2);
CHECK_SHAPE(output, batch_size, num_tokens, num_heads, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, num_tokens, num_heads, head_dim / 16);
} else {
num_tokens = input.size(2);
num_heads = input.size(1);
stride_seq_input = input.stride(2);
stride_seq_output = output.stride(2);
stride_seq_output_sf = output_sf.stride(2);
stride_h_input = input.stride(1);
stride_h_output = output.stride(1);
stride_h_output_sf = output_sf.stride(1);
CHECK_SHAPE(output, batch_size, num_heads, num_tokens, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, num_heads, num_tokens, head_dim / 16);
}
auto input_dtype = input.scalar_type();
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, false, c_type>
<<<grid, block, 0, stream>>>(
reinterpret_cast<c_type*>(input.data_ptr()),
reinterpret_cast<uint8_t*>(output.data_ptr()),
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
batch_size, num_heads, num_tokens,
stride_bz_input, stride_h_input, stride_seq_input,
stride_bz_output, stride_h_output, stride_seq_output,
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
});
});
}
void scaled_fp4_quant_permute(torch::Tensor const& input,
torch::Tensor const& output,
torch::Tensor const& output_sf,
int tensor_layout) {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
CHECK_CUDA(input);
CHECK_CUDA(output);
CHECK_CUDA(output_sf);
CHECK_LASTDIM_CONTIGUOUS(input);
CHECK_LASTDIM_CONTIGUOUS(output);
CHECK_LASTDIM_CONTIGUOUS(output_sf);
CHECK_DTYPE(output, at::ScalarType::Byte);
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
CHECK_DIMS(input, 4);
CHECK_DIMS(output, 4);
CHECK_DIMS(output_sf, 4);
const int batch_size = input.size(0);
const int head_dim = input.size(3);
const int stride_bz_input = input.stride(0);
const int stride_bz_output = output.stride(0);
const int stride_bz_output_sf = output_sf.stride(0);
int num_tokens, num_heads;
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
int stride_h_input, stride_h_output, stride_h_output_sf;
if (tensor_layout == 0) {
num_tokens = input.size(1);
num_heads = input.size(2);
stride_seq_input = input.stride(1);
stride_seq_output = output.stride(1);
stride_seq_output_sf = output_sf.stride(1);
stride_h_input = input.stride(2);
stride_h_output = output.stride(2);
stride_h_output_sf = output_sf.stride(2);
CHECK_SHAPE(output, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 16);
} else {
num_tokens = input.size(2);
num_heads = input.size(1);
stride_seq_input = input.stride(2);
stride_seq_output = output.stride(2);
stride_seq_output_sf = output_sf.stride(2);
stride_h_input = input.stride(1);
stride_h_output = output.stride(1);
stride_h_output_sf = output_sf.stride(1);
CHECK_SHAPE(output, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 16);
}
auto input_dtype = input.scalar_type();
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, true, c_type>
<<<grid, block, 0, stream>>>(
reinterpret_cast<c_type*>(input.data_ptr()),
reinterpret_cast<uint8_t*>(output.data_ptr()),
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
batch_size, num_heads, num_tokens,
stride_bz_input, stride_h_input, stride_seq_input,
stride_bz_output, stride_h_output, stride_seq_output,
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
});
});
}
void scaled_fp4_quant_trans(torch::Tensor const& input,
torch::Tensor const& output,
torch::Tensor const& output_sf,
int tensor_layout) {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
CHECK_CUDA(input);
CHECK_CUDA(output);
CHECK_CUDA(output_sf);
CHECK_LASTDIM_CONTIGUOUS(input);
CHECK_LASTDIM_CONTIGUOUS(output);
CHECK_LASTDIM_CONTIGUOUS(output_sf);
CHECK_DTYPE(output, at::ScalarType::Byte);
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
CHECK_DIMS(input, 4);
CHECK_DIMS(output, 4);
CHECK_DIMS(output_sf, 4);
const int batch_size = input.size(0);
const int head_dim = input.size(3);
const int stride_bz_input = input.stride(0);
const int stride_bz_output = output.stride(0);
const int stride_bz_output_sf = output_sf.stride(0);
int num_tokens, num_heads;
int stride_seq_input;
int stride_d_output, stride_d_output_sf;
int stride_h_input, stride_h_output, stride_h_output_sf;
if (tensor_layout == 0) {
num_tokens = input.size(1);
num_heads = input.size(2);
stride_seq_input = input.stride(1);
stride_d_output = output.stride(1);
stride_d_output_sf = output_sf.stride(1);
stride_h_input = input.stride(2);
stride_h_output = output.stride(2);
stride_h_output_sf = output_sf.stride(2);
CHECK_SHAPE(output, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
CHECK_SHAPE(output_sf, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
} else {
num_tokens = input.size(2);
num_heads = input.size(1);
stride_seq_input = input.stride(2);
stride_d_output = output.stride(2);
stride_d_output_sf = output_sf.stride(2);
stride_h_input = input.stride(1);
stride_h_output = output.stride(1);
stride_h_output_sf = output_sf.stride(1);
CHECK_SHAPE(output, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
CHECK_SHAPE(output_sf, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
}
auto input_dtype = input.scalar_type();
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
scaled_fp4_quant_trans_kernel<HEAD_DIM, BLOCK_SIZE, c_type>
<<<grid, block, 0, stream>>>(
reinterpret_cast<c_type*>(input.data_ptr()),
reinterpret_cast<uint8_t*>(output.data_ptr()),
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
batch_size, num_heads, num_tokens,
stride_bz_input, stride_h_input, stride_seq_input,
stride_bz_output, stride_h_output, stride_d_output,
stride_bz_output_sf, stride_h_output_sf, stride_d_output_sf);
});
});
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("scaled_fp4_quant", &scaled_fp4_quant);
m.def("scaled_fp4_quant_permute", &scaled_fp4_quant_permute);
m.def("scaled_fp4_quant_trans", &scaled_fp4_quant_trans);
}
-14
View File
@@ -1,14 +0,0 @@
"""Make local benchmark scripts runnable from common repo entrypoints."""
from pathlib import Path
import sys
BENCHMARKS_DIR = Path(__file__).resolve().parent
KERNEL_ROOT = BENCHMARKS_DIR.parent
REPO_ROOT = KERNEL_ROOT.parent
for path in (KERNEL_ROOT, REPO_ROOT):
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
@@ -1,287 +0,0 @@
import sys
import traceback
import _bootstrap # noqa: F401
import torch
from attn_qat_infer.api import (
blockscaled_fp4_attn,
preprocess_qkv,
scale_and_quant_fp4,
scale_and_quant_fp4_permute,
scale_and_quant_fp4_transpose,
)
from attn_qat_infer.quantization.bench.bench_utils import bench
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def benchmark_blockscaled_fp4_attn(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
per_block_mean=True, single_level_p_quant=False,
num_warmups=100, num_tests=1000):
"""
Benchmark blockscaled_fp4_attn function (excluding quantization overhead).
This benchmarks ONLY the core FP4 attention kernel, with pre-quantized inputs.
The quantization step is performed once before benchmarking.
Args:
batch_size: Batch size
num_heads: Number of attention heads
seq_len: Sequence length (same for Q, K, V)
head_dim: Head dimension
is_causal: Whether to use causal masking
dtype: Data type (torch.bfloat16 or torch.float16)
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization for P matrix
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
Returns:
dict with performance metrics
"""
device = 'cuda'
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
# Create input tensors
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
# Pre-process and quantize inputs (done once, not included in benchmark)
is_bf16 = dtype == torch.bfloat16
KL = k.size(2)
q_processed, k_processed, v_processed, delta_s = preprocess_qkv(q, k, v, per_block_mean)
qlist = scale_and_quant_fp4(q_processed)
klist = scale_and_quant_fp4_permute(k_processed)
vlist = scale_and_quant_fp4_transpose(v_processed)
# Synchronize to ensure quantization is complete
torch.cuda.synchronize()
# Create closure for benchmarking (only the attention kernel)
def run_attention():
return blockscaled_fp4_attn(
qlist, klist, vlist,
delta_s,
KL,
is_causal=is_causal,
per_block_mean=per_block_mean,
is_bf16=is_bf16,
single_level_p_quant=single_level_p_quant
)
# Benchmark using the bench utility (handles warmup and timing)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
avg_time_s = avg_time_ms / 1000.0
# Calculate FLOPs (use processed sequence length after padding)
processed_seq_len = q_processed.size(2)
total_flops = calculate_attention_flops(
batch_size, num_heads, processed_seq_len, processed_seq_len, head_dim, is_causal
)
# Calculate TFLOPs
tflops = total_flops / (avg_time_s * 1e12)
# Calculate throughput (tokens/sec) - use original seq_len for meaningful metric
tokens_per_second = (batch_size * seq_len) / avg_time_s
return {
'batch_size': batch_size,
'num_heads': num_heads,
'seq_len': seq_len,
'processed_seq_len': processed_seq_len,
'head_dim': head_dim,
'is_causal': is_causal,
'dtype': str(dtype),
'per_block_mean': per_block_mean,
'single_level_p_quant': single_level_p_quant,
'avg_time_ms': avg_time_ms,
'avg_time_s': avg_time_s,
'total_flops': total_flops,
'tflops': tflops,
'tokens_per_second': tokens_per_second,
}
def print_results(results):
"""Print benchmark results in a formatted table."""
print("\n" + "="*100)
print("blockscaled_fp4_attn Benchmark Results (Kernel Only, No Quantization)")
print("="*100)
print(f"Configuration:")
print(f" Batch Size: {results['batch_size']}")
print(f" Num Heads: {results['num_heads']}")
print(f" Sequence Length: {results['seq_len']}")
print(f" Processed Seq Length: {results['processed_seq_len']} (after padding)")
print(f" Head Dimension: {results['head_dim']}")
print(f" Causal: {results['is_causal']}")
print(f" Data Type: {results['dtype']}")
print(f" Per Block Mean: {results['per_block_mean']}")
print(f" Single Level P Quant: {results['single_level_p_quant']}")
print(f"\nPerformance:")
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
print("="*100 + "\n")
sys.stdout.flush()
def run_benchmark_suite():
"""Run a comprehensive benchmark suite with various configurations."""
print("Starting blockscaled_fp4_attn Benchmark Suite (Kernel Only)...")
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}\n")
print("Note: This benchmark measures only the FP4 attention kernel,")
print(" excluding quantization overhead.\n")
sys.stdout.flush()
# Default configurations to test
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
configs = [
(1, 16, 512, 64, False, torch.bfloat16),
(1, 16, 1024, 64, False, torch.bfloat16),
(1, 16, 2048, 64, False, torch.bfloat16),
(1, 16, 4096, 64, False, torch.bfloat16),
(1, 16, 8192, 64, False, torch.bfloat16),
(1, 16, 16384, 64, False, torch.bfloat16),
(1, 16, 512, 128, False, torch.bfloat16),
(1, 16, 1024, 128, False, torch.bfloat16),
(1, 16, 2048, 128, False, torch.bfloat16),
(1, 16, 4096, 128, False, torch.bfloat16),
(1, 16, 8192, 128, False, torch.bfloat16),
(1, 16, 16384, 128, False, torch.bfloat16),
(1, 32, 512, 64, False, torch.bfloat16),
(1, 32, 1024, 64, False, torch.bfloat16),
(1, 32, 2048, 64, False, torch.bfloat16),
(1, 32, 4096, 64, False, torch.bfloat16),
(1, 32, 8192, 64, False, torch.bfloat16),
(1, 32, 16384, 64, False, torch.bfloat16),
(1, 32, 512, 128, False, torch.bfloat16),
(1, 32, 1024, 128, False, torch.bfloat16),
(1, 32, 2048, 128, False, torch.bfloat16),
(1, 32, 4096, 128, False, torch.bfloat16),
(1, 32, 8192, 128, False, torch.bfloat16),
(1, 32, 16384, 128, False, torch.bfloat16),
]
all_results = []
for config in configs:
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
f"Causal={is_causal}, dtype={dtype}...")
sys.stdout.flush()
try:
results = benchmark_blockscaled_fp4_attn(
batch_size=batch_size,
num_heads=num_heads,
seq_len=seq_len,
head_dim=head_dim,
is_causal=is_causal,
dtype=dtype,
num_warmups=10,
num_tests=50
)
print_results(results)
all_results.append(results)
except Exception as e:
print(f"Error benchmarking configuration {config}:")
print(f" Exception: {e}")
traceback.print_exc()
sys.stdout.flush()
continue
# Print summary table
print("\n" + "="*120)
print("Summary Table (blockscaled_fp4_attn Kernel Only)")
print("="*120)
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
print("-"*120)
for r in all_results:
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
f"{r['tokens_per_second']:<15,.0f}")
print("="*120 + "\n")
sys.stdout.flush()
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description='Benchmark blockscaled_fp4_attn kernel (excluding quantization)')
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
help='Data type')
parser.add_argument('--per-block-mean', action='store_true', default=True,
help='Use per-block mean for Q smoothing (default: True)')
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
help='Disable per-block mean for Q smoothing')
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
help='Use single-level P quantization (default: False)')
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
help='Use two-level P quantization')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
args = parser.parse_args()
dtype_map = {
'bfloat16': torch.bfloat16,
'float16': torch.float16
}
if args.suite:
run_benchmark_suite()
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
results = benchmark_blockscaled_fp4_attn(
batch_size=args.batch_size,
num_heads=args.num_heads,
seq_len=args.seq_len,
head_dim=args.head_dim,
is_causal=args.causal,
dtype=dtype_map[args.dtype],
per_block_mean=args.per_block_mean,
single_level_p_quant=args.single_level_p_quant,
num_warmups=args.num_warmups,
num_tests=args.num_tests
)
print_results(results)
else:
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
print(
"Example: python benchmarks/benchmark_blockscaled_fp4_attn.py "
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
)
print("\nNote: This benchmark measures only the FP4 attention kernel,")
print(" excluding quantization overhead.")
sys.stdout.flush()
run_benchmark_suite()
@@ -1,380 +0,0 @@
import argparse
import sys
from typing import Dict, List, Optional
import _bootstrap # noqa: F401
import matplotlib
matplotlib.use('Agg') # Use non-interactive backend for server environments
import matplotlib.pyplot as plt
import numpy as np
import torch
from flash_attn import flash_attn_func
from attn_qat_infer.quantization.bench.bench_utils import bench
# Import SageAttn components for direct control
from attn_qat_infer.api import (
preprocess_qkv,
scale_and_quant_fp4,
scale_and_quant_fp4_permute,
scale_and_quant_fp4_transpose,
blockscaled_fp4_attn
)
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def sageattn_blackwell_configurable(q, k, v, is_causal=False, per_block_mean=True,
single_level_p_quant=True,
enable_smoothing_q=False, enable_smoothing_k=False):
"""
Configurable SageAttention3 Blackwell kernel with explicit smoothing control.
Args:
q: Query tensor [B, H, L, D]
k: Key tensor [B, H, L, D]
v: Value tensor [B, H, L, D]
is_causal: Whether to use causal masking
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization for P matrix
enable_smoothing_q: Enable Q smoothing
enable_smoothing_k: Enable K smoothing
Returns:
Output tensor [B, H, L, D]
"""
QL = q.size(2)
KL = k.size(2)
is_bf16 = q.dtype == torch.bfloat16
# Preprocess with explicit smoothing control
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean, enable_smoothing_q, enable_smoothing_k)
qlist_from_cuda = scale_and_quant_fp4(q)
klist_from_cuda = scale_and_quant_fp4_permute(k)
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
o_fp4 = blockscaled_fp4_attn(
qlist_from_cuda,
klist_from_cuda,
vlist_from_cuda,
delta_s,
KL,
is_causal,
per_block_mean,
is_bf16,
single_level_p_quant
)[0][:, :, :QL, :].contiguous()
return o_fp4
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
num_warmups=10, num_tests=50):
"""Benchmark FlashAttention2."""
device = 'cuda'
# FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
q = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
k = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
v = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
def run_attention():
return flash_attn_func(q, k, v, causal=is_causal)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
return avg_time_ms
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
per_block_mean=True, single_level_p_quant=False,
enable_smoothing_q=True, enable_smoothing_k=True,
num_warmups=10, num_tests=50):
"""Benchmark SageAttention3 with configurable smoothing."""
device = 'cuda'
# SageAttn expects (batch, num_heads, seq_len, head_dim)
q = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
k = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
v = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
def run_attention():
return sageattn_blackwell_configurable(
q, k, v,
is_causal=is_causal,
per_block_mean=per_block_mean,
single_level_p_quant=single_level_p_quant,
enable_smoothing_q=enable_smoothing_q,
enable_smoothing_k=enable_smoothing_k
)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
return avg_time_ms
def time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal=False):
"""Convert time to TFLOPs (Tera FLOPs per Second)."""
total_flops = calculate_attention_flops(batch_size, num_heads, seq_len, seq_len, head_dim, is_causal)
time_s = time_ms / 1000.0
tflops = total_flops / (time_s * 1e12)
return tflops
def run_benchmark_suite(head_dim=64, is_causal=False, num_heads=12, batch_size=1,
num_warmups=10, num_tests=50,
seq_lens=None, output_file="benchmark_attention.png"):
"""
Run comprehensive benchmark suite and generate plot.
Args:
head_dim: Head dimension (64 or 128)
is_causal: Whether to use causal attention
num_heads: Number of attention heads
batch_size: Batch size
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
seq_lens: List of sequence lengths to test
output_file: Output plot filename
"""
if seq_lens is None:
seq_lens = [1024, 2048, 4096, 8192, 16384, 32768]
device_name = torch.cuda.get_device_name(0)
# Extract short name (e.g., "RTX5090" from full name)
short_name = device_name.split()[-1] if 'RTX' in device_name or 'A100' in device_name else device_name[:20]
print(f"Starting Combined Attention Benchmark Suite...")
print(f"CUDA Device: {device_name}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}")
print(f"Head Dim: {head_dim}, Causal: {is_causal}, Num Heads: {num_heads}, Batch Size: {batch_size}")
print("="*80)
sys.stdout.flush()
# Results storage: {method_name: {seq_len: tflops}}
results: Dict[str, Dict[int, Optional[float]]] = {
'FlashAttn': {},
'SageAttn3': {},
'FP4': {},
}
dtype = torch.bfloat16
for seq_len in seq_lens:
print(f"\n--- Sequence Length: {seq_len} ---")
sys.stdout.flush()
# FlashAttention2
print(f" Benchmarking FlashAttn2...", end=" ")
sys.stdout.flush()
try:
time_ms = benchmark_flashattn2(
batch_size, num_heads, seq_len, head_dim,
is_causal=is_causal, dtype=dtype,
num_warmups=num_warmups, num_tests=num_tests
)
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
results['FlashAttn'][seq_len] = tflops
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
except Exception as e:
print(f"OOM or Error: {e}")
results['FlashAttn'][seq_len] = None
sys.stdout.flush()
# SageAttn3 (with smoothing: single_level_p_quant=False, enable_smoothing_q=True, enable_smoothing_k=True)
print(f" Benchmarking SageAttn3 (smoothing ON)...", end=" ")
sys.stdout.flush()
try:
time_ms = benchmark_sageattn3(
batch_size, num_heads, seq_len, head_dim,
is_causal=is_causal, dtype=dtype,
per_block_mean=True,
single_level_p_quant=False,
enable_smoothing_q=True,
enable_smoothing_k=True,
num_warmups=num_warmups, num_tests=num_tests
)
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
results['SageAttn3'][seq_len] = tflops
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
except Exception as e:
print(f"OOM or Error: {e}")
results['SageAttn3'][seq_len] = None
sys.stdout.flush()
# FP4 (no smoothing: single_level_p_quant=True, enable_smoothing_q=False, enable_smoothing_k=False)
print(f" Benchmarking FP4 (smoothing OFF)...", end=" ")
sys.stdout.flush()
try:
time_ms = benchmark_sageattn3(
batch_size, num_heads, seq_len, head_dim,
is_causal=is_causal, dtype=dtype,
per_block_mean=True,
single_level_p_quant=True,
enable_smoothing_q=False,
enable_smoothing_k=False,
num_warmups=num_warmups, num_tests=num_tests
)
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
results['FP4'][seq_len] = tflops
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
except Exception as e:
print(f"OOM or Error: {e}")
results['FP4'][seq_len] = None
sys.stdout.flush()
# Print summary table
print("\n" + "="*100)
print("Summary Table (TFLOPs)")
print("="*100)
header = f"{'SeqLen':<10}"
for method in results.keys():
header += f"{method:<15}"
print(header)
print("-"*100)
for seq_len in seq_lens:
row = f"{seq_len:<10}"
for method in results.keys():
val = results[method].get(seq_len)
if val is not None:
row += f"{val:<15.0f}"
else:
row += f"{'OOM':<15}"
print(row)
print("="*100)
sys.stdout.flush()
# Generate plot
generate_plot(results, seq_lens, head_dim, is_causal, short_name, output_file)
return results
def generate_plot(results: Dict[str, Dict[int, Optional[float]]],
seq_lens: List[int],
head_dim: int,
is_causal: bool,
device_name: str,
output_file: str):
"""Generate bar plot comparing attention implementations."""
# Prepare data
methods = list(results.keys())
x_labels = [f"{sl//1024}K" for sl in seq_lens]
# Colors for each method (red, blue, green scheme)
colors = {
'FlashAttn': '#1E90FF', # Blue (Dodger Blue)
'SageAttn3': '#228B22', # Green (Forest Green)
'FP4': '#DC143C', # Red (Crimson)
}
# Number of methods and positions
n_methods = len(methods)
n_positions = len(seq_lens)
# Bar width and positions
bar_width = 0.25
x = np.arange(n_positions)
# Create figure
fig, ax = plt.subplots(figsize=(12, 6))
# Plot bars for each method
for i, method in enumerate(methods):
values = []
for seq_len in seq_lens:
val = results[method].get(seq_len)
values.append(val if val is not None else 0)
offset = (i - n_methods/2 + 0.5) * bar_width
bars = ax.bar(x + offset, values, bar_width,
label=method, color=colors.get(method, f'C{i}'),
edgecolor='black', linewidth=0.5)
# Add value labels on top of bars
for bar, val, seq_len in zip(bars, values, seq_lens):
if results[method].get(seq_len) is None:
label = 'OOM'
else:
label = f'{int(val)}'
height = bar.get_height()
ax.annotate(label,
xy=(bar.get_x() + bar.get_width() / 2, height),
xytext=(0, 3), # 3 points vertical offset
textcoords="offset points",
ha='center', va='bottom',
fontsize=8, rotation=0)
# Customize plot
ax.set_xlabel('Sequence Length', fontsize=12, fontweight='bold')
ax.set_ylabel('Speed (TFLOPs)', fontsize=12, fontweight='bold')
ax.set_title(f'{device_name}, (Head dim = {head_dim}, causal = {is_causal})', fontsize=14, fontweight='bold')
ax.set_xticks(x)
ax.set_xticklabels(x_labels)
legend = ax.legend(loc='upper left', ncol=len(methods), fontsize=10)
# Make legend text bold
for text in legend.get_texts():
text.set_fontweight('bold')
# Set y-axis to start from 0
ax.set_ylim(bottom=0)
# Add grid for readability
ax.yaxis.grid(True, linestyle='--', alpha=0.7)
ax.set_axisbelow(True)
# Tight layout
plt.tight_layout()
# Save plot
plt.savefig(output_file, dpi=150, bbox_inches='tight')
print(f"\nPlot saved to: {output_file}")
# Also save as PDF for high quality
pdf_file = output_file.rsplit('.', 1)[0] + '.pdf'
plt.savefig(pdf_file, bbox_inches='tight')
print(f"PDF saved to: {pdf_file}")
plt.close()
if __name__ == "__main__":
parser = argparse.ArgumentParser(description='Combined Attention Benchmark (FlashAttn2 vs SageAttn3 vs FP4)')
parser.add_argument('--batch-size', type=int, default=1, help='Batch size')
parser.add_argument('--num-heads', type=int, default=16, help='Number of attention heads')
parser.add_argument('--head-dim', type=int, default=64, choices=[64, 128], help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--seq-lens', type=int, nargs='+',
default=[1024, 2048, 4096, 8192, 16384, 32768],
help='Sequence lengths to benchmark')
parser.add_argument('--output', type=str, default='benchmark_attention.png',
help='Output plot filename')
args = parser.parse_args()
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
run_benchmark_suite(
head_dim=args.head_dim,
is_causal=args.causal,
num_heads=args.num_heads,
batch_size=args.batch_size,
num_warmups=args.num_warmups,
num_tests=args.num_tests,
seq_lens=args.seq_lens,
output_file=args.output
)
@@ -1,234 +0,0 @@
import sys
import traceback
import _bootstrap # noqa: F401
import torch
from flash_attn import flash_attn_func
from attn_qat_infer.quantization.bench.bench_utils import bench
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
num_warmups=100, num_tests=1000):
"""
Benchmark FlashAttention2 and return performance metrics.
Args:
batch_size: Batch size
num_heads: Number of attention heads
seq_len: Sequence length (same for Q, K, V)
head_dim: Head dimension
is_causal: Whether to use causal masking
dtype: Data type (torch.bfloat16 or torch.float16)
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
Returns:
dict with performance metrics
"""
device = 'cuda'
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
# Create input tensors - FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
q = torch.randn(batch_size, seq_len, num_heads, head_dim,
device=device, dtype=dtype)
k = torch.randn(batch_size, seq_len, num_heads, head_dim,
device=device, dtype=dtype)
v = torch.randn(batch_size, seq_len, num_heads, head_dim,
device=device, dtype=dtype)
# Create closure for benchmarking
def run_attention():
return flash_attn_func(
q, k, v,
causal=is_causal,
)
# Benchmark using the bench utility (handles warmup and timing)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
avg_time_s = avg_time_ms / 1000.0
# Calculate FLOPs
total_flops = calculate_attention_flops(
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
)
# Calculate TFLOPs
tflops = total_flops / (avg_time_s * 1e12)
# Calculate throughput (tokens/sec)
tokens_per_second = (batch_size * seq_len) / avg_time_s
return {
'batch_size': batch_size,
'num_heads': num_heads,
'seq_len': seq_len,
'head_dim': head_dim,
'is_causal': is_causal,
'dtype': str(dtype),
'avg_time_ms': avg_time_ms,
'avg_time_s': avg_time_s,
'total_flops': total_flops,
'tflops': tflops,
'tokens_per_second': tokens_per_second,
}
def print_results(results):
"""Print benchmark results in a formatted table."""
print("\n" + "="*100)
print("FlashAttention2 Benchmark Results")
print("="*100)
print(f"Configuration:")
print(f" Batch Size: {results['batch_size']}")
print(f" Num Heads: {results['num_heads']}")
print(f" Sequence Length: {results['seq_len']}")
print(f" Head Dimension: {results['head_dim']}")
print(f" Causal: {results['is_causal']}")
print(f" Data Type: {results['dtype']}")
print(f"\nPerformance:")
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
print("="*100 + "\n")
sys.stdout.flush()
def run_benchmark_suite():
"""Run a comprehensive benchmark suite with various configurations."""
print("Starting FlashAttention2 Benchmark Suite...")
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}\n")
sys.stdout.flush()
# Default configurations to test
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
configs = [
(1, 16, 1024, 64, False, torch.bfloat16),
(1, 16, 2048, 64, False, torch.bfloat16),
(1, 16, 4096, 64, False, torch.bfloat16),
(1, 16, 8192, 64, False, torch.bfloat16),
(1, 16, 16384, 64, False, torch.bfloat16),
(1, 16, 1024, 128, False, torch.bfloat16),
(1, 16, 2048, 128, False, torch.bfloat16),
(1, 16, 4096, 128, False, torch.bfloat16),
(1, 16, 8192, 128, False, torch.bfloat16),
(1, 16, 16384, 128, False, torch.bfloat16),
(1, 32, 1024, 64, False, torch.bfloat16),
(1, 32, 2048, 64, False, torch.bfloat16),
(1, 32, 4096, 64, False, torch.bfloat16),
(1, 32, 8192, 64, False, torch.bfloat16),
(1, 32, 16384, 64, False, torch.bfloat16),
(1, 32, 1024, 128, False, torch.bfloat16),
(1, 32, 2048, 128, False, torch.bfloat16),
(1, 32, 4096, 128, False, torch.bfloat16),
(1, 32, 8192, 128, False, torch.bfloat16),
(1, 32, 16384, 128, False, torch.bfloat16),
]
all_results = []
for config in configs:
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
f"Causal={is_causal}, dtype={dtype}...")
sys.stdout.flush()
try:
results = benchmark_flashattn2(
batch_size=batch_size,
num_heads=num_heads,
seq_len=seq_len,
head_dim=head_dim,
is_causal=is_causal,
dtype=dtype,
num_warmups=10,
num_tests=50
)
print_results(results)
all_results.append(results)
except Exception as e:
print(f"Error benchmarking configuration {config}:")
print(f" Exception: {e}")
traceback.print_exc()
sys.stdout.flush()
continue
# Print summary table
print("\n" + "="*120)
print("Summary Table")
print("="*120)
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
print("-"*120)
for r in all_results:
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
f"{r['tokens_per_second']:<15,.0f}")
print("="*120 + "\n")
sys.stdout.flush()
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description='Benchmark FlashAttention2 in TFLOPs')
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
help='Data type')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
args = parser.parse_args()
dtype_map = {
'bfloat16': torch.bfloat16,
'float16': torch.float16
}
if args.suite:
run_benchmark_suite()
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
results = benchmark_flashattn2(
batch_size=args.batch_size,
num_heads=args.num_heads,
seq_len=args.seq_len,
head_dim=args.head_dim,
is_causal=args.causal,
dtype=dtype_map[args.dtype],
num_warmups=args.num_warmups,
num_tests=args.num_tests
)
print_results(results)
else:
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
print(
"Example: python benchmarks/benchmark_flashattn2.py "
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
)
sys.stdout.flush()
run_benchmark_suite()
@@ -1,253 +0,0 @@
import sys
import traceback
import _bootstrap # noqa: F401
import torch
from attn_qat_infer.api import sageattn_blackwell
from attn_qat_infer.quantization.bench.bench_utils import bench
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
per_block_mean=True, single_level_p_quant=False,
num_warmups=100, num_tests=1000):
"""
Benchmark SageAttention3 and return performance metrics.
Args:
batch_size: Batch size
num_heads: Number of attention heads
seq_len: Sequence length (same for Q, K, V)
head_dim: Head dimension
is_causal: Whether to use causal masking
dtype: Data type (torch.bfloat16 or torch.float16)
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization for P matrix
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
Returns:
dict with performance metrics
"""
device = 'cuda'
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
# Create input tensors
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
# Create closure for benchmarking (no extra stream needed - bench handles synchronization)
def run_attention():
return sageattn_blackwell(
q, k, v,
is_causal=is_causal,
per_block_mean=per_block_mean,
single_level_p_quant=single_level_p_quant
)
# Benchmark using the bench utility (handles warmup and timing)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
avg_time_s = avg_time_ms / 1000.0
# Calculate FLOPs
total_flops = calculate_attention_flops(
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
)
# Calculate TFLOPs
tflops = total_flops / (avg_time_s * 1e12)
# Calculate throughput (tokens/sec)
tokens_per_second = (batch_size * seq_len) / avg_time_s
return {
'batch_size': batch_size,
'num_heads': num_heads,
'seq_len': seq_len,
'head_dim': head_dim,
'is_causal': is_causal,
'dtype': str(dtype),
'per_block_mean': per_block_mean,
'single_level_p_quant': single_level_p_quant,
'avg_time_ms': avg_time_ms,
'avg_time_s': avg_time_s,
'total_flops': total_flops,
'tflops': tflops,
'tokens_per_second': tokens_per_second,
}
def print_results(results):
"""Print benchmark results in a formatted table."""
print("\n" + "="*100)
print("SageAttention3 Benchmark Results")
print("="*100)
print(f"Configuration:")
print(f" Batch Size: {results['batch_size']}")
print(f" Num Heads: {results['num_heads']}")
print(f" Sequence Length: {results['seq_len']}")
print(f" Head Dimension: {results['head_dim']}")
print(f" Causal: {results['is_causal']}")
print(f" Data Type: {results['dtype']}")
print(f" Per Block Mean: {results['per_block_mean']}")
print(f" Single Level P Quant: {results['single_level_p_quant']}")
print(f"\nPerformance:")
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
print("="*100 + "\n")
sys.stdout.flush()
def run_benchmark_suite():
"""Run a comprehensive benchmark suite with various configurations."""
print("Starting SageAttention3 Benchmark Suite...")
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}\n")
sys.stdout.flush()
# Default configurations to test
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
configs = [
(1, 16, 1024, 64, False, torch.bfloat16),
(1, 16, 2048, 64, False, torch.bfloat16),
(1, 16, 4096, 64, False, torch.bfloat16),
(1, 16, 8192, 64, False, torch.bfloat16),
(1, 16, 16384, 64, False, torch.bfloat16),
(1, 16, 1024, 128, False, torch.bfloat16),
(1, 16, 2048, 128, False, torch.bfloat16),
(1, 16, 4096, 128, False, torch.bfloat16),
(1, 16, 8192, 128, False, torch.bfloat16),
(1, 16, 16384, 128, False, torch.bfloat16),
(1, 32, 1024, 64, False, torch.bfloat16),
(1, 32, 2048, 64, False, torch.bfloat16),
(1, 32, 4096, 64, False, torch.bfloat16),
(1, 32, 8192, 64, False, torch.bfloat16),
(1, 32, 16384, 64, False, torch.bfloat16),
(1, 32, 1024, 128, False, torch.bfloat16),
(1, 32, 2048, 128, False, torch.bfloat16),
(1, 32, 4096, 128, False, torch.bfloat16),
(1, 32, 8192, 128, False, torch.bfloat16),
(1, 32, 16384, 128, False, torch.bfloat16),
]
all_results = []
for config in configs:
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
f"Causal={is_causal}, dtype={dtype}...")
sys.stdout.flush()
try:
results = benchmark_sageattn3(
batch_size=batch_size,
num_heads=num_heads,
seq_len=seq_len,
head_dim=head_dim,
is_causal=is_causal,
dtype=dtype,
num_warmups=10,
num_tests=50
)
print_results(results)
all_results.append(results)
except Exception as e:
print(f"Error benchmarking configuration {config}:")
print(f" Exception: {e}")
traceback.print_exc()
sys.stdout.flush()
continue
# Print summary table
print("\n" + "="*120)
print("Summary Table")
print("="*120)
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
print("-"*120)
for r in all_results:
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
f"{r['tokens_per_second']:<15,.0f}")
print("="*120 + "\n")
sys.stdout.flush()
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description='Benchmark SageAttention3 in TFLOPs')
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
help='Data type')
parser.add_argument('--per-block-mean', action='store_true', default=True,
help='Use per-block mean for Q smoothing (default: True)')
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
help='Disable per-block mean for Q smoothing')
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
help='Use single-level P quantization (default: True)')
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
help='Use two-level P quantization')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
args = parser.parse_args()
dtype_map = {
'bfloat16': torch.bfloat16,
'float16': torch.float16
}
if args.suite:
run_benchmark_suite()
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
results = benchmark_sageattn3(
batch_size=args.batch_size,
num_heads=args.num_heads,
seq_len=args.seq_len,
head_dim=args.head_dim,
is_causal=args.causal,
dtype=dtype_map[args.dtype],
per_block_mean=args.per_block_mean,
single_level_p_quant=args.single_level_p_quant,
num_warmups=args.num_warmups,
num_tests=args.num_tests
)
print_results(results)
else:
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
print(
"Example: python benchmarks/benchmark_sageattn3.py "
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
)
sys.stdout.flush()
run_benchmark_suite()
+2 -2
View File
@@ -23,7 +23,7 @@ classifiers = [
]
dependencies = [
"torch>=2.5.0",
"triton>=2.0.0; sys_platform == 'linux'",
"triton>=2.0.0",
]
[project.urls]
@@ -32,4 +32,4 @@ dependencies = [
[tool.scikit-build]
cmake.build-type = "Release"
minimum-version = "build-system.requires"
wheel.packages = ["python/fastvideo_kernel", "attn_qat_infer"]
wheel.packages = ["python/fastvideo_kernel"]
@@ -1,5 +0,0 @@
"""Triton kernel entrypoints exposed by ``fastvideo_kernel``."""
from .fused_attention import attention as fused_attention
__all__ = ["fused_attention"]
File diff suppressed because it is too large Load Diff
@@ -1,55 +0,0 @@
"""Compatibility shim for the legacy non-QAT Triton attention import path.
Historically callers imported
``fastvideo_kernel.triton_kernels.fused_attention`` directly. The shared
implementation now lives in ``attn_qat_train.py`` and is parameterized by the
``IS_QAT`` flag. This module preserves the original public API for tests and
downstream users while always dispatching to the non-QAT configuration.
"""
from __future__ import annotations
import torch
from .attn_qat_train import attention as _attention
def attention(
q: torch.Tensor,
k: torch.Tensor,
v: torch.Tensor,
causal: bool,
sm_scale: float,
warp_specialize: bool = True,
) -> torch.Tensor:
"""Run the shared Triton attention kernel in non-QAT mode."""
use_qat_qkv_backward = True
smooth_k = False
is_qat = False
two_level_quant_p = False
fake_quant_p = False
use_high_prec_o = False
smooth_q = False
use_global_sf_p = False
use_global_sf_qkv = False
return _attention(
q,
k,
v,
causal,
sm_scale,
use_qat_qkv_backward,
smooth_k,
warp_specialize,
is_qat,
two_level_quant_p,
fake_quant_p,
use_high_prec_o,
smooth_q,
use_global_sf_p,
use_global_sf_qkv,
)
__all__ = ["attention"]
@@ -1,237 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
# Adapted from https://github.com/triton-lang/triton/blob/main/python/triton_kernels/triton_kernels/numerics_details/mxfp_details/_upcast_from_mxfp.py
# and https://github.com/triton-lang/triton/blob/main/python/triton_kernels/triton_kernels/numerics_details/mxfp_details/_downcast_to_mxfp.py
import triton
import triton.language as tl
from triton.language.target_info import cuda_capability_geq
MXFP_BLOCK_SIZE = tl.constexpr(16)
@triton.jit
def _compute_quant_and_scale(
src_tensor,
valid_src_mask,
mx_tensor_dtype: tl.constexpr = tl.uint8,
use_global_sf=True,
two_level_quant_P=False,
):
BLOCK_SIZE_OUT_DIM: tl.constexpr = src_tensor.shape[0]
BLOCK_SIZE_QUANT_DIM: tl.constexpr = src_tensor.shape[1]
BLOCK_SIZE_QUANT_MX_SCALE: tl.constexpr = src_tensor.shape[1] // MXFP_BLOCK_SIZE
is_fp4: tl.constexpr = mx_tensor_dtype == tl.uint8
tl.static_assert(
is_fp4
or mx_tensor_dtype == tl.float8e4nv
or mx_tensor_dtype == tl.float8e5,
"mx_tensor_dtype must be uint8, float8e4nv, or float8e5",
)
# Explicit cast to fp32 since most ops are not supported on bfloat16. We avoid needless conversions to and from bf16
f32_tensor = src_tensor.to(tl.float32)
abs_tensor = tl.abs(f32_tensor)
abs_tensor = tl.where(valid_src_mask, abs_tensor, -1.0) # Don't consider padding tensors in scale computation
if two_level_quant_P:
# row max from SageAttn3 paper
global_max_val = tl.max(f32_tensor, axis=1, keep_dims=True) # (BLOCK_SIZE_OUT_DIM, 1)
global_max_val = tl.maximum(global_max_val, 1e-8)
s_enc = ((6 * 448) / global_max_val).reshape([BLOCK_SIZE_OUT_DIM, 1, 1])
s_dec = (1 / s_enc)
abs_tensor = tl.reshape(abs_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
if use_global_sf and not two_level_quant_P:
global_max_val = tl.max(abs_tensor)
# Avoid division by zero: if all values are padding (max is 0), use a default scale
global_max_val = tl.maximum(global_max_val, 1e-8)
s_enc = (6 * 448) / global_max_val
s_dec = (1 / s_enc)
elif not two_level_quant_P and not use_global_sf:
s_dec = 1.0
s_enc = 1.0
max_val = tl.max(abs_tensor, axis=2, keep_dims=True) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1) # per block maxima
s_dec_b = max_val / 6 # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
s_dec_b_e4m3 = (s_dec_b * s_enc).to(tl.float8e4nv) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
s_enc_b = 1 / (s_dec_b_e4m3.to(tl.float32) * s_dec) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
f32_tensor = tl.reshape(f32_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
quant_tensor = f32_tensor * s_enc_b
# Reshape the tensors after scaling
quant_tensor = quant_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM])
# Set the invalid portions of the tensor to 0. This will ensure that any padding tensors are 0 in the mx format.
quant_tensor = tl.where(valid_src_mask, quant_tensor, 0.0)
dequant_scale = s_dec_b_e4m3.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE])
if is_fp4 and cuda_capability_geq(10, 0):
# Convert scaled values to two f32 lanes and use PTX cvt to e2m1x2 with two f32 operands.
pairs = tl.reshape(quant_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM // 2, 2])
lo_f, hi_f = tl.split(pairs)
lo_f32 = lo_f.to(tl.float32)
hi_f32 = hi_f.to(tl.float32)
# Inline PTX: cvt.rn.satfinite.e2m1x2.f32 takes two f32 sources and produces one .b8 packed e2m1x2.
out_tensor = tl.inline_asm_elementwise(
"""
{
.reg .b8 r;
cvt.rn.satfinite.e2m1x2.f32 r, $1, $2;
mov.b32 $0, {r, r, r, r};
}
""",
constraints="=r,f,f",
args=[hi_f32, lo_f32],
dtype=tl.uint8,
is_pure=True,
pack=1,
)
elif is_fp4:
quant_tensor = quant_tensor.to(tl.uint32, bitcast=True)
signs = quant_tensor & 0x80000000
exponents = (quant_tensor >> 23) & 0xFF
mantissas_orig = (quant_tensor & 0x7FFFFF)
# For RTNE: 0.25 < x < 0.75 maps to 0.5 (denormal); exactly 0.25 maps to 0.0
E8_BIAS = 127
E2_BIAS = 1
# Move implicit bit 1 at the beginning to mantissa for denormals
is_subnormal = exponents < E8_BIAS
adjusted_exponents = tl.core.sub(E8_BIAS, exponents + 1, sanitize_overflow=False)
mantissas_pre = (0x400000 | (mantissas_orig >> 1))
mantissas = tl.where(is_subnormal, mantissas_pre >> adjusted_exponents, mantissas_orig)
# For normal numbers, we change the bias from 127 to 1, and for subnormals, we keep exponent as 0.
exponents = tl.maximum(exponents, E8_BIAS - E2_BIAS) - (E8_BIAS - E2_BIAS)
# Combine sign, exponent, and mantissa, while saturating
# Round to nearest, ties to even (RTNE): use guard/sticky and LSB to decide increment
m2bits = mantissas >> 21
lsb_keep = (m2bits >> 1) & 0x1
guard = m2bits & 0x1
IS_SRC_FP32: tl.constexpr = src_tensor.dtype == tl.float32
if IS_SRC_FP32:
bit0_dropped = (mantissas_orig & 0x1) != 0
mask = (1 << tl.minimum(adjusted_exponents, 31)) - 1
dropped_post = (mantissas_pre & mask) != 0
sticky = is_subnormal & (bit0_dropped | dropped_post)
sticky |= ((mantissas & 0x1FFFFF) != 0).to(tl.uint32)
else:
sticky = ((mantissas & 0x1FFFFF) != 0).to(tl.uint32)
round_inc = guard & (sticky | lsb_keep)
e2m1_tmp = tl.minimum((((exponents << 2) | m2bits) + round_inc) >> 1, 0x7)
e2m1_value = ((signs >> 28) | e2m1_tmp).to(tl.uint8)
e2m1_value = tl.reshape(e2m1_value, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM // 2, 2])
evens, odds = tl.split(e2m1_value)
out_tensor = evens | (odds << 4)
else:
out_tensor = quant_tensor.to(mx_tensor_dtype)
return out_tensor, dequant_scale, s_dec
@triton.jit
def _compute_dequant(
mx_tensor,
scale,
s_dec,
BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
):
tl.static_assert(BLOCK_SIZE_QUANT_DIM % MXFP_BLOCK_SIZE == 0, f"Block size along quantization block must be a multiple of {MXFP_BLOCK_SIZE=}")
# uint8 signifies two fp4 e2m1 values packed into a single byte
mx_tensor_dtype: tl.constexpr = mx_tensor.dtype
tl.static_assert(dst_dtype == tl.float16 or dst_dtype == tl.bfloat16 or dst_dtype == tl.float32)
tl.static_assert(
mx_tensor_dtype == tl.uint8
or ((mx_tensor_dtype == tl.float8e4nv or mx_tensor_dtype == tl.float8e5) or mx_tensor_dtype == dst_dtype),
"mx_tensor_ptr must be uint8 or float8 or dst_dtype")
tl.static_assert(scale.dtype == tl.float8e4nv, "scale must be float8e4nv")
# Determine if we are dealing with fp8 types.
is_fp4: tl.constexpr = mx_tensor_dtype == tl.uint8
BLOCK_SIZE_QUANT_MX_SCALE: tl.constexpr = BLOCK_SIZE_QUANT_DIM // MXFP_BLOCK_SIZE
# Upcast the scale to the destination type.
if dst_dtype == tl.bfloat16:
dst_scale = scale.to(tl.bfloat16)
else:
dst_scale = scale.to(tl.float32)
if dst_dtype == tl.float16:
dst_scale = dst_scale.to(tl.float16)
# Now upcast the tensor.
intermediate_dtype: tl.constexpr = tl.bfloat16 if dst_dtype == tl.float32 else dst_dtype
if cuda_capability_geq(10, 0):
assert is_fp4
packed_u32 = tl.inline_asm_elementwise(
asm="""
{
.reg .b8 in_8;
.reg .f16x2 out;
cvt.u8.u32 in_8, $1;
cvt.rn.f16x2.e2m1x2 out, in_8;
mov.b32 $0, out;
}
""",
constraints="=r,r",
args=[mx_tensor], # tl.uint8 passed in as a 32-bit reg with value in low 8 bits
dtype=tl.uint32,
is_pure=True,
pack=1,
)
lo_u16 = (packed_u32 & 0xFFFF).to(tl.uint16)
hi_u16 = (packed_u32 >> 16).to(tl.uint16)
lo_f16 = lo_u16.to(tl.float16, bitcast=True)
hi_f16 = hi_u16.to(tl.float16, bitcast=True)
if intermediate_dtype == tl.float16:
x0, x1 = lo_f16, hi_f16
else:
x0 = lo_f16.to(intermediate_dtype)
x1 = hi_f16.to(intermediate_dtype)
dst_tensor = tl.interleave(x0, x1)
else:
assert is_fp4
dst_bias: tl.constexpr = 127 if intermediate_dtype == tl.bfloat16 else 15 # exponent bias
dst_0p5: tl.constexpr = 16128 if intermediate_dtype == tl.bfloat16 else 0x3800
dst_m_bits: tl.constexpr = 7 if intermediate_dtype == tl.bfloat16 else 10 # mantissa bits
# e2m1
em0 = mx_tensor & 0x07
em1 = mx_tensor & 0x70
x0 = (em0.to(tl.uint16) << (dst_m_bits - 1)) | ((mx_tensor & 0x08).to(tl.uint16) << 12)
x1 = (em1.to(tl.uint16) << (dst_m_bits - 5)) | ((mx_tensor & 0x80).to(tl.uint16) << 8)
# Three cases:
# 1) x is normal and non-zero: Correct bias
x0 = tl.where((em0 & 0x06) != 0, x0 + ((dst_bias - 1) << dst_m_bits), x0)
x1 = tl.where((em1 & 0x60) != 0, x1 + ((dst_bias - 1) << dst_m_bits), x1)
# 2) x is subnormal (x == 0bs001 where s is the sign): Map to +-0.5 in the dst type
x0 = tl.where(em0 == 0x01, dst_0p5 | (x0 & 0x8000), x0)
x1 = tl.where(em1 == 0x10, dst_0p5 | (x1 & 0x8000), x1)
# 3) x is zero, do nothing
dst_tensor = tl.interleave(x0, x1).to(intermediate_dtype, bitcast=True)
dst_tensor = dst_tensor.to(dst_dtype)
# Reshape for proper broadcasting: the scale was stored with a 16‐sized “inner” grouping.
dst_tensor = dst_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
dst_scale = dst_scale.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1])
scale = scale.reshape(dst_scale.shape)
out_tensor = dst_tensor * dst_scale * s_dec # NVFP4 has the additional global scale factor
if dst_dtype == tl.float32:
max_fin = 3.4028234663852886e+38
elif dst_dtype == tl.bfloat16:
max_fin = 3.3895313892515355e+38
else:
tl.static_assert(dst_dtype == tl.float16)
max_fin = 65504
out_tensor = tl.clamp(out_tensor, min=-max_fin, max=max_fin)
out_tensor = out_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM])
out_tensor = out_tensor.to(dst_dtype)
return out_tensor
@@ -1,80 +0,0 @@
import triton
import triton.language as tl
from .nvfp4_utils import _compute_quant_and_scale, _compute_dequant
@triton.jit
def fake_quantize(src_tensor, valid_src_mask, BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
mx_tensor_dtype: tl.constexpr = tl.uint8,
use_global_sf: tl.constexpr = True,
two_level_quant_P: tl.constexpr = False):
high_prec_src_tensor = src_tensor
src_tensor, src_scale, src_s_dec = _compute_quant_and_scale(src_tensor=src_tensor,
valid_src_mask=valid_src_mask,
mx_tensor_dtype=mx_tensor_dtype,
use_global_sf=use_global_sf,
two_level_quant_P=two_level_quant_P)
src_tensor = _compute_dequant(mx_tensor=src_tensor,
scale=src_scale,
s_dec=src_s_dec,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=dst_dtype)
return src_tensor, high_prec_src_tensor.to(src_tensor.dtype)
@triton.jit
def fake_quantize_q(Q, fake_Q, stride_z_q, stride_h_q,
stride_tok_q, stride_d_q,
fake_stride_z_q, fake_stride_h_q,
fake_stride_tok_q, fake_stride_d_q,
H, N_CTX_Q,
BLOCK_M: tl.constexpr,
HEAD_DIM: tl.constexpr,
use_global_sf: tl.constexpr = True):
bhid = tl.program_id(1)
adj_q = (stride_h_q * (bhid % H) + stride_z_q * (bhid // H))
fake_adj_q = (fake_stride_h_q * (bhid % H) + fake_stride_z_q * (bhid // H))
Q += adj_q
fake_Q += fake_adj_q
pid = tl.program_id(0)
start_m = pid * BLOCK_M
offs_m = start_m + tl.arange(0, BLOCK_M)
offs_k = tl.arange(0, HEAD_DIM)
q_valid = offs_m < N_CTX_Q
q = tl.load(Q + offs_m[:, None] * stride_tok_q + offs_k[None, :] * stride_d_q, mask=q_valid[:, None], other=0.0)
q, _ = fake_quantize(src_tensor=q, valid_src_mask=q_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_M, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=q.dtype, use_global_sf=use_global_sf)
tl.store(fake_Q + offs_m[:, None] * fake_stride_tok_q + offs_k[None, :] * fake_stride_d_q, q, mask=q_valid[:, None])
@triton.jit
def fake_quantize_kv(K, V, fake_K, fake_V, stride_z_kv, stride_h_kv,
stride_tok_kv, stride_d_kv,
fake_stride_z_kv, fake_stride_h_kv,
fake_stride_tok_kv, fake_stride_d_kv,
H, N_CTX_KV,
BLOCK_N: tl.constexpr,
HEAD_DIM: tl.constexpr,
use_global_sf: tl.constexpr = True):
bhid = tl.program_id(1)
adj_kv = (stride_h_kv * (bhid % H) + stride_z_kv * (bhid // H))
fake_adj_kv = (fake_stride_h_kv * (bhid % H) + fake_stride_z_kv * (bhid // H))
K += adj_kv
V += adj_kv
fake_K += fake_adj_kv
fake_V += fake_adj_kv
pid = tl.program_id(0)
start_n = pid * BLOCK_N
offs_n = start_n + tl.arange(0, BLOCK_N)
offs_k = tl.arange(0, HEAD_DIM)
kv_valid = offs_n < N_CTX_KV
k_block = tl.load(K + offs_n[:, None] * stride_tok_kv + offs_k[None, :] * stride_d_kv, mask=kv_valid[:, None], other=0.0)
v_block = tl.load(V + offs_n[:, None] * stride_tok_kv + offs_k[None, :] * stride_d_kv, mask=kv_valid[:, None], other=0.0)
k, _ = fake_quantize(src_tensor=k_block, valid_src_mask=kv_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_N, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=k_block.dtype, use_global_sf=use_global_sf)
v, _ = fake_quantize(src_tensor=v_block, valid_src_mask=kv_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_N, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=v_block.dtype, use_global_sf=use_global_sf)
tl.store(fake_K + offs_n[:, None] * fake_stride_tok_kv + offs_k[None, :] * fake_stride_d_kv, k, mask=kv_valid[:, None])
tl.store(fake_V + offs_n[:, None] * fake_stride_tok_kv + offs_k[None, :] * fake_stride_d_kv, v, mask=kv_valid[:, None])
-4
View File
@@ -1,4 +0,0 @@
from ._bootstrap import ensure_local_kernel_sources_first
ensure_local_kernel_sources_first()
-56
View File
@@ -1,56 +0,0 @@
"""Test import helpers for preferring the in-tree fastvideo-kernel sources."""
from __future__ import annotations
import importlib
import sys
from pathlib import Path
def _prepend_import_path(path: Path) -> None:
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
def _purge_package(name: str) -> None:
prefix = f"{name}."
for module_name in tuple(sys.modules):
if module_name == name or module_name.startswith(prefix):
sys.modules.pop(module_name, None)
def _module_is_from_checkout(module_file: str | None, checkout_root: Path) -> bool:
if module_file is None:
return False
try:
module_path = Path(module_file).resolve()
except OSError:
return False
return module_path.is_relative_to(checkout_root.resolve())
def ensure_local_kernel_sources_first() -> None:
tests_root = Path(__file__).resolve().parent
kernel_root = tests_root.parent
repo_root = kernel_root.parent
kernel_python_root = kernel_root / "python"
# Keep the in-tree kernel sources ahead of any preinstalled wheel so tests
# exercise the checkout under review.
for path in (repo_root, kernel_root, kernel_python_root):
_prepend_import_path(path)
importlib.invalidate_caches()
loaded_kernel = sys.modules.get("fastvideo_kernel")
if loaded_kernel is None:
return
if not _module_is_from_checkout(
getattr(loaded_kernel, "__file__", None),
kernel_python_root,
):
_purge_package("fastvideo_kernel")
-4
View File
@@ -1,4 +0,0 @@
from ._bootstrap import ensure_local_kernel_sources_first
ensure_local_kernel_sources_first()
File diff suppressed because it is too large Load Diff
-30
View File
@@ -1,30 +0,0 @@
from __future__ import annotations
import sys
import types
from pathlib import Path
from tests._bootstrap import ensure_local_kernel_sources_first
def test_bootstrap_prefers_local_kernel_checkout(monkeypatch) -> None:
tests_root = Path(__file__).resolve().parent
kernel_root = tests_root.parent
repo_root = kernel_root.parent
kernel_python_root = kernel_root / "python"
stale_module = types.ModuleType("fastvideo_kernel")
stale_module.__file__ = (
"/tmp/site-packages/fastvideo_kernel/__init__.py"
)
monkeypatch.setitem(sys.modules, "fastvideo_kernel", stale_module)
monkeypatch.setattr(sys, "path", ["/tmp/site-packages"])
ensure_local_kernel_sources_first()
assert "fastvideo_kernel" not in sys.modules
assert sys.path[:3] == [
str(kernel_python_root),
str(kernel_root),
str(repo_root),
]
-503
View File
@@ -1,503 +0,0 @@
#!/usr/bin/env python3
"""
Precision test to compare numerical differences for fake_quantize triton function:
1. fake_quantize (triton implementation) vs reference implementations
2. Tests various shapes, dtypes, and value ranges
3. Evaluates cosine similarity, max diff, and mean diff between input and output
"""
import torch
import triton
import triton.language as tl
from flashinfer import SfLayout, nvfp4_quantize, e2m1_and_ufp8sf_scale_to_float
from fastvideo_kernel.triton_kernels.nvfp4_utils import (
_compute_quant_and_scale,
_compute_dequant,
)
from typing import Optional
# MXFP_BLOCK_SIZE is 16 - use Python int for runtime checks
MXFP_BLOCK_SIZE = 16
DEVICE = torch.device("cuda")
def cosine_similarity(tensor1, tensor2):
"""
Compute cosine similarity between two tensors.
Args:
tensor1: First tensor
tensor2: Second tensor (same shape as tensor1)
Returns:
Cosine similarity value (scalar)
- Returns 1.0 if both tensors are zero (identical zero vectors)
- Returns 0.0 if only one tensor is zero (orthogonal to non-zero vector)
"""
# Flatten tensors for computation
t1_flat = tensor1.flatten().float()
t2_flat = tensor2.flatten().float()
# Compute cosine similarity: (A · B) / (||A|| * ||B||)
dot_product = torch.dot(t1_flat, t2_flat)
norm1 = torch.norm(t1_flat)
norm2 = torch.norm(t2_flat)
# Handle zero vectors
if norm1 == 0 and norm2 == 0:
# Both are zero vectors - they are identical, so similarity is 1.0
return 1.0
elif norm1 == 0 or norm2 == 0:
# One is zero, one is not - they are orthogonal, so similarity is 0.0
return 0.0
cos_sim = dot_product / (norm1 * norm2)
return cos_sim.item()
@triton.jit
def fake_quantize(src_tensor, valid_src_mask, BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
mx_tensor_dtype: tl.constexpr = tl.uint8):
"""
Fake quantize function - matches API from attn_qat_train.py.
"""
high_prec_src_tensor = src_tensor
src_tensor, src_scale, src_s_dec = _compute_quant_and_scale(
src_tensor=src_tensor,
valid_src_mask=valid_src_mask,
mx_tensor_dtype=mx_tensor_dtype
)
src_tensor = _compute_dequant(
mx_tensor=src_tensor,
scale=src_scale,
s_dec=src_s_dec,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=dst_dtype
)
return src_tensor, high_prec_src_tensor
def get_fake_quant_reference(x: torch.Tensor):
"""
Reference implementation using FlashInfer for comparison.
"""
orig_shape = x.shape
orig_dtype = x.dtype
device = x.device
x = x.view(-1, x.shape[-1])
x_global_sf = (448 * 6) / x.float().abs().nan_to_num().max()
x_fp4, x_scale = nvfp4_quantize(x, x_global_sf, sfLayout=SfLayout.layout_128x4, do_shuffle=False)
x_dequant = e2m1_and_ufp8sf_scale_to_float(x_fp4, x_scale, 1 / x_global_sf)
return x_dequant.view(orig_shape).to(orig_dtype).to(device)
@triton.jit
def fake_quantize_kernel(
src_ptr,
dst_ptr,
BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
mx_tensor_dtype: tl.constexpr,
stride_src_outer,
stride_src_quant,
stride_dst_outer,
stride_dst_quant,
outer_dim,
quant_dim,
):
"""
Kernel wrapper to call fake_quantize on a block of data.
"""
outer_idx = tl.program_id(0)
quant_idx = tl.program_id(1)
# Compute offsets
start_outer = outer_idx * BLOCK_SIZE_OUT_DIM
start_quant = quant_idx * BLOCK_SIZE_QUANT_DIM
# Create offset arrays
offs_outer = tl.arange(0, BLOCK_SIZE_OUT_DIM)[:, None]
offs_quant = tl.arange(0, BLOCK_SIZE_QUANT_DIM)[None, :]
# Create masks for valid elements
mask_outer = (start_outer + offs_outer) < outer_dim
mask_quant = (start_quant + offs_quant) < quant_dim
full_mask = mask_outer & mask_quant
# Load source tensor
src_offsets = (start_outer + offs_outer) * stride_src_outer + (start_quant + offs_quant) * stride_src_quant
src_tensor = tl.load(src_ptr + src_offsets, mask=full_mask, other=0.0)
# Call fake_quantize with valid_src_mask parameter
quantized_tensor, high_prec_tensor = fake_quantize(
src_tensor=src_tensor,
valid_src_mask=full_mask,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=dst_dtype,
mx_tensor_dtype=mx_tensor_dtype
)
# Store result
dst_offsets = (start_outer + offs_outer) * stride_dst_outer + (start_quant + offs_quant) * stride_dst_quant
tl.store(dst_ptr + dst_offsets, quantized_tensor, mask=full_mask)
def triton_fake_quantize(
x: torch.Tensor,
BLOCK_SIZE_OUT_DIM: int = 128,
BLOCK_SIZE_QUANT_DIM: int = 128,
use_fp4: bool = True, # True for fp4 (uint8), False for fp8 (float8e4nv)
dst_dtype: Optional[torch.dtype] = None
) -> torch.Tensor:
"""
Call fake_quantize triton function on a tensor.
Args:
x: Input tensor (2D or can be reshaped to 2D)
BLOCK_SIZE_OUT_DIM: Block size for outer dimension
BLOCK_SIZE_QUANT_DIM: Block size for quantization dimension (must be multiple of 16)
use_fp4: If True, use fp4 (uint8), else use fp8 (float8e4nv)
dst_dtype: Output dtype (defaults to input dtype)
Returns:
Fake quantized tensor
"""
assert x.is_cuda, "Input must be on CUDA"
assert BLOCK_SIZE_QUANT_DIM % 16 == 0, f"BLOCK_SIZE_QUANT_DIM must be multiple of 16"
orig_shape = x.shape
orig_dtype = x.dtype
# Reshape to 2D
x_2d = x.view(-1, x.shape[-1])
outer_dim, quant_dim = x_2d.shape
if dst_dtype is None:
dst_dtype = orig_dtype
# Map torch dtype to triton dtype
dtype_map = {
torch.float32: tl.float32,
torch.float16: tl.float16,
torch.bfloat16: tl.bfloat16,
}
triton_dst_dtype = dtype_map.get(dst_dtype, tl.float16)
# Allocate output
output = torch.empty_like(x_2d, dtype=dst_dtype)
# Launch kernel with appropriate quantization dtype
grid = (
triton.cdiv(outer_dim, BLOCK_SIZE_OUT_DIM),
triton.cdiv(quant_dim, BLOCK_SIZE_QUANT_DIM),
)
if use_fp4:
# Use fp4 (uint8)
fake_quantize_kernel[grid](
src_ptr=x_2d,
dst_ptr=output,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=triton_dst_dtype,
mx_tensor_dtype=tl.uint8, # fp4 uses uint8
stride_src_outer=x_2d.stride(0),
stride_src_quant=x_2d.stride(1),
stride_dst_outer=output.stride(0),
stride_dst_quant=output.stride(1),
outer_dim=outer_dim,
quant_dim=quant_dim,
)
else:
# Use fp8 (float8e4nv)
fake_quantize_kernel[grid](
src_ptr=x_2d,
dst_ptr=output,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=triton_dst_dtype,
mx_tensor_dtype=tl.float8e4nv, # fp8 uses float8e4nv
stride_src_outer=x_2d.stride(0),
stride_src_quant=x_2d.stride(1),
stride_dst_outer=output.stride(0),
stride_dst_quant=output.stride(1),
outer_dim=outer_dim,
quant_dim=quant_dim,
)
return output.view(orig_shape).to(dst_dtype)
def test_fake_quantize_basic():
"""Test basic functionality of fake_quantize."""
torch.manual_seed(42)
# Test parameters
shape = (128, 128)
dtype = torch.bfloat16
# Create input tensor
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
# Check that outputs have correct shape
assert x_fq.shape == x.shape
assert x_fq.dtype == x.dtype
# Check that outputs are finite
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Input vs Output - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print("✓ Basic test passed.")
def test_fake_quantize_different_shapes():
"""Test fake_quantize with different input shapes."""
torch.manual_seed(42)
test_configs = [
(128, 64), # Small
(256, 128), # Medium
(512, 256), # Large
(128, 256), # Rectangular
(256, 128), # Rectangular (reversed)
]
dtype = torch.bfloat16
for outer_dim, quant_dim in test_configs:
# Create input tensor
x = torch.randn((outer_dim, quant_dim), dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Shape ({outer_dim}, {quant_dim}) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Shape test passed for ({outer_dim}, {quant_dim})")
def test_fake_quantize_different_dtypes():
"""Test fake_quantize with different input dtypes."""
torch.manual_seed(42)
shape = (128, 128)
dtypes = [torch.float32, torch.float16, torch.bfloat16]
for dtype in dtypes:
# Create input tensor
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert x_fq.dtype == x.dtype
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Dtype {dtype} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Dtype test passed for {dtype}")
def test_fake_quantize_3d_4d_tensors():
"""Test fake_quantize with 3D and 4D tensors (reshaped to 2D internally)."""
torch.manual_seed(42)
dtype = torch.bfloat16
# Test 3D tensor (B, H, D)
print("\nTesting 3D tensor (B, H, D)")
x_3d = torch.randn((2, 8, 128), dtype=dtype, device=DEVICE)
x_fq_3d = triton_fake_quantize(x_3d, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq_3d.shape == x_3d.shape
assert torch.isfinite(x_fq_3d).all()
max_diff = (x_fq_3d - x_3d).abs().max()
mean_diff = (x_fq_3d - x_3d).abs().mean()
cos_sim = cosine_similarity(x_fq_3d, x_3d)
print(f" 3D (2, 8, 128) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
# Test 4D tensor (B, H, L, D)
print("\nTesting 4D tensor (B, H, L, D)")
x_4d = torch.randn((1, 8, 256, 128), dtype=dtype, device=DEVICE)
x_fq_4d = triton_fake_quantize(x_4d, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq_4d.shape == x_4d.shape
assert torch.isfinite(x_fq_4d).all()
max_diff = (x_fq_4d - x_4d).abs().max()
mean_diff = (x_fq_4d - x_4d).abs().mean()
cos_sim = cosine_similarity(x_fq_4d, x_4d)
print(f" 4D (1, 8, 256, 128) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print("✓ 3D/4D tensor test passed.")
def test_fake_quantize_edge_cases():
"""Test edge cases like zeros, ones, extreme values."""
torch.manual_seed(42)
dtype = torch.bfloat16
shape = (128, 128)
test_cases = [
("zeros", torch.zeros(shape, dtype=dtype, device=DEVICE)),
("ones", torch.ones(shape, dtype=dtype, device=DEVICE)),
("negative_ones", -torch.ones(shape, dtype=dtype, device=DEVICE)),
("very_small", torch.randn(shape, dtype=dtype, device=DEVICE) * 1e-6),
("very_large", torch.randn(shape, dtype=dtype, device=DEVICE) * 1e6),
]
for name, x in test_cases:
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" {name} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print("✓ Edge cases test passed.")
def test_fake_quantize_vs_reference():
"""Test triton fake_quantize vs reference implementation."""
torch.manual_seed(42)
dtype = torch.bfloat16
shape = (128, 128)
# Create input tensor
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq_triton = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
# Test reference implementation
x_fq_ref = get_fake_quant_reference(x)
# Compare triton vs reference
max_diff = (x_fq_triton - x_fq_ref).abs().max()
mean_diff = (x_fq_triton - x_fq_ref).abs().mean()
cos_sim = cosine_similarity(x_fq_triton, x_fq_ref)
print(f" Triton vs Reference - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
# Also compare each vs input
max_diff_triton = (x_fq_triton - x).abs().max()
mean_diff_triton = (x_fq_triton - x).abs().mean()
cos_sim_triton = cosine_similarity(x_fq_triton, x)
print(f" Triton vs Input - Max diff: {max_diff_triton.item():.6f}, Mean diff: {mean_diff_triton.item():.6f}, Cosine sim: {cos_sim_triton:.6f}")
max_diff_ref = (x_fq_ref - x).abs().max()
mean_diff_ref = (x_fq_ref - x).abs().mean()
cos_sim_ref = cosine_similarity(x_fq_ref, x)
print(f" Reference vs Input - Max diff: {max_diff_ref.item():.6f}, Mean diff: {mean_diff_ref.item():.6f}, Cosine sim: {cos_sim_ref:.6f}")
print("✓ Reference comparison test passed.")
def test_fake_quantize_attention_shapes():
"""Test fake_quantize with attention-like shapes (similar to test_attn_qat_train.py)."""
torch.manual_seed(42)
test_configs = [
(2, 4, 128, 64), # Z, H, N_CTX, HEAD_DIM - Medium
(1, 8, 256, 128), # Large head dim
(1, 40, 9360, 128), # WAN shape
]
dtype = torch.bfloat16
for Z, H, N_CTX, HEAD_DIM in test_configs:
# Create input tensor in BLHD format (B, L, H, D) = (Z, N_CTX, H, HEAD_DIM)
x = torch.randn((Z, N_CTX, H, HEAD_DIM), dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" (Z={Z}, H={H}, N_CTX={N_CTX}, HEAD_DIM={HEAD_DIM}) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Attention shape test passed for (Z={Z}, H={H}, N_CTX={N_CTX}, HEAD_DIM={HEAD_DIM})")
def test_fake_quantize_non_divisible_blocks():
"""Test fake_quantize with shapes that are not divisible by block sizes."""
torch.manual_seed(42)
dtype = torch.bfloat16
# Test with non-divisible dimensions
test_shapes = [
(100, 100), # Not divisible by 128
(150, 200), # Not divisible by 128
(256, 100), # One dimension divisible, one not
]
for shape in test_shapes:
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Shape {shape} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Non-divisible block test passed for {shape}")
if __name__ == "__main__":
print("Running fake_quantize tests...")
print(f"Device: {DEVICE}")
print()
test_fake_quantize_basic()
test_fake_quantize_different_shapes()
test_fake_quantize_different_dtypes()
test_fake_quantize_3d_4d_tensors()
test_fake_quantize_edge_cases()
test_fake_quantize_vs_reference()
test_fake_quantize_attention_shapes()
test_fake_quantize_non_divisible_blocks()
print()
print("All tests passed! ✓")
-65
View File
@@ -1,65 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
from fastvideo.api.schema import (
CompileConfig,
ComponentConfig,
ContinuationState,
EngineConfig,
GenerationPlan,
GenerationRequest,
GeneratorConfig,
InputConfig,
OffloadConfig,
OutputConfig,
ParallelismConfig,
PipelineSelection,
PlannedStage,
QuantizationConfig,
RequestRuntimeConfig,
RunConfig,
SamplingConfig,
ServeConfig,
ServerConfig,
)
from fastvideo.api.errors import ConfigValidationError
from fastvideo.api.overrides import apply_overrides, parse_cli_overrides
from fastvideo.api.parser import (
config_to_dict,
load_config,
load_raw_config,
load_run_config,
load_serve_config,
parse_config,
)
from fastvideo.api.results import GenerationResult
__all__ = [
"CompileConfig",
"ComponentConfig",
"ContinuationState",
"ConfigValidationError",
"EngineConfig",
"GenerationResult",
"GenerationPlan",
"GenerationRequest",
"GeneratorConfig",
"InputConfig",
"OffloadConfig",
"OutputConfig",
"ParallelismConfig",
"PipelineSelection",
"PlannedStage",
"QuantizationConfig",
"RequestRuntimeConfig",
"RunConfig",
"SamplingConfig",
"ServeConfig",
"ServerConfig",
"apply_overrides",
"config_to_dict",
"load_config",
"load_raw_config",
"load_run_config",
"load_serve_config",
"parse_cli_overrides",
"parse_config",
]
-503
View File
@@ -1,503 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
from __future__ import annotations
from collections.abc import Mapping
from copy import deepcopy
from dataclasses import fields, is_dataclass
from pathlib import Path
from typing import Any
from fastvideo.api.overrides import apply_overrides, parse_cli_overrides
from fastvideo.api.parser import config_to_dict, load_raw_config, parse_config
from fastvideo.api.schema import (
GenerationRequest,
GeneratorConfig,
InputConfig,
OutputConfig,
RequestRuntimeConfig,
SamplingConfig,
)
from fastvideo.configs.sample import SamplingParam
from fastvideo.fastvideo_args import FastVideoArgs
from fastvideo.utils import shallow_asdict
_EXPLICIT_REQUEST_ATTR = "_fastvideo_explicit_request"
_INPUT_FIELD_NAMES = {field.name for field in fields(InputConfig)}
_SAMPLING_FIELD_NAMES = {field.name for field in fields(SamplingConfig)}
_RUNTIME_FIELD_NAMES = {field.name for field in fields(RequestRuntimeConfig)}
_OUTPUT_FIELD_NAMES = {field.name for field in fields(OutputConfig)}
_MISSING = object()
_LEGACY_REQUEST_ALIASES = {
"neg_prompt": "negative_prompt",
}
_REQUEST_PIPELINE_OVERRIDE_FIELDS = frozenset({
"embedded_cfg_scale",
})
def normalize_generator_config(config: GeneratorConfig | Mapping[str, Any], ) -> GeneratorConfig:
if isinstance(config, GeneratorConfig):
return config
return parse_config(GeneratorConfig, config)
def load_generator_config_from_file(
path: str | Path,
overrides: list[str] | Mapping[str, Any] | None = None,
) -> GeneratorConfig:
raw = load_raw_config(path)
normalized_overrides = _normalize_overrides(overrides)
if _looks_like_run_or_serve_config(raw):
if normalized_overrides:
raw = apply_overrides(raw, normalized_overrides)
return parse_config(GeneratorConfig, raw["generator"])
if normalized_overrides:
adjusted = normalized_overrides
if all(key.startswith("generator.") for key in adjusted):
adjusted = {key[len("generator."):]: value for key, value in adjusted.items()}
raw = apply_overrides(raw, adjusted)
return parse_config(GeneratorConfig, raw)
def legacy_from_pretrained_to_config(
model_path: str,
kwargs: Mapping[str, Any],
) -> GeneratorConfig:
raw: dict[str, Any] = {"model_path": model_path}
engine: dict[str, Any] = {}
parallelism: dict[str, Any] = {}
offload: dict[str, Any] = {}
compile_config: dict[str, Any] = {}
pipeline: dict[str, Any] = {}
components: dict[str, Any] = {}
quantization: dict[str, Any] = {}
experimental: dict[str, Any] = {}
for key, value in kwargs.items():
if key == "revision":
raw["revision"] = value
elif key == "trust_remote_code":
raw["trust_remote_code"] = value
elif key == "num_gpus":
engine["num_gpus"] = value
elif key == "distributed_executor_backend":
engine["execution_backend"] = value
elif key in {"tp_size", "sp_size", "hsdp_replicate_dim", "hsdp_shard_dim", "dist_timeout"}:
parallelism[key] = value
elif key == "dit_cpu_offload":
offload["dit"] = value
elif key == "dit_layerwise_offload":
offload["dit_layerwise"] = value
elif key == "text_encoder_cpu_offload":
offload["text_encoder"] = value
elif key == "image_encoder_cpu_offload":
offload["image_encoder"] = value
elif key == "vae_cpu_offload":
offload["vae"] = value
elif key == "pin_cpu_memory":
offload["pin_cpu_memory"] = value
elif key == "enable_torch_compile":
compile_config["enabled"] = value
elif key == "torch_compile_kwargs":
compile_config["kwargs"] = deepcopy(value)
elif key in {"enable_stage_verification", "use_fsdp_inference", "disable_autocast"}:
engine[key] = value
elif key == "override_text_encoder_quant":
quantization["text_encoder_quant"] = value
elif key == "transformer_quant":
quantization["transformer_quant"] = value
elif key == "workload_type":
pipeline["workload_type"] = value
elif key == "lora_path":
components["lora_path"] = value
elif key == "override_pipeline_cls_name":
components["override_pipeline_cls_name"] = value
elif key == "override_transformer_cls_name":
components["override_transformer_cls_name"] = value
elif key == "pipeline_config":
if isinstance(value, str):
components["pipeline_config_path"] = value
else:
experimental[key] = deepcopy(value)
elif key == "override_text_encoder_safetensors":
components["text_encoder_weights"] = value
elif key == "init_weights_from_safetensors":
components["transformer_weights"] = value
elif key == "init_weights_from_safetensors_2":
components["transformer_2_weights"] = value
else:
experimental[key] = deepcopy(value)
if parallelism:
engine["parallelism"] = parallelism
if offload:
engine["offload"] = offload
if compile_config:
engine["compile"] = compile_config
if quantization:
engine["quantization"] = quantization
if engine:
raw["engine"] = engine
if components:
pipeline["components"] = components
if experimental:
pipeline["experimental"] = experimental
if pipeline:
raw["pipeline"] = pipeline
return parse_config(GeneratorConfig, raw)
def generator_config_to_fastvideo_args(config: GeneratorConfig | Mapping[str, Any], ) -> FastVideoArgs:
normalized = normalize_generator_config(config)
unsupported = []
if normalized.pipeline.profile is not None:
unsupported.append("pipeline.profile")
if normalized.pipeline.profile_version is not None:
unsupported.append("pipeline.profile_version")
if normalized.pipeline.components.config_root is not None:
unsupported.append("pipeline.components.config_root")
if normalized.pipeline.components.vae_weights is not None:
unsupported.append("pipeline.components.vae_weights")
if normalized.pipeline.components.upsampler_weights is not None:
unsupported.append("pipeline.components.upsampler_weights")
if unsupported:
joined = ", ".join(unsupported)
raise NotImplementedError(f"VideoGenerator compatibility adapter does not support {joined} yet")
engine = normalized.engine
kwargs: dict[str, Any] = {
"model_path": normalized.model_path,
"revision": normalized.revision,
"trust_remote_code": normalized.trust_remote_code,
"num_gpus": engine.num_gpus,
"distributed_executor_backend": engine.execution_backend,
"tp_size": engine.parallelism.tp_size,
"sp_size": engine.parallelism.sp_size,
"hsdp_replicate_dim": engine.parallelism.hsdp_replicate_dim,
"hsdp_shard_dim": engine.parallelism.hsdp_shard_dim,
"dist_timeout": engine.parallelism.dist_timeout,
"dit_cpu_offload": engine.offload.dit,
"dit_layerwise_offload": engine.offload.dit_layerwise,
"text_encoder_cpu_offload": engine.offload.text_encoder,
"image_encoder_cpu_offload": engine.offload.image_encoder,
"vae_cpu_offload": engine.offload.vae,
"pin_cpu_memory": engine.offload.pin_cpu_memory,
"enable_torch_compile": engine.compile.enabled,
"torch_compile_kwargs": deepcopy(engine.compile.kwargs),
"enable_stage_verification": engine.enable_stage_verification,
"use_fsdp_inference": engine.use_fsdp_inference,
"disable_autocast": engine.disable_autocast,
}
if normalized.pipeline.workload_type is not None:
kwargs["workload_type"] = normalized.pipeline.workload_type
quantization = engine.quantization
if quantization is not None and quantization.text_encoder_quant is not None:
kwargs["override_text_encoder_quant"] = quantization.text_encoder_quant
if quantization is not None and quantization.transformer_quant is not None:
kwargs["transformer_quant"] = quantization.transformer_quant
components = normalized.pipeline.components
if components.pipeline_config_path is not None:
kwargs["pipeline_config"] = components.pipeline_config_path
if components.lora_path is not None:
kwargs["lora_path"] = components.lora_path
if components.override_pipeline_cls_name is not None:
kwargs["override_pipeline_cls_name"] = components.override_pipeline_cls_name
if components.override_transformer_cls_name is not None:
kwargs["override_transformer_cls_name"] = components.override_transformer_cls_name
if components.text_encoder_weights is not None:
kwargs["override_text_encoder_safetensors"] = components.text_encoder_weights
if components.transformer_weights is not None:
kwargs["init_weights_from_safetensors"] = components.transformer_weights
if components.transformer_2_weights is not None:
kwargs["init_weights_from_safetensors_2"] = components.transformer_2_weights
kwargs.update(deepcopy(normalized.pipeline.profile_overrides))
kwargs.update(deepcopy(normalized.pipeline.experimental))
return FastVideoArgs.from_kwargs(**kwargs)
def normalize_generation_request(request: GenerationRequest | Mapping[str, Any], ) -> GenerationRequest:
normalized = (request if isinstance(request, GenerationRequest) else parse_config(GenerationRequest, request))
if not hasattr(normalized, _EXPLICIT_REQUEST_ATTR):
setattr(normalized, _EXPLICIT_REQUEST_ATTR, _serialize_generation_request(normalized))
return normalized
def legacy_generate_call_to_request(
prompt: str | None,
sampling_param: SamplingParam | None,
*,
mouse_cond: Any | None = None,
keyboard_cond: Any | None = None,
grid_sizes: Any | None = None,
legacy_kwargs: Mapping[str, Any] | None = None,
) -> GenerationRequest:
raw = _sampling_param_to_request_raw(sampling_param)
if prompt is not None:
raw["prompt"] = prompt
for key, value in (legacy_kwargs or {}).items():
_apply_request_field(raw, key, value)
if mouse_cond is not None:
raw.setdefault("inputs", {})["mouse_cond"] = mouse_cond
if keyboard_cond is not None:
raw.setdefault("inputs", {})["keyboard_cond"] = keyboard_cond
if grid_sizes is not None:
raw.setdefault("inputs", {})["grid_sizes"] = grid_sizes
normalized = parse_config(GenerationRequest, raw)
setattr(normalized, _EXPLICIT_REQUEST_ATTR, deepcopy(raw))
return normalized
def request_to_sampling_param(
request: GenerationRequest,
*,
model_path: str,
) -> SamplingParam:
if request.plan is not None:
raise NotImplementedError("GenerationRequest.plan is not wired into VideoGenerator yet")
if request.state is not None:
raise NotImplementedError("GenerationRequest.state is not wired into VideoGenerator yet")
sampling_param = SamplingParam.from_pretrained(model_path)
updates = _explicit_request_updates(request)
for key, value in updates.items():
if hasattr(sampling_param, key):
setattr(sampling_param, key, deepcopy(value))
elif key in _REQUEST_PIPELINE_OVERRIDE_FIELDS or _is_supported_as_default_only(key, value):
continue
else:
raise ValueError(f"Request field {key!r} is not supported by sampling params for {model_path}")
sampling_param.__post_init__()
sampling_param.check_sampling_param()
return sampling_param
def expand_request_prompt_batch(request: GenerationRequest, ) -> list[GenerationRequest]:
if not isinstance(request.prompt, list):
return [request]
requests: list[GenerationRequest] = []
for index, prompt in enumerate(request.prompt):
single_request = deepcopy(request)
single_request.prompt = prompt
_fan_out_batched_input_value(request, single_request, "image_path", index)
_fan_out_batched_input_value(request, single_request, "video_path", index)
_fan_out_explicit_request_metadata(request, single_request, index, prompt)
requests.append(single_request)
return requests
def _looks_like_run_or_serve_config(raw: Mapping[str, Any]) -> bool:
return isinstance(raw.get("generator"), Mapping)
def _normalize_overrides(overrides: list[str] | Mapping[str, Any] | None, ) -> dict[str, Any] | None:
if not overrides:
return None
if isinstance(overrides, list):
return parse_cli_overrides(overrides)
return dict(overrides)
def _sampling_param_to_request_raw(sampling_param: SamplingParam | None, ) -> dict[str, Any]:
if sampling_param is None:
return {}
raw: dict[str, Any] = {}
for key, value in shallow_asdict(sampling_param).items():
if key == "prompt":
continue
_apply_request_field(raw, key, deepcopy(value))
return raw
def _apply_request_field(
raw: dict[str, Any],
key: str,
value: Any,
) -> None:
key = _LEGACY_REQUEST_ALIASES.get(key, key)
if key == "negative_prompt":
raw["negative_prompt"] = value
return
if key in _INPUT_FIELD_NAMES:
raw.setdefault("inputs", {})[key] = value
return
if key in _SAMPLING_FIELD_NAMES:
raw.setdefault("sampling", {})[key] = value
return
if key in _RUNTIME_FIELD_NAMES:
raw.setdefault("runtime", {})[key] = value
return
if key in _OUTPUT_FIELD_NAMES:
raw.setdefault("output", {})[key] = value
return
raw.setdefault("extensions", {})[key] = value
def request_to_pipeline_overrides(request: GenerationRequest) -> dict[str, Any]:
overrides: dict[str, Any] = {}
for key, value in _explicit_request_updates(request).items():
if key in _REQUEST_PIPELINE_OVERRIDE_FIELDS:
overrides[key] = deepcopy(value)
return overrides
def _explicit_request_updates(request: GenerationRequest) -> dict[str, Any]:
raw = getattr(request, _EXPLICIT_REQUEST_ATTR, None)
if raw is None:
raw = _serialize_generation_request(request)
return _extract_request_updates(raw)
def _extract_request_updates(raw: Mapping[str, Any]) -> dict[str, Any]:
updates: dict[str, Any] = {}
if "negative_prompt" in raw:
updates["negative_prompt"] = deepcopy(raw["negative_prompt"])
for section_name in ("inputs", "sampling", "runtime", "output"):
section = raw.get(section_name)
if not isinstance(section, Mapping):
continue
for key, value in section.items():
updates[key] = deepcopy(value)
stage_overrides = raw.get("stage_overrides")
if stage_overrides:
updates.update(_flatten_stage_overrides(stage_overrides))
extensions = raw.get("extensions")
if isinstance(extensions, Mapping):
for key, value in extensions.items():
updates[key] = deepcopy(value)
return updates
def _flatten_stage_overrides(stage_overrides: Any) -> dict[str, Any]:
if not isinstance(stage_overrides, Mapping):
raise ValueError("GenerationRequest.stage_overrides must be a mapping")
flattened: dict[str, Any] = {}
for stage_name, overrides in stage_overrides.items():
if not isinstance(overrides, Mapping):
raise ValueError(f"GenerationRequest.stage_overrides.{stage_name} must be a mapping")
for key, value in overrides.items():
if key in flattened and flattened[key] != value:
raise ValueError(f"Conflicting stage override for {key!r} across stages")
flattened[key] = deepcopy(value)
return flattened
def _serialize_generation_request(request: GenerationRequest) -> dict[str, Any]:
return deepcopy(config_to_dict(request))
def _fan_out_batched_input_value(
source_request: GenerationRequest,
target_request: GenerationRequest,
field_name: str,
index: int,
) -> None:
value = getattr(source_request.inputs, field_name)
if not isinstance(value, list):
return
_validate_batched_input_length(source_request.prompt, value, field_name)
setattr(target_request.inputs, field_name, deepcopy(value[index]))
def _fan_out_explicit_request_metadata(
source_request: GenerationRequest,
target_request: GenerationRequest,
index: int,
prompt: str,
) -> None:
raw = getattr(source_request, _EXPLICIT_REQUEST_ATTR, None)
if raw is None:
return
raw = deepcopy(raw)
raw["prompt"] = prompt
inputs = raw.get("inputs")
if isinstance(inputs, dict):
for field_name in ("image_path", "video_path"):
value = inputs.get(field_name)
if isinstance(value, list):
_validate_batched_input_length(source_request.prompt, value, field_name)
inputs[field_name] = deepcopy(value[index])
setattr(target_request, _EXPLICIT_REQUEST_ATTR, raw)
def _validate_batched_input_length(
prompts: str | list[str] | None,
values: list[Any],
field_name: str,
) -> None:
if not isinstance(prompts, list):
return
if len(values) != len(prompts):
raise ValueError(f"GenerationRequest.inputs.{field_name} must have the same length as request.prompt")
def _is_supported_as_default_only(key: str, value: Any) -> bool:
default_value = _DEFAULT_REQUEST_UPDATES.get(key, _MISSING)
return default_value is not _MISSING and _values_equal(value, default_value)
def _collect_non_default_fields(
value: Any,
default: Any,
) -> dict[str, Any]:
if not (is_dataclass(value) and is_dataclass(default)):
return {}
result: dict[str, Any] = {}
for field in fields(value):
current = getattr(value, field.name)
default_value = getattr(default, field.name)
if is_dataclass(current) and is_dataclass(default_value):
nested = _collect_non_default_fields(current, default_value)
if nested:
result[field.name] = nested
continue
if not _values_equal(current, default_value):
result[field.name] = deepcopy(current)
return result
def _values_equal(left: Any, right: Any) -> bool:
if left is right:
return True
try:
return bool(left == right)
except Exception:
return False
_DEFAULT_REQUEST_UPDATES = _extract_request_updates(config_to_dict(GenerationRequest()))
__all__ = [
"generator_config_to_fastvideo_args",
"legacy_from_pretrained_to_config",
"legacy_generate_call_to_request",
"load_generator_config_from_file",
"normalize_generation_request",
"normalize_generator_config",
"request_to_pipeline_overrides",
"request_to_sampling_param",
]
-16
View File
@@ -1,16 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
from __future__ import annotations
class ConfigValidationError(ValueError):
"""Validation error that keeps track of the nested config path."""
def __init__(self, path: str, message: str):
self.path = path
self.message = message
super().__init__(str(self))
def __str__(self) -> str:
if self.path:
return f"{self.path}: {self.message}"
return self.message
-97
View File
@@ -1,97 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
from __future__ import annotations
from copy import deepcopy
from typing import Any
from collections.abc import Mapping
import yaml
from fastvideo.api.errors import ConfigValidationError
def parse_cli_overrides(overrides: list[str]) -> dict[str, Any]:
"""Parse ``--dotted.key value`` style overrides into a flat mapping."""
parsed: dict[str, Any] = {}
index = 0
while index < len(overrides):
token = overrides[index]
if not token.startswith("--"):
raise ValueError(f"Expected --dotted.key, got {token!r}")
key = token[2:]
if not key:
raise ValueError("Override key cannot be empty")
if "=" in key:
key, raw_value = key.split("=", 1)
else:
index += 1
if index >= len(overrides):
raise ValueError(f"Missing value for override {token!r}")
raw_value = overrides[index]
parsed[key] = _cast_override_value(raw_value)
index += 1
return parsed
def apply_overrides(config: Mapping[str, Any], overrides: Mapping[str, Any]) -> dict[str, Any]:
"""Return a copy of ``config`` with dotted-key overrides applied."""
merged = deepcopy(dict(config))
for dotted_key, value in overrides.items():
_apply_single_override(merged, dotted_key, value)
return merged
def _apply_single_override(config: dict[str, Any], dotted_key: str, value: Any) -> None:
parts = dotted_key.split(".")
if not all(parts):
raise ValueError(f"Invalid override path {dotted_key!r}")
cursor = config
for depth, part in enumerate(parts[:-1]):
existing = cursor.get(part)
if existing is None:
existing = {}
cursor[part] = existing
elif not isinstance(existing, dict):
raise ConfigValidationError(
".".join(parts[:depth + 1]),
"cannot apply nested override through a non-mapping value",
)
cursor = existing
cursor[parts[-1]] = value
def _cast_override_value(raw: str) -> Any:
lowered = raw.lower()
if lowered == "true":
return True
if lowered == "false":
return False
if lowered in {"none", "null"}:
return None
try:
return int(raw)
except ValueError:
pass
try:
return float(raw)
except ValueError:
pass
if raw.startswith("[") or raw.startswith("{"):
try:
return yaml.safe_load(raw)
except yaml.YAMLError:
pass
return raw
__all__ = ["apply_overrides", "parse_cli_overrides"]
-320
View File
@@ -1,320 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
from __future__ import annotations
import dataclasses
import json
import types
from pathlib import Path
from collections.abc import Mapping
from typing import Any, Literal, TypeVar, Union, get_args, get_origin, get_type_hints
import yaml
from fastvideo.api.errors import ConfigValidationError
from fastvideo.api.overrides import apply_overrides, parse_cli_overrides
from fastvideo.api.schema import RunConfig, ServeConfig
T = TypeVar("T")
_UNION_ORIGINS = {types.UnionType, Union}
@dataclasses.dataclass(frozen=True)
class _DataclassSpec:
cls: type[Any]
type_hints: dict[str, Any]
fields_by_name: dict[str, dataclasses.Field[Any]]
def parse_config(config_type: type[T], raw: Mapping[str, Any] | T) -> T:
"""Parse a nested mapping into a typed inference config object."""
if isinstance(raw, config_type):
return raw
if not isinstance(raw, Mapping):
raise ConfigValidationError("", f"expected mapping for {config_type.__name__}")
return _SchemaParser().parse_dataclass(config_type, raw, "")
def config_to_dict(config: Any) -> Any:
"""Serialize a typed config object into plain Python containers."""
if dataclasses.is_dataclass(config) and not isinstance(config, type):
return {field.name: config_to_dict(getattr(config, field.name)) for field in dataclasses.fields(config)}
if isinstance(config, list):
return [config_to_dict(item) for item in config]
if isinstance(config, dict):
return {key: config_to_dict(value) for key, value in config.items()}
return config
def load_config(
config_type: type[T],
path: str | Path,
overrides: list[str] | Mapping[str, Any] | None = None,
) -> T:
"""Load a typed config object from YAML or JSON."""
raw = load_raw_config(path)
normalized_overrides = _normalize_overrides(overrides)
if normalized_overrides:
raw = apply_overrides(raw, normalized_overrides)
return parse_config(config_type, raw)
def load_run_config(
path: str | Path,
overrides: list[str] | Mapping[str, Any] | None = None,
) -> RunConfig:
return load_config(RunConfig, path, overrides)
def load_serve_config(
path: str | Path,
overrides: list[str] | Mapping[str, Any] | None = None,
) -> ServeConfig:
return load_config(ServeConfig, path, overrides)
def load_raw_config(path: str | Path) -> dict[str, Any]:
config_path = Path(path)
if not config_path.exists():
raise FileNotFoundError(f"Config file not found: {config_path}")
with config_path.open(encoding="utf-8") as handle:
raw = _load_raw_mapping(handle, config_path)
if raw is None:
return {}
if not isinstance(raw, Mapping):
raise ConfigValidationError("", f"{config_path} must contain a top-level mapping")
return dict(raw)
def _load_raw_mapping(handle: Any, config_path: Path) -> Any:
suffix = config_path.suffix.lower()
if suffix in {".yaml", ".yml"}:
return yaml.safe_load(handle)
if suffix == ".json":
return json.load(handle)
raise ValueError(f"Unsupported config file format: {config_path}")
def _normalize_overrides(overrides: list[str] | Mapping[str, Any] | None, ) -> dict[str, Any] | None:
if not overrides:
return None
if isinstance(overrides, list):
return parse_cli_overrides(overrides)
return dict(overrides)
class _SchemaParser:
def parse_dataclass(
self,
config_type: type[T],
raw: Mapping[str, Any],
path: str,
) -> T:
if not isinstance(raw, Mapping):
raise ConfigValidationError(path, f"expected mapping for {config_type.__name__}")
spec = _get_dataclass_spec(config_type)
self._validate_keys(raw, spec, path)
values: dict[str, Any] = {}
for name, field in spec.fields_by_name.items():
field_path = _join_path(path, name)
if name in raw:
values[name] = self.parse_value(spec.type_hints[name], raw[name], field_path)
continue
if _field_is_required(field):
raise ConfigValidationError(field_path, "missing required field")
return config_type(**values)
def parse_value(self, annotation: Any, value: Any, path: str) -> Any:
if annotation is Any:
return value
origin = get_origin(annotation)
if origin in _UNION_ORIGINS:
return self._parse_union(annotation, value, path)
if origin is Literal:
return self._parse_literal(annotation, value, path)
if origin is list:
return self._parse_list(annotation, value, path)
if origin is dict:
return self._parse_dict(annotation, value, path)
if origin is tuple:
return self._parse_tuple(annotation, value, path)
if isinstance(annotation, type) and dataclasses.is_dataclass(annotation):
return self.parse_dataclass(annotation, value, path)
scalar_parser = _SCALAR_PARSERS.get(annotation)
if scalar_parser is not None:
return scalar_parser(value, path)
return self._parse_instance(annotation, value, path)
def _validate_keys(
self,
raw: Mapping[str, Any],
spec: _DataclassSpec,
path: str,
) -> None:
for key in raw:
if not isinstance(key, str):
raise ConfigValidationError(path, "expected mapping keys to be strings")
if key not in spec.fields_by_name:
raise ConfigValidationError(_join_path(path, key), "unknown field")
def _parse_union(self, annotation: Any, value: Any, path: str) -> Any:
candidates = [candidate for candidate in get_args(annotation) if candidate is not type(None)]
if value is None and len(candidates) != len(get_args(annotation)):
return None
if len(candidates) == 1:
return self.parse_value(candidates[0], value, path)
errors: list[str] = []
for candidate in candidates:
try:
return self.parse_value(candidate, value, path)
except ConfigValidationError as exc:
errors.append(exc.message)
expected = ", ".join(_type_name(candidate) for candidate in candidates)
detail = errors[0] if errors else f"expected one of ({expected})"
raise ConfigValidationError(path, detail)
def _parse_literal(self, annotation: Any, value: Any, path: str) -> Any:
allowed = get_args(annotation)
if value not in allowed:
raise ConfigValidationError(path, f"expected one of {sorted(allowed)!r}")
return value
def _parse_list(self, annotation: Any, value: Any, path: str) -> list[Any]:
if not isinstance(value, list):
raise ConfigValidationError(path, "expected list")
item_type = get_args(annotation)[0] if get_args(annotation) else Any
return [self.parse_value(item_type, item, f"{path}[{index}]") for index, item in enumerate(value)]
def _parse_dict(self, annotation: Any, value: Any, path: str) -> dict[Any, Any]:
if not isinstance(value, Mapping):
raise ConfigValidationError(path, "expected mapping")
key_type, value_type = (get_args(annotation) + (Any, Any))[:2]
parsed: dict[Any, Any] = {}
for key, item in value.items():
parsed_key = self._parse_dict_key(key_type, key, path)
item_path = _join_path(path, str(key))
parsed[parsed_key] = self.parse_value(value_type, item, item_path)
return parsed
def _parse_tuple(self, annotation: Any, value: Any, path: str) -> tuple[Any, ...]:
if not isinstance(value, list | tuple):
raise ConfigValidationError(path, "expected tuple")
item_types = get_args(annotation)
if len(item_types) == 2 and item_types[1] is Ellipsis:
return tuple(self.parse_value(item_types[0], item, f"{path}[{index}]") for index, item in enumerate(value))
if len(value) != len(item_types):
raise ConfigValidationError(path, f"expected tuple of length {len(item_types)}")
return tuple(
self.parse_value(item_type, item, f"{path}[{index}]")
for index, (item_type, item) in enumerate(zip(item_types, value, strict=True)))
def _parse_dict_key(self, annotation: Any, value: Any, path: str) -> Any:
if annotation is Any:
return value
if annotation is str:
if not isinstance(value, str):
raise ConfigValidationError(path, "expected string dictionary keys")
return value
if annotation is int:
if not isinstance(value, int) or isinstance(value, bool):
raise ConfigValidationError(path, "expected integer dictionary keys")
return value
return value
def _parse_instance(self, annotation: Any, value: Any, path: str) -> Any:
if isinstance(annotation, type) and not isinstance(value, annotation):
raise ConfigValidationError(path, f"expected {annotation.__name__}")
return value
def _parse_bool(value: Any, path: str) -> bool:
if type(value) is not bool:
raise ConfigValidationError(path, "expected bool")
return value
def _parse_int(value: Any, path: str) -> int:
if not isinstance(value, int) or isinstance(value, bool):
raise ConfigValidationError(path, "expected int")
return value
def _parse_float(value: Any, path: str) -> float:
if not isinstance(value, int | float) or isinstance(value, bool):
raise ConfigValidationError(path, "expected float")
return float(value)
def _parse_str(value: Any, path: str) -> str:
if not isinstance(value, str):
raise ConfigValidationError(path, "expected str")
return value
_SCALAR_PARSERS: dict[Any, Any] = {
bool: _parse_bool,
int: _parse_int,
float: _parse_float,
str: _parse_str,
}
def _field_is_required(field: dataclasses.Field[Any]) -> bool:
return (field.default is dataclasses.MISSING and field.default_factory is dataclasses.MISSING)
def _get_dataclass_spec(config_type: type[Any]) -> _DataclassSpec:
spec = _DATACLASS_SPEC_CACHE.get(config_type)
if spec is not None:
return spec
spec = _DataclassSpec(
cls=config_type,
type_hints=get_type_hints(config_type),
fields_by_name={field.name: field
for field in dataclasses.fields(config_type)},
)
_DATACLASS_SPEC_CACHE[config_type] = spec
return spec
_DATACLASS_SPEC_CACHE: dict[type[Any], _DataclassSpec] = {}
def _join_path(prefix: str, suffix: str) -> str:
if not prefix:
return suffix
return f"{prefix}.{suffix}"
def _type_name(annotation: Any) -> str:
origin = get_origin(annotation)
if origin is not None:
return str(annotation)
if hasattr(annotation, "__name__"):
return annotation.__name__
return str(annotation)
__all__ = [
"config_to_dict",
"load_config",
"load_raw_config",
"load_run_config",
"load_serve_config",
"parse_config",
]
-101
View File
@@ -1,101 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Any
from collections.abc import Mapping
from fastvideo.api.schema import ContinuationState
@dataclass
class GenerationResult:
prompt: str | None = None
prompt_index: int | None = None
samples: Any | None = None
frames: Any | None = None
audio: Any | None = None
size: tuple[int, int, int] | None = None
generation_time: float | None = None
logging_info: Any | None = None
trajectory: Any | None = None
trajectory_timesteps: Any | None = None
trajectory_decoded: Any | None = None
video_path: str | None = None
peak_memory_mb: float | None = None
state: ContinuationState | None = None
extra: dict[str, Any] = field(default_factory=dict)
@classmethod
def from_legacy_result(
cls,
result: Mapping[str, Any],
) -> GenerationResult:
prompt = result.get("prompt")
if prompt is None:
prompt = result.get("prompts")
extra = {
key: value
for key, value in result.items() if key not in {
"prompt",
"prompt_index",
"prompts",
"samples",
"frames",
"audio",
"size",
"generation_time",
"logging_info",
"trajectory",
"trajectory_timesteps",
"trajectory_decoded",
"video_path",
"peak_memory_mb",
"state",
}
}
return cls(
prompt=prompt,
prompt_index=result.get("prompt_index"),
samples=result.get("samples"),
frames=result.get("frames"),
audio=result.get("audio"),
size=result.get("size"),
generation_time=result.get("generation_time"),
logging_info=result.get("logging_info"),
trajectory=result.get("trajectory"),
trajectory_timesteps=result.get("trajectory_timesteps"),
trajectory_decoded=result.get("trajectory_decoded"),
video_path=result.get("video_path"),
peak_memory_mb=result.get("peak_memory_mb"),
state=result.get("state"),
extra=extra,
)
def to_legacy_dict(self) -> dict[str, Any]:
result = {
"prompts": self.prompt,
"samples": self.samples,
"frames": self.frames,
"audio": self.audio,
"size": self.size,
"generation_time": self.generation_time,
"logging_info": self.logging_info,
"trajectory": self.trajectory,
"trajectory_timesteps": self.trajectory_timesteps,
"trajectory_decoded": self.trajectory_decoded,
"video_path": self.video_path,
"peak_memory_mb": self.peak_memory_mb,
}
if self.prompt_index is not None:
result["prompt_index"] = self.prompt_index
result["prompt"] = self.prompt
if self.state is not None:
result["state"] = self.state
result.update(self.extra)
return result
__all__ = ["GenerationResult"]
-210
View File
@@ -1,210 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Any, Literal
@dataclass
class ServerConfig:
host: str = "0.0.0.0"
port: int = 8000
output_dir: str = "outputs/"
@dataclass
class ParallelismConfig:
tp_size: int = -1
sp_size: int = -1
hsdp_replicate_dim: int = 1
hsdp_shard_dim: int = -1
dist_timeout: int | None = None
@dataclass
class OffloadConfig:
dit: bool = True
dit_layerwise: bool = True
text_encoder: bool = True
image_encoder: bool = True
vae: bool = True
pin_cpu_memory: bool = True
@dataclass
class CompileConfig:
enabled: bool = False
kwargs: dict[str, Any] = field(default_factory=dict)
@dataclass
class QuantizationConfig:
text_encoder_quant: str | None = None
transformer_quant: str | None = None
@dataclass
class EngineConfig:
num_gpus: int = 1
execution_backend: Literal["mp", "ray"] = "mp"
parallelism: ParallelismConfig = field(default_factory=ParallelismConfig)
offload: OffloadConfig = field(default_factory=OffloadConfig)
compile: CompileConfig = field(default_factory=CompileConfig)
enable_stage_verification: bool = True
use_fsdp_inference: bool = False
disable_autocast: bool = False
quantization: QuantizationConfig | None = None
@dataclass
class ComponentConfig:
config_root: str | None = None
pipeline_config_path: str | None = None
text_encoder_weights: str | None = None
transformer_weights: str | None = None
transformer_2_weights: str | None = None
vae_weights: str | None = None
upsampler_weights: str | None = None
lora_path: str | None = None
override_pipeline_cls_name: str | None = None
override_transformer_cls_name: str | None = None
@dataclass
class PipelineSelection:
workload_type: Literal["t2v", "i2v", "t2i", "i2i"] | None = None
profile: str | None = None
profile_version: str | None = None
components: ComponentConfig = field(default_factory=ComponentConfig)
profile_overrides: dict[str, Any] = field(default_factory=dict)
experimental: dict[str, Any] = field(default_factory=dict)
@dataclass
class GeneratorConfig:
model_path: str
revision: str | None = None
trust_remote_code: bool = False
engine: EngineConfig = field(default_factory=EngineConfig)
pipeline: PipelineSelection = field(default_factory=PipelineSelection)
@dataclass
class InputConfig:
prompt_path: str | None = None
image_path: str | list[str] | None = None
video_path: str | list[str] | None = None
pil_image: Any | None = None
pose: str | None = None
mouse_cond: Any | None = None
keyboard_cond: Any | None = None
grid_sizes: Any | None = None
c2ws_plucker_emb: Any | None = None
refine_from: str | None = None
stage1_video: Any | None = None
@dataclass
class SamplingConfig:
num_videos_per_prompt: int = 1
seed: int = 1024
num_frames: int = 125
height: int = 720
width: int = 1280
height_sr: int = 1072
width_sr: int = 1920
fps: int = 24
num_inference_steps: int = 50
num_inference_steps_sr: int = 50
guidance_scale: float = 1.0
guidance_scale_2: float | None = None
guidance_rescale: float = 0.0
true_cfg_scale: float | None = None
boundary_ratio: float | None = None
sigmas: list[float] | None = None
@dataclass
class RequestRuntimeConfig:
enable_teacache: bool = False
return_trajectory_latents: bool = False
return_trajectory_decoded: bool = False
@dataclass
class OutputConfig:
output_path: str = "outputs/"
output_video_name: str | None = None
save_video: bool = True
return_frames: bool = True
return_state: bool = False
@dataclass
class ContinuationState:
kind: str
payload: dict[str, Any]
@dataclass
class PlannedStage:
name: str
kind: str
source: str | None = None
overrides: dict[str, Any] = field(default_factory=dict)
@dataclass
class GenerationPlan:
stages: list[PlannedStage]
final_stage: str | None = None
@dataclass
class GenerationRequest:
prompt: str | list[str] | None = None
negative_prompt: str | None = None
inputs: InputConfig = field(default_factory=InputConfig)
sampling: SamplingConfig = field(default_factory=SamplingConfig)
runtime: RequestRuntimeConfig = field(default_factory=RequestRuntimeConfig)
output: OutputConfig = field(default_factory=OutputConfig)
stage_overrides: dict[str, Any] = field(default_factory=dict)
state: ContinuationState | None = None
plan: GenerationPlan | None = None
extensions: dict[str, Any] = field(default_factory=dict)
@dataclass
class RunConfig:
generator: GeneratorConfig
request: GenerationRequest
@dataclass
class ServeConfig:
generator: GeneratorConfig
server: ServerConfig = field(default_factory=ServerConfig)
default_request: GenerationRequest = field(default_factory=GenerationRequest)
__all__ = [
"CompileConfig",
"ComponentConfig",
"ContinuationState",
"EngineConfig",
"GenerationPlan",
"GenerationRequest",
"GeneratorConfig",
"InputConfig",
"OffloadConfig",
"OutputConfig",
"ParallelismConfig",
"PipelineSelection",
"PlannedStage",
"QuantizationConfig",
"RequestRuntimeConfig",
"RunConfig",
"SamplingConfig",
"ServeConfig",
"ServerConfig",
]
+2 -6
View File
@@ -2,7 +2,7 @@
# Adapted from vllm: https://github.com/vllm-project/vllm/blob/v0.7.3/vllm/attention/backends/abstract.py
from abc import ABC, abstractmethod
from dataclasses import dataclass, field, fields
from dataclasses import dataclass, fields
from typing import TYPE_CHECKING, Any, Generic, Protocol, TypeVar
if TYPE_CHECKING:
@@ -53,10 +53,6 @@ class AttentionMetadata:
"""Attention metadata for prefill and decode batched together."""
# Current step of diffusion process
current_timestep: int
VSA_sparsity: float = field(default=0.0, kw_only=True)
def __getattr__(self, name: str) -> Any:
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
def asdict_zerocopy(self, skip_fields: set[str] | None = None) -> dict[str, Any]:
"""Similar to dataclasses.asdict, but avoids deepcopying."""
@@ -86,7 +82,7 @@ class AttentionMetadataBuilder(ABC, Generic[T]):
@abstractmethod
def build(
self,
**kwargs: Any,
**kwargs: dict[str, Any],
) -> AttentionMetadata:
"""Build attention metadata with on-device tensors."""
raise NotImplementedError
@@ -1,121 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
import importlib
import sys
from collections.abc import Callable
from pathlib import Path
import torch
from fastvideo.attention.backends.abstract import (
AttentionBackend,
AttentionImpl,
AttentionMetadata,
AttentionMetadataBuilder,
)
from fastvideo.logger import init_logger
logger = init_logger(__name__)
_project_root = Path(__file__).resolve().parent.parent.parent.parent
_kernel_root = _project_root / "fastvideo-kernel"
_kernel_python_root = _kernel_root / "python"
_attn_qat_infer: Callable[..., torch.Tensor] | None = None
_attn_qat_infer_import_attempted = False
def _ensure_kernel_paths() -> None:
for path in (_project_root, _kernel_root, _kernel_python_root):
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
def _get_attn_qat_infer() -> Callable[..., torch.Tensor] | None:
global _attn_qat_infer
global _attn_qat_infer_import_attempted
if _attn_qat_infer_import_attempted:
return _attn_qat_infer
_attn_qat_infer_import_attempted = True
_ensure_kernel_paths()
try:
# Prefer the in-repo kernel implementation during local development.
_attn_qat_infer = importlib.import_module("attn_qat_infer").sageattn_blackwell
except ImportError:
_attn_qat_infer = None
return _attn_qat_infer
def is_attn_qat_infer_available() -> bool:
return _get_attn_qat_infer() is not None
class AttnQatInferBackend(AttentionBackend):
accept_output_buffer: bool = True
@staticmethod
def get_supported_head_sizes() -> list[int]:
return [64, 128]
@staticmethod
def get_name() -> str:
return "ATTN_QAT_INFER"
@staticmethod
def get_impl_cls() -> type["AttnQatInferImpl"]:
return AttnQatInferImpl
@staticmethod
def get_metadata_cls() -> type["AttentionMetadata"]:
raise NotImplementedError
@staticmethod
def get_builder_cls() -> type["AttentionMetadataBuilder"]:
raise NotImplementedError
class AttnQatInferImpl(AttentionImpl):
def __init__(
self,
num_heads: int,
head_size: int,
causal: bool,
softmax_scale: float,
num_kv_heads: int | None = None,
prefix: str = "",
**extra_impl_args,
) -> None:
self.causal = causal
self.softmax_scale = softmax_scale
self.dropout = extra_impl_args.get("dropout_p", 0.0)
def forward(
self,
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
attn_metadata: AttentionMetadata,
) -> torch.Tensor:
attn_qat_infer = _get_attn_qat_infer()
if attn_qat_infer is None:
raise ImportError("attn_qat_infer is not available. Please ensure the "
"attn_qat_infer kernel package is installed.")
query = query.transpose(1, 2).contiguous()
key = key.transpose(1, 2).contiguous()
value = value.transpose(1, 2).contiguous()
output = attn_qat_infer(
query,
key,
value,
attn_mask=None,
is_causal=self.causal,
)
return output.transpose(1, 2).contiguous()
@@ -1,145 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
import importlib
import sys
from collections.abc import Callable
from pathlib import Path
import torch
from fastvideo.attention.backends.abstract import (
AttentionBackend,
AttentionImpl,
AttentionMetadata,
AttentionMetadataBuilder,
)
from fastvideo.logger import init_logger
logger = init_logger(__name__)
_project_root = Path(__file__).resolve().parent.parent.parent.parent
_kernel_root = _project_root / "fastvideo-kernel"
_kernel_python_root = _kernel_root / "python"
_attn_qat_train_attention: Callable[..., torch.Tensor] | None = None
_attn_qat_train_import_attempted = False
def _ensure_kernel_paths() -> None:
for path in (_project_root, _kernel_root, _kernel_python_root):
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
def _get_attn_qat_train_attention() -> Callable[..., torch.Tensor] | None:
global _attn_qat_train_attention
global _attn_qat_train_import_attempted
if _attn_qat_train_import_attempted:
return _attn_qat_train_attention
_attn_qat_train_import_attempted = True
_ensure_kernel_paths()
try:
_attn_qat_train_attention = importlib.import_module("fastvideo_kernel.triton_kernels.attn_qat_train").attention
except ImportError:
_attn_qat_train_attention = None
return _attn_qat_train_attention
def attn_qat_train(q_BLHD: torch.Tensor,
k_BLHD: torch.Tensor,
v_BLHD: torch.Tensor,
is_causal: bool = False) -> torch.Tensor:
attention = _get_attn_qat_train_attention()
if attention is None:
raise ImportError("fastvideo_kernel.triton_kernels.attn_qat_train is not available. "
"Please ensure the FastVideo kernel package is installed.")
q_BHLD = q_BLHD.permute(0, 2, 1, 3).contiguous()
k_BHLD = k_BLHD.permute(0, 2, 1, 3).contiguous()
v_BHLD = v_BLHD.permute(0, 2, 1, 3).contiguous()
use_qat_qkv_backward = True
smooth_k = False
warp_specialize = True
is_qat = True
two_level_quant_p_sage3 = False
fake_quant_p_bwd = True
use_high_prec_o = True
smooth_q = False
sm_scale = 1.0 / (q_BHLD.shape[-1]**0.5)
use_global_sf_qkv = False
use_global_sf_p = False
o_BHLD = attention(
q_BHLD,
k_BHLD,
v_BHLD,
is_causal,
sm_scale,
use_qat_qkv_backward,
smooth_k,
warp_specialize,
is_qat,
two_level_quant_p_sage3,
fake_quant_p_bwd,
use_high_prec_o,
smooth_q,
use_global_sf_p,
use_global_sf_qkv,
)
return o_BHLD.permute(0, 2, 1, 3).contiguous()
class AttnQatTrainBackend(AttentionBackend):
accept_output_buffer: bool = True
@staticmethod
def get_supported_head_sizes() -> list[int]:
return [64, 96, 128, 160, 192, 224, 256]
@staticmethod
def get_name() -> str:
return "ATTN_QAT_TRAIN"
@staticmethod
def get_impl_cls() -> type["AttnQatTrainImpl"]:
return AttnQatTrainImpl
@staticmethod
def get_metadata_cls() -> type["AttentionMetadata"]:
raise NotImplementedError
@staticmethod
def get_builder_cls() -> type["AttentionMetadataBuilder"]:
raise NotImplementedError
class AttnQatTrainImpl(AttentionImpl):
def __init__(
self,
num_heads: int,
head_size: int,
causal: bool,
softmax_scale: float,
num_kv_heads: int | None = None,
prefix: str = "",
**extra_impl_args,
) -> None:
self.causal = causal
self.softmax_scale = softmax_scale
self.dropout = extra_impl_args.get("dropout_p", 0.0)
def forward(
self,
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
attn_metadata: AttentionMetadata,
) -> torch.Tensor:
return attn_qat_train(query, key, value, is_causal=self.causal)
-736
View File
@@ -1,736 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
"""
Bidirectional Sparse Attention (BSA) backend for FastVideo.
Pure-PyTorch reference implementation from:
"Bidirectional Sparse Attention for Faster Video Diffusion Training"
(arXiv:2509.01085)
BSA sparsifies both queries (pruning redundant tokens per block) and
key-value pairs (keeping only relevant KV blocks per query block).
This is a training-free inference backend: it works with any model
trained with full attention by applying BSA sparsity at inference time.
"""
import functools
import math
from dataclasses import dataclass
from typing import Any
import torch
import torch.nn.functional as F
from fastvideo.attention.backends.abstract import (
AttentionBackend,
AttentionImpl,
AttentionMetadata,
AttentionMetadataBuilder,
)
from fastvideo.distributed import get_sp_group
from fastvideo.logger import init_logger
try:
from fastvideo.attention.utils.flash_attn_no_pad import (
flash_attn_varlen_func_impl, )
FLASH_ATTN_AVAILABLE = True
except ImportError:
flash_attn_varlen_func_impl = None
FLASH_ATTN_AVAILABLE = False
logger = init_logger(__name__)
BSA_TILE_SIZE = (4, 4, 4)
# ---------------------------------------------------------------------------
# Cached index helpers (same pattern as VSA)
# ---------------------------------------------------------------------------
@functools.lru_cache(maxsize=10)
def get_tile_partition_indices(
dit_seq_shape: tuple[int, int, int],
tile_size: tuple[int, int, int],
device: torch.device,
) -> torch.LongTensor:
"""Map raster-order tokens to tile-contiguous order."""
T, H, W = dit_seq_shape
ts, hs, ws = tile_size
indices = torch.arange(T * H * W, device=device, dtype=torch.long).reshape(T, H, W)
ls = []
for t in range(math.ceil(T / ts)):
for h in range(math.ceil(H / hs)):
for w in range(math.ceil(W / ws)):
ls.append(indices[
t * ts:min(t * ts + ts, T),
h * hs:min(h * hs + hs, H),
w * ws:min(w * ws + ws, W),
].flatten())
return torch.cat(ls, dim=0)
@functools.lru_cache(maxsize=10)
def get_reverse_tile_partition_indices(
dit_seq_shape: tuple[int, int, int],
tile_size: tuple[int, int, int],
device: torch.device,
) -> torch.LongTensor:
"""Inverse mapping: tile-contiguous order back to raster order."""
return torch.argsort(get_tile_partition_indices(dit_seq_shape, tile_size, device))
# ---------------------------------------------------------------------------
# BSA core operations
# ---------------------------------------------------------------------------
def _prune_queries(
q_blocks: torch.Tensor,
keep_ratio: float,
) -> tuple[torch.Tensor, torch.Tensor, int]:
"""
Prune redundant query tokens within each block.
Scores tokens by cosine similarity to the block center.
Keeps the LEAST similar (most informative) tokens.
Args:
q_blocks: [B, N_heads, N_blocks, block_size, D]
keep_ratio: fraction of tokens to keep
Returns:
sparse_q: [B, N_heads, N_blocks, keep_size, D]
keep_indices: [B, N_heads, N_blocks, keep_size]
keep_size: int
"""
B, H, N, S, D = q_blocks.shape
keep_size = max(1, int(S * keep_ratio))
if keep_size >= S:
idx = torch.arange(S, device=q_blocks.device)
idx = idx.view(1, 1, 1, S).expand(B, H, N, S)
return q_blocks, idx, S
center_idx = S // 2
center = q_blocks[:, :, :, center_idx:center_idx + 1, :]
q_norm = F.normalize(q_blocks, dim=-1)
c_norm = F.normalize(center, dim=-1)
similarity = (q_norm * c_norm).sum(dim=-1) # [B, H, N, S]
# lowest similarity = most distinctive = keep
_, indices = similarity.topk(keep_size, dim=-1, largest=False)
indices, _ = indices.sort(dim=-1)
idx_expand = indices.unsqueeze(-1).expand(-1, -1, -1, -1, D)
sparse_q = torch.gather(q_blocks, 3, idx_expand)
return sparse_q, indices, keep_size
def _select_kv_blocks(
sparse_q: torch.Tensor,
k_blocks: torch.Tensor,
cumulative_threshold: float,
min_kv_blocks: int,
) -> torch.Tensor:
"""
Dynamically select KV blocks for each query block.
Mean-pools to block level, computes block attention scores,
admits blocks in descending order until cumulative mass
exceeds threshold.
Args:
sparse_q: [B, H, N, Sq, D]
k_blocks: [B, H, N, Sk, D]
cumulative_threshold: e.g. 0.9
min_kv_blocks: minimum blocks to keep
Returns:
kv_mask: [B, H, N, N] boolean
"""
B, H, N, _, D = sparse_q.shape
q_repr = sparse_q.mean(dim=3)
k_repr = k_blocks.mean(dim=3)
scores = torch.matmul(q_repr, k_repr.transpose(-1, -2)) / (D**0.5)
block_attn = F.softmax(scores, dim=-1)
sorted_attn, sorted_idx = block_attn.sort(dim=-1, descending=True)
cumsum = sorted_attn.cumsum(dim=-1)
keep_sorted = torch.ones_like(cumsum, dtype=torch.bool)
keep_sorted[..., 1:] = cumsum[..., :-1] < cumulative_threshold
min_mask = torch.zeros_like(keep_sorted)
min_mask[..., :min(min_kv_blocks, N)] = True
keep_sorted = keep_sorted | min_mask
kv_mask = torch.zeros_like(block_attn, dtype=torch.bool)
kv_mask.scatter_(-1, sorted_idx, keep_sorted)
return kv_mask
def _compute_sparse_attention(
sparse_q: torch.Tensor,
k_blocks: torch.Tensor,
v_blocks: torch.Tensor,
kv_mask: torch.Tensor,
) -> torch.Tensor:
"""
Compute attention for each query block against selected KV blocks.
Handles per-batch and per-head KV masks correctly.
Uses flash_attn_varlen_func when available on GPU.
Falls back to pure-PyTorch reference on CPU.
Args:
sparse_q: [B, H, N, Sq, D]
k_blocks: [B, H, N, Sk, D]
v_blocks: [B, H, N, Sk, D]
kv_mask: [B, H, N, N] boolean (per-batch, per-head)
Returns:
output: [B, H, N, Sq, D]
"""
if FLASH_ATTN_AVAILABLE and sparse_q.is_cuda:
return _compute_sparse_attention_flash(sparse_q, k_blocks, v_blocks, kv_mask)
else:
return _compute_sparse_attention_reference(sparse_q, k_blocks, v_blocks, kv_mask)
def _compute_sparse_attention_reference(
sparse_q: torch.Tensor,
k_blocks: torch.Tensor,
v_blocks: torch.Tensor,
kv_mask: torch.Tensor,
) -> torch.Tensor:
"""Pure-PyTorch fallback with per-batch, per-head mask support."""
B, H, N, Sq, D = sparse_q.shape
output = torch.zeros_like(sparse_q)
for b in range(B):
for h in range(H):
for qb in range(N):
selected = kv_mask[b, h, qb] # [N] boolean
sel_idx = selected.nonzero(as_tuple=True)[0]
if sel_idx.shape[0] == 0:
continue
# [num_sel * Sk, D]
sel_k = k_blocks[b, h, sel_idx].reshape(-1, D)
sel_v = v_blocks[b, h, sel_idx].reshape(-1, D)
q = sparse_q[b, h, qb] # [Sq, D]
scores = torch.matmul(q, sel_k.transpose(-1, -2)) / (D**0.5)
weights = F.softmax(scores, dim=-1)
output[b, h, qb] = torch.matmul(weights, sel_v)
return output
def _compute_sparse_attention_flash(
sparse_q: torch.Tensor,
k_blocks: torch.Tensor,
v_blocks: torch.Tensor,
kv_mask: torch.Tensor,
) -> torch.Tensor:
"""
FlashAttention implementation with per-batch, per-head mask support.
Strategy: check if all heads share the same mask. If so, use a single
FlashAttention call per batch (fast path). If not, process each head
separately (correct path).
Args:
sparse_q: [B, H, N, Sq, D]
k_blocks: [B, H, N, Sk, D]
v_blocks: [B, H, N, Sk, D]
kv_mask: [B, H, N, N] boolean
Returns:
output: [B, H, N, Sq, D]
"""
B, H, N, Sq, D = sparse_q.shape
Sk = k_blocks.shape[3]
device = sparse_q.device
output = torch.zeros_like(sparse_q)
for b in range(B):
# Check if all heads share the same mask for this batch element
# Compare each head's mask to head 0's mask
head0_mask = kv_mask[b, 0] # [N, N]
all_heads_same = all(torch.equal(kv_mask[b, h], head0_mask) for h in range(1, H))
if all_heads_same:
# Fast path: all heads share the same mask, single FA call
_flash_attn_single_mask(
sparse_q[b],
k_blocks[b],
v_blocks[b],
head0_mask,
output[b],
H,
N,
Sq,
Sk,
D,
device,
)
else:
# Per-head path: process each head individually
for h in range(H):
head_mask = kv_mask[b, h] # [N, N]
# Process single head: squeeze head dim, run FA, put back
_flash_attn_single_head(
sparse_q[b, h],
k_blocks[b, h],
v_blocks[b, h],
head_mask,
output,
b,
h,
N,
Sq,
Sk,
D,
device,
)
return output
def _flash_attn_single_mask(
sparse_q_b: torch.Tensor, # [H, N, Sq, D]
k_blocks_b: torch.Tensor, # [H, N, Sk, D]
v_blocks_b: torch.Tensor, # [H, N, Sk, D]
mask: torch.Tensor, # [N, N] boolean
output_b: torch.Tensor, # [H, N, Sq, D] (modified in-place)
H: int,
N: int,
Sq: int,
Sk: int,
D: int,
device: torch.device,
) -> None:
"""Run FlashAttention for all heads sharing the same KV mask."""
q_list = []
k_list = []
v_list = []
cu_seqlens_q = [0]
cu_seqlens_k = [0]
active_blocks = []
for qb in range(N):
selected = mask[qb] # [N] boolean
sel_idx = selected.nonzero(as_tuple=True)[0]
if sel_idx.shape[0] == 0:
continue
active_blocks.append(qb)
num_kv_tokens = sel_idx.shape[0] * Sk
# [H, Sq, D] -> [Sq, H, D]
q_block = sparse_q_b[:, qb].permute(1, 0, 2)
q_list.append(q_block)
# [H, num_sel, Sk, D] -> [num_kv_tokens, H, D]
sel_k = k_blocks_b[:, sel_idx].permute(1, 2, 0, 3).reshape(num_kv_tokens, H, D)
sel_v = v_blocks_b[:, sel_idx].permute(1, 2, 0, 3).reshape(num_kv_tokens, H, D)
k_list.append(sel_k)
v_list.append(sel_v)
cu_seqlens_q.append(cu_seqlens_q[-1] + Sq)
cu_seqlens_k.append(cu_seqlens_k[-1] + num_kv_tokens)
if not q_list:
return
flat_q = torch.cat(q_list, dim=0)
flat_k = torch.cat(k_list, dim=0)
flat_v = torch.cat(v_list, dim=0)
cu_seqlens_q_t = torch.tensor(cu_seqlens_q, dtype=torch.int32, device=device)
cu_seqlens_k_t = torch.tensor(cu_seqlens_k, dtype=torch.int32, device=device)
max_seqlen_q = Sq
max_seqlen_k = int((cu_seqlens_k_t[1:] - cu_seqlens_k_t[:-1]).max().item())
orig_dtype = flat_q.dtype
compute_dtype = orig_dtype
if compute_dtype not in (torch.float16, torch.bfloat16):
compute_dtype = torch.bfloat16
flat_q = flat_q.to(compute_dtype)
flat_k = flat_k.to(compute_dtype)
flat_v = flat_v.to(compute_dtype)
flat_out = flash_attn_varlen_func_impl(
flat_q,
flat_k,
flat_v,
cu_seqlens_q_t,
cu_seqlens_k_t,
max_seqlen_q,
max_seqlen_k,
causal=False,
)
if compute_dtype != orig_dtype:
flat_out = flat_out.to(orig_dtype)
idx = 0
for qb in active_blocks:
block_out = flat_out[idx:idx + Sq] # [Sq, H, D]
output_b[:, qb] = block_out.permute(1, 0, 2) # [H, Sq, D]
idx += Sq
def _flash_attn_single_head(
sparse_q_bh: torch.Tensor, # [N, Sq, D]
k_blocks_bh: torch.Tensor, # [N, Sk, D]
v_blocks_bh: torch.Tensor, # [N, Sk, D]
mask: torch.Tensor, # [N, N] boolean
output: torch.Tensor, # [B, H, N, Sq, D] (modified in-place)
b: int,
h: int,
N: int,
Sq: int,
Sk: int,
D: int,
device: torch.device,
) -> None:
"""Run FlashAttention for a single head with its own KV mask."""
q_list = []
k_list = []
v_list = []
cu_seqlens_q = [0]
cu_seqlens_k = [0]
active_blocks = []
for qb in range(N):
selected = mask[qb]
sel_idx = selected.nonzero(as_tuple=True)[0]
if sel_idx.shape[0] == 0:
continue
active_blocks.append(qb)
num_kv_tokens = sel_idx.shape[0] * Sk
# [Sq, D] -> [Sq, 1, D] (single head)
q_block = sparse_q_bh[qb].unsqueeze(1)
q_list.append(q_block)
# [num_sel, Sk, D] -> [num_kv_tokens, 1, D]
sel_k = k_blocks_bh[sel_idx].reshape(num_kv_tokens, 1, D)
sel_v = v_blocks_bh[sel_idx].reshape(num_kv_tokens, 1, D)
k_list.append(sel_k)
v_list.append(sel_v)
cu_seqlens_q.append(cu_seqlens_q[-1] + Sq)
cu_seqlens_k.append(cu_seqlens_k[-1] + num_kv_tokens)
if not q_list:
return
flat_q = torch.cat(q_list, dim=0)
flat_k = torch.cat(k_list, dim=0)
flat_v = torch.cat(v_list, dim=0)
cu_seqlens_q_t = torch.tensor(cu_seqlens_q, dtype=torch.int32, device=device)
cu_seqlens_k_t = torch.tensor(cu_seqlens_k, dtype=torch.int32, device=device)
max_seqlen_q = Sq
max_seqlen_k = int((cu_seqlens_k_t[1:] - cu_seqlens_k_t[:-1]).max().item())
orig_dtype = flat_q.dtype
compute_dtype = orig_dtype
if compute_dtype not in (torch.float16, torch.bfloat16):
compute_dtype = torch.bfloat16
flat_q = flat_q.to(compute_dtype)
flat_k = flat_k.to(compute_dtype)
flat_v = flat_v.to(compute_dtype)
flat_out = flash_attn_varlen_func_impl(
flat_q,
flat_k,
flat_v,
cu_seqlens_q_t,
cu_seqlens_k_t,
max_seqlen_q,
max_seqlen_k,
causal=False,
)
if compute_dtype != orig_dtype:
flat_out = flat_out.to(orig_dtype)
idx = 0
for qb in active_blocks:
block_out = flat_out[idx:idx + Sq] # [Sq, 1, D]
output[b, h, qb] = block_out.squeeze(1) # [Sq, D]
idx += Sq
def _reconstruct_pruned(
sparse_output: torch.Tensor,
keep_indices: torch.Tensor,
block_size: int,
) -> torch.Tensor:
"""
Scatter sparse output back to full block size.
Pruned positions get nearest kept token's output.
Handles per-batch, per-head indices correctly.
Args:
sparse_output: [B, H, N, keep_size, D]
keep_indices: [B, H, N, keep_size]
block_size: original tokens per block
Returns:
full_output: [B, H, N, block_size, D]
"""
B, H, N, keep_size, D = sparse_output.shape
device = sparse_output.device
if keep_size >= block_size:
return sparse_output
full_output = torch.zeros(B, H, N, block_size, D, device=device, dtype=sparse_output.dtype)
# Scatter kept tokens
idx_expand = keep_indices.unsqueeze(-1).expand(-1, -1, -1, -1, D)
full_output.scatter_(3, idx_expand, sparse_output)
# Fill pruned positions with nearest kept token (vectorized)
all_pos = torch.arange(block_size, device=device)
for b in range(B):
for h in range(H):
for n in range(N):
kept = keep_indices[b, h, n] # [keep_size]
# Distance from every position to every kept position
dists = (all_pos.view(-1, 1) - kept.view(1, -1)).abs()
nearest_local_idx = dists.argmin(dim=1) # [block_size]
# Identify pruned positions
is_pruned = torch.ones(block_size, dtype=torch.bool, device=device)
is_pruned[kept] = False
pruned_indices = is_pruned.nonzero(as_tuple=True)[0]
if pruned_indices.numel() > 0:
src_indices = nearest_local_idx[pruned_indices]
full_output[b, h, n, pruned_indices] = sparse_output[b, h, n, src_indices]
return full_output
# ---------------------------------------------------------------------------
# FastVideo backend classes
# ---------------------------------------------------------------------------
class BSAAttentionBackend(AttentionBackend):
accept_output_buffer: bool = False
@staticmethod
def get_supported_head_sizes() -> list[int]:
return [64, 128]
@staticmethod
def get_name() -> str:
return "BSA_ATTN"
@staticmethod
def get_impl_cls() -> type["BSAAttentionImpl"]:
return BSAAttentionImpl
@staticmethod
def get_metadata_cls() -> type["BSAAttentionMetadata"]:
return BSAAttentionMetadata
@staticmethod
def get_builder_cls() -> type["BSAAttentionMetadataBuilder"]:
return BSAAttentionMetadataBuilder
@dataclass
class BSAAttentionMetadata(AttentionMetadata):
current_timestep: int
dit_seq_shape: tuple[int, int, int]
total_seq_length: int
num_blocks: int
block_size: int
tile_partition_indices: torch.LongTensor
reverse_tile_partition_indices: torch.LongTensor
# BSA-specific config
query_keep_ratio: float
kv_cumulative_threshold: float
min_kv_blocks: int
class BSAAttentionMetadataBuilder(AttentionMetadataBuilder):
def __init__(self):
pass
def prepare(self):
pass
def build(
self,
current_timestep: int,
raw_latent_shape: tuple[int, int, int],
patch_size: tuple[int, int, int],
device: torch.device,
bsa_query_keep_ratio: float = 0.5,
bsa_kv_cumulative_threshold: float = 0.9,
bsa_min_kv_blocks: int = 4,
**kwargs: dict[str, Any],
) -> "BSAAttentionMetadata":
# Ensure patching does not drop tokens silently.
assert all(r % p == 0 for r, p in zip(raw_latent_shape, patch_size, strict=False)), (
"raw_latent_shape must be divisible by patch_size for BSA", )
dit_seq_shape = (
raw_latent_shape[0] // patch_size[0],
raw_latent_shape[1] // patch_size[1],
raw_latent_shape[2] // patch_size[2],
)
total_seq_length = math.prod(dit_seq_shape)
block_size = math.prod(BSA_TILE_SIZE)
# Require exact tiling to avoid reshape failures later.
assert all(d % t == 0 for d, t in zip(dit_seq_shape, BSA_TILE_SIZE, strict=False)), (
"dit_seq_shape must be divisible by BSA_TILE_SIZE", )
num_blocks = total_seq_length // block_size
tile_partition_indices = get_tile_partition_indices(dit_seq_shape, BSA_TILE_SIZE, device)
reverse_tile_partition_indices = get_reverse_tile_partition_indices(dit_seq_shape, BSA_TILE_SIZE, device)
return BSAAttentionMetadata(
current_timestep=current_timestep,
dit_seq_shape=dit_seq_shape,
total_seq_length=total_seq_length,
num_blocks=num_blocks,
block_size=block_size,
tile_partition_indices=tile_partition_indices,
reverse_tile_partition_indices=reverse_tile_partition_indices,
query_keep_ratio=bsa_query_keep_ratio,
kv_cumulative_threshold=bsa_kv_cumulative_threshold,
min_kv_blocks=bsa_min_kv_blocks,
)
class BSAAttentionImpl(AttentionImpl):
def __init__(
self,
num_heads: int,
head_size: int,
causal: bool,
softmax_scale: float,
num_kv_heads: int | None = None,
prefix: str = "",
**extra_impl_args,
) -> None:
self.prefix = prefix
self.num_heads = num_heads
self.head_size = head_size
if num_kv_heads is not None and num_kv_heads != num_heads:
raise ValueError("BSA backend does not support grouped-query attention")
if causal:
raise ValueError("BSA backend is bidirectional; causal=True is unsupported")
if softmax_scale is not None:
expected_scale = 1.0 / math.sqrt(self.head_size)
if not math.isclose(softmax_scale, expected_scale, rel_tol=1e-4, abs_tol=1e-5):
raise ValueError("softmax_scale must be default (1/sqrt(d)) for BSA")
try:
sp_group = get_sp_group()
self.sp_size = sp_group.world_size
except (AssertionError, RuntimeError):
self.sp_size = 1
def preprocess_qkv(
self,
qkv: torch.Tensor,
attn_metadata: BSAAttentionMetadata,
) -> torch.Tensor:
"""Reorder tokens from raster order to tile-contiguous order."""
# qkv: [B, L, num_heads, D]
return qkv[:, attn_metadata.tile_partition_indices]
def postprocess_output(
self,
output: torch.Tensor,
attn_metadata: BSAAttentionMetadata,
) -> torch.Tensor:
"""Reorder tokens from tile-contiguous order back to raster order."""
return output[:, attn_metadata.reverse_tile_partition_indices]
def forward(
self,
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
attn_metadata: BSAAttentionMetadata,
) -> torch.Tensor:
"""
BSA attention forward pass.
Input tensors are already in tile-contiguous order from preprocess_qkv.
Args:
query: [B, L, num_heads, D] (tile-ordered)
key: [B, L, num_heads, D] (tile-ordered)
value: [B, L, num_heads, D] (tile-ordered)
attn_metadata: BSA metadata
Returns:
output: [B, L, num_heads, D] (tile-ordered)
"""
B, L, H, D = query.shape
block_size = attn_metadata.block_size
num_blocks = attn_metadata.num_blocks
assert num_blocks * block_size == L, "Sequence length must match tiling"
# Reshape to [B, H, L, D] for attention computation
q = query.transpose(1, 2).contiguous() # [B, H, L, D]
k = key.transpose(1, 2).contiguous()
v = value.transpose(1, 2).contiguous()
# Reshape into blocks: [B, H, num_blocks, block_size, D]
q_blocks = q.view(B, H, num_blocks, block_size, D)
k_blocks = k.view(B, H, num_blocks, block_size, D)
v_blocks = v.view(B, H, num_blocks, block_size, D)
# --- Query sparsification ---
sparse_q, keep_indices, keep_size = _prune_queries(q_blocks, attn_metadata.query_keep_ratio)
# --- KV block selection ---
kv_mask = _select_kv_blocks(
sparse_q,
k_blocks,
attn_metadata.kv_cumulative_threshold,
attn_metadata.min_kv_blocks,
)
# --- Sparse attention ---
sparse_output = _compute_sparse_attention(sparse_q, k_blocks, v_blocks, kv_mask)
# --- Reconstruct pruned positions ---
full_output = _reconstruct_pruned(sparse_output, keep_indices, block_size)
# Reshape back: [B, H, num_blocks, block_size, D] -> [B, H, L, D] -> [B, L, H, D]
hidden_states = full_output.view(B, H, L, D).transpose(1, 2)
return hidden_states
+1 -1
View File
@@ -16,7 +16,7 @@ class SageAttention3Backend(AttentionBackend):
@staticmethod
def get_supported_head_sizes() -> list[int]:
return [64, 128]
return [64, 128, 256]
@staticmethod
def get_name() -> str:
@@ -133,6 +133,7 @@ class VideoSparseAttentionBackend(AttentionBackend):
class VideoSparseAttentionMetadata(AttentionMetadata):
current_timestep: int
dit_seq_shape: list[int]
VSA_sparsity: float
num_tiles: list[int]
total_seq_length: int
tile_partition_indices: torch.LongTensor
@@ -143,10 +144,10 @@ class VideoSparseAttentionMetadata(AttentionMetadata):
class VideoSparseAttentionMetadataBuilder(AttentionMetadataBuilder):
def __init__(self) -> None:
def __init__(self):
pass
def prepare(self) -> None:
def prepare(self):
pass
def build( # type: ignore
+4 -8
View File
@@ -61,10 +61,10 @@ class VideoMobaAttentionMetadata(AttentionMetadata):
class VideoMobaAttentionMetadataBuilder(AttentionMetadataBuilder):
def __init__(self) -> None:
def __init__(self):
pass
def prepare(self) -> None:
def prepare(self):
pass
def build( # type: ignore
@@ -81,7 +81,7 @@ class VideoMobaAttentionMetadataBuilder(AttentionMetadataBuilder):
moba_select_mode: str = 'threshold',
moba_threshold: float = 0.25,
moba_threshold_type: str = 'query_head',
device: torch.device | None = None,
device: torch.device = None,
first_full_layer: int = 0,
first_full_step: int = 12,
temporal_layer: int = 1,
@@ -142,7 +142,7 @@ class VMOBAAttentionImpl(AttentionImpl):
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
attn_metadata: VideoMobaAttentionMetadata,
attn_metadata: AttentionMetadata,
) -> torch.Tensor:
"""
query: [B, L, H, D]
@@ -154,9 +154,7 @@ class VMOBAAttentionImpl(AttentionImpl):
# select chunk type according to layer idx:
loop_layer_num = attn_metadata.temporal_layer + attn_metadata.spatial_layer + attn_metadata.st_layer
assert self.layer_idx is not None, "VMoBA attention requires layer_idx to be set"
moba_layer = self.layer_idx - attn_metadata.first_full_layer
moba_chunk_size: int | tuple[int, int] | tuple[int, int, int]
if moba_layer % loop_layer_num < attn_metadata.temporal_layer:
moba_chunk_size = attn_metadata.temporal_chunk_size
moba_topk = attn_metadata.temporal_topk
@@ -166,8 +164,6 @@ class VMOBAAttentionImpl(AttentionImpl):
elif moba_layer % loop_layer_num < attn_metadata.temporal_layer + attn_metadata.spatial_layer + attn_metadata.st_layer:
moba_chunk_size = attn_metadata.st_chunk_size
moba_topk = attn_metadata.st_topk
else:
raise ValueError(f"Invalid MoBA layer selection for layer {moba_layer}")
query, chunk_size = process_moba_input(query, attn_metadata.patch_resolution, moba_chunk_size)
key, chunk_size = process_moba_input(key, attn_metadata.patch_resolution, moba_chunk_size)
-2
View File
@@ -158,7 +158,6 @@ class DistributedAttention_VSA(DistributedAttention):
replicated_v: torch.Tensor | None = None,
gate_compress: torch.Tensor | None = None,
freqs_cis: tuple[torch.Tensor, torch.Tensor] | None = None,
attention_mask: torch.Tensor | None = None,
) -> tuple[torch.Tensor, torch.Tensor | None]:
"""Forward pass for distributed attention.
@@ -171,7 +170,6 @@ class DistributedAttention_VSA(DistributedAttention):
replicated_q (Optional[torch.Tensor]): Replicated query tensor, typically for text tokens
replicated_k (Optional[torch.Tensor]): Replicated key tensor
replicated_v (Optional[torch.Tensor]): Replicated value tensor
attention_mask (Optional[torch.Tensor]): Attention mask [batch_size, seq_len]
Returns:
Tuple[torch.Tensor, Optional[torch.Tensor]]: A tuple containing:
+16 -23
View File
@@ -85,26 +85,7 @@ def get_attn_backend(
supported_attention_backends: tuple[AttentionBackendEnum, ...]
| None = None,
) -> type[AttentionBackend]:
selected_backend, is_forced = _resolve_backend_override()
return _cached_get_attn_backend(
head_size,
dtype,
supported_attention_backends,
selected_backend,
is_forced,
)
def _resolve_backend_override() -> tuple[AttentionBackendEnum | None, bool]:
backend_by_global_setting = get_global_forced_attn_backend()
if backend_by_global_setting is not None:
return backend_by_global_setting, True
backend_by_env_var: str | None = envs.FASTVIDEO_ATTENTION_BACKEND
if backend_by_env_var is not None:
return backend_name_to_enum(backend_by_env_var), False
return None, False
return _cached_get_attn_backend(head_size, dtype, supported_attention_backends)
@cache
@@ -113,16 +94,28 @@ def _cached_get_attn_backend(
dtype: torch.dtype,
supported_attention_backends: tuple[AttentionBackendEnum, ...]
| None = None,
selected_backend: AttentionBackendEnum | None = None,
is_forced_backend: bool = False,
) -> type[AttentionBackend]:
# Check whether a particular choice of backend was
# previously forced.
#
# THIS SELECTION OVERRIDES THE FASTVIDEO_ATTENTION_BACKEND
# ENVIRONMENT VARIABLE.
if not supported_attention_backends:
raise ValueError("supported_attention_backends is empty")
selected_backend = None
backend_by_global_setting: AttentionBackendEnum | None = (get_global_forced_attn_backend())
if backend_by_global_setting is not None:
selected_backend = backend_by_global_setting
else:
# Check the environment variable and override if specified
backend_by_env_var: str | None = envs.FASTVIDEO_ATTENTION_BACKEND
if backend_by_env_var is not None:
selected_backend = backend_name_to_enum(backend_by_env_var)
# get device-specific attn_backend
from fastvideo.platforms import current_platform
if not is_forced_backend and selected_backend not in supported_attention_backends:
if selected_backend not in supported_attention_backends:
selected_backend = None
attention_cls = current_platform.get_attn_backend_cls(selected_backend, head_size, dtype)
if not attention_cls:
+33 -47
View File
@@ -14,44 +14,30 @@
# of rights and permissions under this agreement.
# See the License for the specific language governing permissions and limitations under the License.
from typing import Any
import torch
from einops import rearrange
from flash_attn import flash_attn_varlen_qkvpacked_func
from flash_attn.bert_padding import pad_input, unpad_input
from flash_attn import flash_attn_varlen_qkvpacked_func
def _resolve_flash_attn_varlen_func() -> Any:
try:
from fastvideo.attention.utils.flash_attn_cute import (
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
except ImportError:
try:
from fastvideo.attention.utils.flash_attn_cute import (
flash_attn_varlen_func as flash_attn_varlen_func_cute, )
return flash_attn_varlen_func_cute
from flash_attn_interface import (
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
except ImportError:
try:
from flash_attn_interface import (
flash_attn_varlen_func as flash_attn_varlen_func_interface, )
return flash_attn_varlen_func_interface
except ImportError:
from flash_attn import (
flash_attn_varlen_func as flash_attn_varlen_func_flash, )
return flash_attn_varlen_func_flash
flash_attn_varlen_func_impl = _resolve_flash_attn_varlen_func()
from flash_attn import (
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
def flash_attn_no_pad(
qkv: torch.Tensor,
key_padding_mask: torch.Tensor,
causal: bool = False,
dropout_p: float = 0.0,
softmax_scale: float | None = None,
deterministic: bool = False,
) -> torch.Tensor:
qkv,
key_padding_mask,
causal=False,
dropout_p=0.0,
softmax_scale=None,
deterministic=False,
):
batch_size = qkv.shape[0]
seqlen = qkv.shape[1]
nheads = qkv.shape[-2]
@@ -82,13 +68,13 @@ def flash_attn_no_pad(
def flash_attn_no_pad_v3(
qkv: torch.Tensor,
key_padding_mask: torch.Tensor,
causal: bool = False,
dropout_p: float = 0.0,
softmax_scale: float | None = None,
deterministic: bool = False,
) -> torch.Tensor:
qkv,
key_padding_mask,
causal=False,
dropout_p=0.0,
softmax_scale=None,
deterministic=False,
):
from flash_attn_interface import (
flash_attn_varlen_func as flash_attn_varlen_func_v3, )
@@ -134,16 +120,16 @@ def flash_attn_no_pad_v3(
def flash_attn_varlen_qk_no_pad(
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
query_padding_mask: torch.Tensor,
key_padding_mask: torch.Tensor,
causal: bool = False,
dropout_p: float = 0.0,
softmax_scale: float | None = None,
deterministic: bool = False,
) -> torch.Tensor:
query,
key,
value,
query_padding_mask,
key_padding_mask,
causal=False,
dropout_p=0.0,
softmax_scale=None,
deterministic=False,
):
batch_size, q_seqlen, nheads, _ = query.shape
query_unpad, q_indices, cu_seqlens_q, max_seqlen_q, _ = unpad_input(rearrange(query, "b s h d -> b s (h d)"),
+4 -14
View File
@@ -12,15 +12,8 @@ logger = init_logger(__name__)
# 3. Any field in ArchConfig is fixed upon initialization, and should be hidden away from users
@dataclass
class ArchConfig:
stacked_params_mapping: list[tuple[str, str, str | int]] = field(
stacked_params_mapping: list[tuple[str, str, str]] = field(
default_factory=list) # mapping from huggingface weight names to custom names
output_hidden_states: bool = False
def __post_init__(self) -> None:
pass
def __getattr__(self, name: str) -> Any:
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
@dataclass
@@ -31,22 +24,19 @@ class ModelConfig:
# FastVideo-specific parameters here
def __post_init__(self) -> None:
pass
def __getattr__(self, name: str) -> Any:
def __getattr__(self, name):
# Only called if 'name' is not found in ModelConfig directly
if hasattr(self.arch_config, name):
return getattr(self.arch_config, name)
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
def __getstate__(self) -> dict[str, Any]:
def __getstate__(self):
# Return a dictionary of attributes to pickle
# Convert to dict and exclude any problematic attributes
state = self.__dict__.copy()
return state
def __setstate__(self, state: dict[str, Any]) -> None:
def __setstate__(self, state):
# Restore instance attributes from the unpickled state
self.__dict__.update(state)

Some files were not shown because too many files have changed in this diff Show More