Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3a7925c674 |
+2
-186
@@ -9,183 +9,11 @@ notify:
|
||||
- github_commit_status:
|
||||
context: "full-suite-passed"
|
||||
if: build.env("TEST_SCOPE") == "full"
|
||||
- github_commit_status:
|
||||
context: "direct-test-completed"
|
||||
if: build.env("TEST_SCOPE") == "direct"
|
||||
|
||||
steps:
|
||||
# ============================================================
|
||||
# Direct test: triggered by /test <name> slash command.
|
||||
# Labels match fastcheck/full-suite counterparts so the GitHub
|
||||
# check status overwrites the original failed check.
|
||||
# Only ONE step executes per build (gated by TEST_TYPE).
|
||||
# ============================================================
|
||||
|
||||
# --- Fastcheck-scope direct tests ---
|
||||
- label: ":microscope: Encoder Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "encoder"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: VAE Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "vae"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Transformer Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "transformer"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Kernel Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "kernel_tests"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Unit Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "unit_test"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
# --- Full-suite-scope direct tests ---
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "ssim"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "distillation_dmd"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "self_forcing"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_vsa"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_vmoba"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Performance Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "performance"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: API Server Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "api_server"
|
||||
- label: ":dart: Direct Test (${TEST_TYPE})"
|
||||
if: build.env("TEST_SCOPE") == "direct"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
@@ -307,10 +135,6 @@ steps:
|
||||
label: ":bar_chart: SSIM Tests"
|
||||
env:
|
||||
- TEST_TYPE=ssim
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
@@ -371,10 +195,6 @@ steps:
|
||||
label: ":test_tube: LoRA Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training_lora
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
@@ -387,10 +207,6 @@ steps:
|
||||
label: ":test_tube: Training Tests VSA"
|
||||
env:
|
||||
- TEST_TYPE=training_vsa
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
|
||||
+11
-6
@@ -4,10 +4,8 @@ merge_protections:
|
||||
- base = main
|
||||
success_conditions:
|
||||
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model)\\]"
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success~=pre-commit
|
||||
- check-success=fastcheck-passed
|
||||
- check-success=full-suite-passed
|
||||
|
||||
pull_request_rules:
|
||||
|
||||
@@ -274,15 +272,24 @@ pull_request_rules:
|
||||
merge:
|
||||
method: squash
|
||||
|
||||
- name: auto-update when ready
|
||||
- name: auto-rebase when ready and Full Suite passed
|
||||
conditions:
|
||||
- label=ready
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success=full-suite-passed
|
||||
- -conflict
|
||||
- -closed
|
||||
- -draft
|
||||
actions:
|
||||
update: {}
|
||||
rebase: {}
|
||||
|
||||
- name: remove ready label on Full Suite failure
|
||||
conditions:
|
||||
- label=ready
|
||||
- check-failure=full-suite-passed
|
||||
actions:
|
||||
label:
|
||||
remove: [ready]
|
||||
|
||||
# ============================================================
|
||||
# PR title format help
|
||||
@@ -312,5 +319,3 @@ pull_request_rules:
|
||||
|
||||
Please update your PR title and the merge protection check will pass automatically.
|
||||
|
||||
merge_protections_settings:
|
||||
reporting_method: check-runs
|
||||
|
||||
@@ -1,80 +0,0 @@
|
||||
name: Aggregate Test Status
|
||||
|
||||
on:
|
||||
status:
|
||||
|
||||
permissions:
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
aggregate:
|
||||
if: >-
|
||||
github.event.context == 'direct-test-completed'
|
||||
&& github.event.state == 'success'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check and update aggregate status
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const sha = context.payload.sha;
|
||||
|
||||
const { data } = await github.rest.repos.getCombinedStatusForRef({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
ref: sha,
|
||||
per_page: 100,
|
||||
});
|
||||
|
||||
const bkStatuses = data.statuses.filter(
|
||||
s => s.context.startsWith('buildkite/ci/')
|
||||
);
|
||||
|
||||
const FASTCHECK_PREFIX = 'buildkite/ci/microscope-';
|
||||
const FULL_SUITE_PREFIXES = [
|
||||
'buildkite/ci/test-tube-',
|
||||
'buildkite/ci/bar-chart-',
|
||||
];
|
||||
|
||||
const fastcheck = bkStatuses.filter(
|
||||
s => s.context.startsWith(FASTCHECK_PREFIX)
|
||||
);
|
||||
const fullSuite = bkStatuses.filter(
|
||||
s => FULL_SUITE_PREFIXES.some(p => s.context.startsWith(p))
|
||||
);
|
||||
|
||||
if (
|
||||
fastcheck.length > 0
|
||||
&& fastcheck.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
`All ${fastcheck.length} fastcheck tests passed — updating fastcheck-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'fastcheck-passed',
|
||||
description:
|
||||
`All ${fastcheck.length} fastcheck tests passed`,
|
||||
});
|
||||
}
|
||||
|
||||
if (
|
||||
fullSuite.length > 0
|
||||
&& fullSuite.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
`All ${fullSuite.length} full suite tests passed — updating full-suite-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'full-suite-passed',
|
||||
description:
|
||||
`All ${fullSuite.length} full suite tests passed`,
|
||||
});
|
||||
}
|
||||
@@ -4,11 +4,10 @@ on:
|
||||
pull_request:
|
||||
branches: [main]
|
||||
workflow_call:
|
||||
inputs:
|
||||
ref:
|
||||
description: 'Git ref to checkout (defaults to github.ref)'
|
||||
required: false
|
||||
type: string
|
||||
|
||||
concurrency:
|
||||
group: pre-commit-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -19,8 +18,6 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ inputs.ref || '' }}
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
@@ -7,7 +7,6 @@ on:
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
handle-merge:
|
||||
@@ -33,7 +32,6 @@ jobs:
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
|
||||
- name: Add ready label and react
|
||||
id: label
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
@@ -41,6 +39,7 @@ jobs:
|
||||
const owner = context.repo.owner;
|
||||
const repo = context.repo.repo;
|
||||
const prNumber = context.payload.issue.number;
|
||||
// Remove ready first to allow re-trigger (labeled event fires on add, not if already present)
|
||||
try { await github.rest.issues.removeLabel({ owner, repo, issue_number: prNumber, name: 'ready' }); } catch {}
|
||||
await github.rest.issues.addLabels({ owner, repo, issue_number: prNumber, labels: ['ready'] });
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
@@ -48,44 +47,6 @@ jobs:
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: prNumber });
|
||||
core.setOutput('pr_sha', pr.head.sha);
|
||||
core.setOutput('pr_branch', pr.head.ref);
|
||||
core.setOutput('pr_number', String(prNumber));
|
||||
|
||||
- name: Trigger Full Suite
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ steps.label.outputs.pr_sha }}
|
||||
PR_BRANCH: ${{ steps.label.outputs.pr_branch }}
|
||||
PR_NUMBER: ${{ steps.label.outputs.pr_number }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "Full Suite for PR #${PR_NUMBER} (via /merge)" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: "full",
|
||||
FULL_SUITE: "true",
|
||||
PR_NUMBER: ($pr_id | tostring)
|
||||
}
|
||||
}')"
|
||||
|
||||
parse-command:
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
@@ -181,33 +142,20 @@ jobs:
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: React to comment
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
|
||||
pre-commit:
|
||||
needs: parse-command
|
||||
if: >-
|
||||
needs.parse-command.outputs.has_write == 'true'
|
||||
&& needs.parse-command.outputs.test_scope == 'precommit'
|
||||
uses: ./.github/workflows/ci-precommit.yml
|
||||
with:
|
||||
ref: refs/pull/${{ github.event.issue.number }}/merge
|
||||
|
||||
post-precommit-status:
|
||||
needs: [parse-command, pre-commit]
|
||||
if: always() && needs.parse-command.outputs.test_scope == 'precommit'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
- name: Post commit status
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
env:
|
||||
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
|
||||
RESULT: ${{ needs.pre-commit.result }}
|
||||
@@ -230,6 +178,17 @@ jobs:
|
||||
&& needs.parse-command.outputs.test_type != ''
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: React to comment
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
|
||||
- name: Trigger Buildkite
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
name: Trigger Full Suite
|
||||
|
||||
on:
|
||||
pull_request_target:
|
||||
pull_request:
|
||||
types: [labeled, synchronize]
|
||||
|
||||
permissions:
|
||||
@@ -10,7 +10,7 @@ permissions:
|
||||
|
||||
concurrency:
|
||||
group: full-suite-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: false
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
trigger:
|
||||
@@ -42,7 +42,7 @@ jobs:
|
||||
# Find running builds for this branch with TEST_SCOPE=full and cancel them
|
||||
builds=$(curl -sS -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds?branch=${PR_BRANCH}&state=running,scheduled" \
|
||||
| jq -r '.[] | select(try (.env.TEST_SCOPE == "full") catch false) | .number')
|
||||
| jq -r '.[] | select(.env.TEST_SCOPE == "full") | .number')
|
||||
for build_num in $builds; do
|
||||
echo "Cancelling Buildkite build #$build_num"
|
||||
curl -sS -X PUT -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
|
||||
@@ -85,7 +85,6 @@ docs/distillation/examples/
|
||||
dmd_t2v_output/
|
||||
preprocess_output_text/
|
||||
|
||||
# Next.js / Node artifacts under ui/: see ui/.gitignore
|
||||
|
||||
.claude/
|
||||
.codex/
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
WRN 2026-03-26T13:46:33.469 ?.19646 server_start:193: Failed to start server: operation not permitted: /var/folders/z_/h_6myyk14d1b7z87z3vy4mjh0000gn/T/nvim.dsynkd/iSe0el/nvim.19646.0
|
||||
@@ -1 +0,0 @@
|
||||
3.12
|
||||
@@ -1,318 +0,0 @@
|
||||
# Attention QAT
|
||||
|
||||
Attention QAT in FastVideo covers two related, but different, backends:
|
||||
|
||||
- `ATTN_QAT_INFER`: the inference-oriented CUDA kernel path
|
||||
- `ATTN_QAT_TRAIN`: the training-oriented Triton attention path
|
||||
|
||||
Both are selected with `FASTVIDEO_ATTENTION_BACKEND`, but they are not
|
||||
interchangeable. The main practical split is:
|
||||
|
||||
- use `ATTN_QAT_INFER` for standalone inference with the dedicated inference
|
||||
kernel
|
||||
- use `ATTN_QAT_TRAIN` for finetuning, validation during training, or when you
|
||||
specifically want to reproduce the training-side attention path
|
||||
|
||||
## Quick Start
|
||||
|
||||
If your goal is "run Wan 2.1 14B with Attention QAT inference weights", this is
|
||||
the shortest path:
|
||||
|
||||
1. Build the in-repo kernel package so FastVideo can import `attn_qat_infer`.
|
||||
2. Download the Wan 2.1 14B QAT checkpoint.
|
||||
3. Edit the provided inference example to point at the 14B base model and the
|
||||
downloaded QAT safetensors.
|
||||
4. Run the example with `ATTN_QAT_INFER`.
|
||||
|
||||
### Step 1. Build the kernel package
|
||||
|
||||
Before using either Attention QAT backend, build the in-repo
|
||||
`fastvideo-kernel` package from source:
|
||||
|
||||
```bash
|
||||
git submodule update --init --recursive
|
||||
cd fastvideo-kernel
|
||||
./build.sh
|
||||
```
|
||||
|
||||
After a successful build:
|
||||
|
||||
- `ATTN_QAT_TRAIN` should be able to import `fastvideo_kernel`
|
||||
- `ATTN_QAT_INFER` should be able to import `attn_qat_infer`
|
||||
|
||||
`ATTN_QAT_INFER` currently targets the Blackwell CUDA path under
|
||||
`fastvideo-kernel/attn_qat_infer/` and requires CUDA 12.8+.
|
||||
|
||||
### Step 2. Download the Wan 2.1 14B QAT checkpoint
|
||||
|
||||
FastVideo includes a helper script:
|
||||
|
||||
- `examples/inference/optimizations/download_14B_qat.sh`
|
||||
|
||||
By default it downloads:
|
||||
|
||||
- Hugging Face repo: `FastVideo/14B_qat_400`
|
||||
- local directory: `checkpoints/14B_qat_400`
|
||||
|
||||
Prerequisites:
|
||||
|
||||
- `huggingface_hub` installed, for example:
|
||||
`uv pip install huggingface_hub`
|
||||
- access to the model repo if it is private or gated:
|
||||
`huggingface-cli login`
|
||||
|
||||
Run the downloader:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh
|
||||
```
|
||||
|
||||
To download into a custom directory:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
|
||||
```
|
||||
|
||||
The script prints a ready-to-copy `init_weights_from_safetensors=...` value at
|
||||
the end.
|
||||
|
||||
### Step 3. Edit the provided inference example
|
||||
|
||||
The example to start from is:
|
||||
|
||||
- `examples/inference/optimizations/attn_qat_inference_example.py`
|
||||
|
||||
Open that file and update these two values:
|
||||
|
||||
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
|
||||
`Wan-AI/Wan2.1-T2V-14B-Diffusers`
|
||||
2. Replace
|
||||
`init_weights_from_safetensors="safetensors_path"` with the directory that
|
||||
contains the downloaded `.safetensors` files
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
init_weights_from_safetensors="checkpoints/14B_qat_400",
|
||||
)
|
||||
```
|
||||
|
||||
Important:
|
||||
|
||||
- the checked-in example currently uses the `1.3B` base model until you edit it
|
||||
- do not load the 14B QAT weights on top of the `1.3B` base model; the weights
|
||||
and model config will not match
|
||||
|
||||
### Step 4. Run the inference example
|
||||
|
||||
```bash
|
||||
python examples/inference/optimizations/attn_qat_inference_example.py
|
||||
```
|
||||
|
||||
Generated videos are written to `video_samples/` by default.
|
||||
|
||||
## Backend Overview
|
||||
|
||||
| Backend | Best for | Package requirement | Primary kernel location |
|
||||
|---------|----------|---------------------|-------------------------|
|
||||
| `ATTN_QAT_TRAIN` | finetuning, training-time validation, reproducing the training path | `fastvideo_kernel` | `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` |
|
||||
| `ATTN_QAT_INFER` | standalone inference with the dedicated CUDA kernel | `attn_qat_infer` from the in-repo `fastvideo-kernel` checkout | `fastvideo-kernel/attn_qat_infer/` |
|
||||
|
||||
FastVideo routes backend selection through:
|
||||
|
||||
- `fastvideo/envs.py`
|
||||
- `fastvideo/platforms/cuda.py`
|
||||
- `fastvideo/attention/backends/attn_qat_train.py`
|
||||
- `fastvideo/attention/backends/attn_qat_infer.py`
|
||||
|
||||
The legacy training pipeline also contains explicit Attention QAT integration:
|
||||
|
||||
- `fastvideo/training/training_pipeline.py`
|
||||
|
||||
That pipeline forces generator loading through `ATTN_QAT_TRAIN` when
|
||||
`FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN` or `--generator_4bit_attn` is
|
||||
enabled.
|
||||
|
||||
## Inference Workflows
|
||||
|
||||
For standalone inference, prefer `ATTN_QAT_INFER` when the CUDA kernel is
|
||||
available. Use `ATTN_QAT_TRAIN` for inference only if you intentionally want to
|
||||
exercise the training-side attention path for debugging or parity checks.
|
||||
|
||||
### Minimal Python example
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1,
|
||||
)
|
||||
|
||||
generator.generate_video(
|
||||
"A cinematic close-up of rain on a neon street at night.",
|
||||
output_path="video_samples",
|
||||
save_video=True,
|
||||
)
|
||||
```
|
||||
|
||||
### Loading custom safetensors during inference
|
||||
|
||||
FastVideo supports loading custom transformer weights through
|
||||
`init_weights_from_safetensors`.
|
||||
|
||||
This value can point to either:
|
||||
|
||||
- a directory containing one or more `.safetensors` files
|
||||
- a single `.safetensors` file
|
||||
|
||||
For Wan 2.1 14B QAT inference, the common pattern is:
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
init_weights_from_safetensors="checkpoints/14B_qat_400",
|
||||
)
|
||||
```
|
||||
|
||||
### CLI example
|
||||
|
||||
You can also force the backend from the command line:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
|
||||
fastvideo generate \
|
||||
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
|
||||
--num-gpus 1 \
|
||||
--sp-size 1 \
|
||||
--tp-size 1 \
|
||||
--height 480 \
|
||||
--width 832 \
|
||||
--num-frames 77 \
|
||||
--num-inference-steps 50 \
|
||||
--guidance-scale 6.0 \
|
||||
--prompt "A cinematic close-up of rain on a neon street at night." \
|
||||
--output-path outputs_video/
|
||||
```
|
||||
|
||||
If you want to use custom QAT transformer weights from the CLI, pass the same
|
||||
custom weight override that the Python API uses:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
|
||||
fastvideo generate \
|
||||
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
|
||||
--init-weights-from-safetensors checkpoints/14B_qat_400 \
|
||||
--num-gpus 1 \
|
||||
--output-path outputs_video/ \
|
||||
--prompt "A cinematic close-up of rain on a neon street at night."
|
||||
```
|
||||
|
||||
## Training Workflows
|
||||
|
||||
Today the checked-in Attention QAT training launchers use the legacy training
|
||||
pipeline in `fastvideo/training/wan_training_pipeline.py`.
|
||||
|
||||
### Ready-made launchers
|
||||
|
||||
Use the provided SLURM scripts directly:
|
||||
|
||||
```bash
|
||||
sbatch examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh
|
||||
sbatch examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh
|
||||
```
|
||||
|
||||
Both scripts already set:
|
||||
|
||||
```bash
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
```
|
||||
|
||||
Before launching, update the script-local values that depend on your
|
||||
environment:
|
||||
|
||||
- `WANDB_API_KEY`
|
||||
- `MODEL_PATH`
|
||||
- `DATA_DIR`
|
||||
- `VALIDATION_DATASET_FILE`
|
||||
- output directory and SLURM resource requests
|
||||
|
||||
### What the launchers run
|
||||
|
||||
The training scripts eventually invoke:
|
||||
|
||||
```bash
|
||||
torchrun fastvideo/training/wan_training_pipeline.py ...
|
||||
```
|
||||
|
||||
If you are adapting the workflow to your own cluster or running outside SLURM,
|
||||
the main Attention QAT requirement is still:
|
||||
|
||||
```bash
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
```
|
||||
|
||||
Then launch the normal Wan training pipeline with your preferred `torchrun`
|
||||
arguments and training flags.
|
||||
|
||||
## Where The Code Lives
|
||||
|
||||
Use these paths when you want to trace or modify the Attention QAT flow:
|
||||
|
||||
| Location | Purpose |
|
||||
|----------|---------|
|
||||
| `fastvideo/attention/backends/attn_qat_train.py` | FastVideo wrapper that imports and calls the Triton training kernel |
|
||||
| `fastvideo/attention/backends/attn_qat_infer.py` | FastVideo wrapper that imports and calls the inference kernel |
|
||||
| `fastvideo-kernel/CMakeLists.txt` | Kernel build definition that compiles the `attn_qat_infer` inference extensions |
|
||||
| `fastvideo/platforms/cuda.py` | Chooses the concrete attention backend at runtime |
|
||||
| `fastvideo/envs.py` | Documents supported `FASTVIDEO_ATTENTION_BACKEND` values |
|
||||
| `fastvideo/training/training_pipeline.py` | Training-time forcing logic for the generator attention backend |
|
||||
| `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` | Triton implementation for `ATTN_QAT_TRAIN` |
|
||||
| `fastvideo-kernel/attn_qat_infer/api.py` | Python API entrypoint for the inference kernel |
|
||||
| `fastvideo-kernel/benchmarks/benchmark_*.py` | Kernel-side benchmark scripts for FlashAttn2, SageAttention3, FP4, and comparison plots |
|
||||
| `fastvideo-kernel/attn_qat_infer/blackwell/api.cu` | CUDA implementation behind `ATTN_QAT_INFER` |
|
||||
| `fastvideo-kernel/tests/test_attn_qat_train.py` | Kernel-level test coverage for the training path |
|
||||
| `examples/inference/optimizations/attn_qat_inference_example.py` | Ready-to-edit inference example for custom Attention QAT weights |
|
||||
| `examples/inference/optimizations/download_14B_qat.sh` | Helper script for downloading the Wan 2.1 14B QAT checkpoint |
|
||||
| `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 1.3B Attention QAT finetune launcher |
|
||||
| `examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 14B Attention QAT finetune launcher |
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- If `ATTN_QAT_TRAIN` fails to import, verify that `fastvideo-kernel` built
|
||||
successfully and exposes `fastvideo_kernel`.
|
||||
- If `ATTN_QAT_INFER` fails to import, verify that the local build exposes the
|
||||
`attn_qat_infer` package.
|
||||
- If the Wan 2.1 14B example fails after you changed only the checkpoint path,
|
||||
make sure you also changed the base model to
|
||||
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
|
||||
- If you hit issues with CPU memory pressure or obscure CUDA argument errors in
|
||||
the example script, try setting `pin_cpu_memory=False`.
|
||||
- If you want a known-safe fallback for debugging, use
|
||||
`FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA`.
|
||||
|
||||
## Related Pages
|
||||
|
||||
- [Attention Overview](../index.md)
|
||||
- [Inference Optimizations](../../inference/optimizations.md)
|
||||
- [Debugging](../../utilities/debugging.md)
|
||||
@@ -5,8 +5,6 @@ FastVideo provides highly optimized custom attention kernels to accelerate video
|
||||
## Supported Kernels
|
||||
|
||||
* **[Video Sparse Attention (VSA)](vsa/index.md)**: Sparse attention mechanism selecting top-k blocks.
|
||||
* **[Attention QAT](attn_qat/index.md)**: Dedicated guide for Attention QAT
|
||||
inference, training, checkpoint loading, and troubleshooting.
|
||||
* **[Sliding Tile Attention (STA)](sta/index.md)**: STA kernel support is kept in
|
||||
`fastvideo-kernel`; full FastVideo STA pipeline workflow is archived in
|
||||
`sta_do_not_delete`.
|
||||
|
||||
@@ -24,7 +24,7 @@ PR push
|
||||
Runs on the PR branch directly
|
||||
│
|
||||
pass ──► Mergify auto-squash-merges to main, branch deleted
|
||||
fail ──► fix the regression, push, and /merge again
|
||||
fail ──► Mergify removes 'ready' label; fix and /merge again
|
||||
```
|
||||
|
||||
---
|
||||
@@ -102,8 +102,8 @@ failing test's output.
|
||||
| Performance Tests | `performance` | 30 min |
|
||||
| API Server Tests | `api_server` | 30 min |
|
||||
|
||||
If a Full Suite test fails, check the Buildkite build log for the failing step's output.
|
||||
Fix the regression, push, and comment `/merge` again to re-trigger.
|
||||
A Full Suite failure removes the `ready` label automatically. A Mergify comment links to
|
||||
the Buildkite build. Fix the regression, push, and comment `/merge` again.
|
||||
|
||||
---
|
||||
|
||||
@@ -129,8 +129,8 @@ Suite passing directly on the PR branch.
|
||||
- No merge conflicts
|
||||
5. If all conditions pass, Mergify squash-merges to `main` automatically. The branch is
|
||||
deleted after merge.
|
||||
6. If the Full Suite fails, the developer fixes the issue, pushes, and comments `/merge`
|
||||
again to re-trigger.
|
||||
6. If the Full Suite fails, Mergify removes the `ready` label and posts a comment linking to
|
||||
the Buildkite build. The developer fixes the issue, pushes, and comments `/merge` again.
|
||||
|
||||
**Merge conditions summary:**
|
||||
|
||||
@@ -279,30 +279,6 @@ Triggers a specific Buildkite test or suite on the current PR branch.
|
||||
| `/test api` | API server integration tests | `api_server` |
|
||||
| `/test full` | Entire Full Suite | all (with `TEST_SCOPE=full`) |
|
||||
| `/test fastcheck` | Entire Fastcheck suite | fastcheck (with `TEST_SCOPE=fastcheck`) |
|
||||
| `/test pre-commit` | Pre-commit checks on PR code | — (runs `ci-precommit.yml` via `workflow_call`) |
|
||||
|
||||
**Re-running failed tests:** When you use `/test <name>` to re-run a specific failed test,
|
||||
the resulting Buildkite check uses the same name as the original (e.g., `/test encoder`
|
||||
creates `buildkite/ci/microscope-encoder-tests`). This overwrites the failed check status.
|
||||
Once all tests in a tier pass, the aggregate status (`fastcheck-passed` or
|
||||
`full-suite-passed`) is automatically updated to `success` by the `ci-aggregate-status.yml`
|
||||
workflow.
|
||||
|
||||
**How aggregate status refresh works:**
|
||||
|
||||
1. `/test <name>` triggers a Buildkite build with `TEST_SCOPE=direct`. The test step uses
|
||||
the same label as its fastcheck/full-suite counterpart, so the resulting GitHub check
|
||||
overwrites the original.
|
||||
2. When the build completes, Buildkite's `notify` posts a `direct-test-completed` commit
|
||||
status. This is the only signal that triggers the aggregate workflow — intermediate step
|
||||
status updates do not trigger it.
|
||||
3. `ci-aggregate-status.yml` fires, calls `getCombinedStatusForRef` to fetch the latest
|
||||
status for every context on that commit (each context returns only its most recent
|
||||
state), groups them by prefix (`microscope-*` → fastcheck, `test-tube-*`/`bar-chart-*`
|
||||
→ full suite), and posts `fastcheck-passed: success` or `full-suite-passed: success` if
|
||||
all entries in the group are `success`.
|
||||
4. Tests that were never triggered (skipped by monorepo-diff) have no status entry and do
|
||||
not block the aggregate.
|
||||
|
||||
---
|
||||
|
||||
@@ -320,7 +296,6 @@ Protected branches (`main`, `master`, `release/*`) are never deleted.
|
||||
| `ci-precommit.yml` | Every push / PR against `main` | Runs pre-commit hooks (yapf, ruff, mypy, codespell, pymarkdown, actionlint, check-filenames) |
|
||||
| `ci-trigger-full-suite.yml` | `ready` label added to a PR | Calls Buildkite API to run Full Suite on the PR branch |
|
||||
| `ci-slash-commands.yml` | PR comment starting with `/merge` or `/test` | Handles slash commands; adds `ready` label or triggers Buildkite |
|
||||
| `ci-aggregate-status.yml` | Any Buildkite commit status update | Checks if all tests in a tier passed; updates `fastcheck-passed` or `full-suite-passed` |
|
||||
| `community-issue-labeler.yml` | Issue opened or edited | Auto-labels issues by keyword matching against title and body |
|
||||
| `community-welcome.yml` | First contribution | Posts a welcome comment for first-time contributors |
|
||||
| `community-stale.yml` | Scheduled | Marks and closes stale issues and PRs |
|
||||
|
||||
@@ -104,9 +104,8 @@ distillation, self-forcing, VSA, VMoBA, performance benchmarks, and API server t
|
||||
8. If all Full Suite tests pass and all merge conditions are met (approval, valid title,
|
||||
pre-commit green, fastcheck green, no draft, no conflicts), Mergify squash-merges to
|
||||
`main` automatically. Your branch is deleted.
|
||||
9. If a Full Suite test fails, check the Buildkite build log for the failing step. Fix the
|
||||
issue, push, and comment `/merge` again. You can also re-run individual failed tests
|
||||
with `/test <name>` — see below.
|
||||
9. If a Full Suite test fails, Mergify removes the `ready` label and posts a comment with a
|
||||
link to the Buildkite build. Fix the issue, push, and comment `/merge` again.
|
||||
|
||||
!!! note
|
||||
Only contributors with write permission to the repository can trigger slash commands.
|
||||
@@ -150,15 +149,10 @@ Comment on your PR to trigger specific tests independently of the auto-merge flo
|
||||
/test vmoba # VMoBA inference tests
|
||||
/test performance # Performance benchmarks
|
||||
/test api # API server integration tests
|
||||
/test pre-commit # Pre-commit checks on PR code
|
||||
```
|
||||
|
||||
The workflow reacts with a 🚀 emoji to confirm the command was received.
|
||||
|
||||
When you re-run an individual test with `/test <name>`, the new result overwrites the
|
||||
original failed check (same Buildkite check name). Once all tests in a tier pass, the
|
||||
`fastcheck-passed` or `full-suite-passed` status is automatically updated.
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
@@ -205,8 +199,9 @@ Mergify removes the `needs-rebase` label automatically once conflicts are resolv
|
||||
|
||||
### Full Suite failed after `/merge`
|
||||
|
||||
The Full Suite found a regression. Check the failing Buildkite step's output for assertion
|
||||
errors or tracebacks.
|
||||
The Full Suite found a regression. Mergify removes the `ready` label and posts a comment
|
||||
linking to the Buildkite build. Check the failing step's output for assertion errors or
|
||||
tracebacks.
|
||||
|
||||
Common causes:
|
||||
|
||||
|
||||
@@ -1,720 +0,0 @@
|
||||
status_definitions:
|
||||
kept: "Public field remains on a public adapter surface with the same meaning."
|
||||
moved: "Public field remains supported but normalizes into a different nested path."
|
||||
profile_owned: "Public field remains supported only through a model/profile-specific surface."
|
||||
compatibility_only: "Legacy public field remains adapter-only during migration and is not part of the canonical typed schema."
|
||||
private_only: "Field should only be handled by private adapters and is not a public FastVideo compatibility promise."
|
||||
internal_only: "Field is runtime/config plumbing and should not be part of the new public typed inference API."
|
||||
|
||||
surfaces:
|
||||
fastvideo_args:
|
||||
moved:
|
||||
model_path: generator.model_path
|
||||
workload_type: generator.pipeline.workload_type
|
||||
distributed_executor_backend: generator.engine.execution_backend
|
||||
trust_remote_code: generator.trust_remote_code
|
||||
revision: generator.revision
|
||||
num_gpus: generator.engine.num_gpus
|
||||
tp_size: generator.engine.parallelism.tp_size
|
||||
sp_size: generator.engine.parallelism.sp_size
|
||||
hsdp_replicate_dim: generator.engine.parallelism.hsdp_replicate_dim
|
||||
hsdp_shard_dim: generator.engine.parallelism.hsdp_shard_dim
|
||||
dist_timeout: generator.engine.parallelism.dist_timeout
|
||||
lora_path: generator.pipeline.components.lora_path
|
||||
dit_cpu_offload: generator.engine.offload.dit
|
||||
use_fsdp_inference: generator.engine.use_fsdp_inference
|
||||
dit_layerwise_offload: generator.engine.offload.dit_layerwise
|
||||
text_encoder_cpu_offload: generator.engine.offload.text_encoder
|
||||
image_encoder_cpu_offload: generator.engine.offload.image_encoder
|
||||
vae_cpu_offload: generator.engine.offload.vae
|
||||
pin_cpu_memory: generator.engine.offload.pin_cpu_memory
|
||||
enable_torch_compile: generator.engine.compile.enabled
|
||||
torch_compile_kwargs: generator.engine.compile.kwargs
|
||||
disable_autocast: generator.engine.disable_autocast
|
||||
enable_stage_verification: generator.engine.enable_stage_verification
|
||||
prompt_txt: request.inputs.prompt_path
|
||||
override_text_encoder_safetensors: generator.pipeline.components.text_encoder_weights
|
||||
override_text_encoder_quant: generator.engine.quantization.text_encoder_quant
|
||||
transformer_quant: generator.engine.quantization.transformer_quant
|
||||
override_transformer_cls_name: generator.pipeline.components.override_transformer_cls_name
|
||||
init_weights_from_safetensors: generator.pipeline.components.transformer_weights
|
||||
init_weights_from_safetensors_2: generator.pipeline.components.transformer_2_weights
|
||||
override_pipeline_cls_name: generator.pipeline.components.override_pipeline_cls_name
|
||||
boundary_ratio: request.sampling.boundary_ratio
|
||||
profile_owned:
|
||||
ltx2_vae_tiling: generator.pipeline.profile_overrides.ltx2.vae_tiling
|
||||
ltx2_vae_spatial_tile_size_in_pixels: generator.pipeline.profile_overrides.ltx2.vae.spatial_tile_size_in_pixels
|
||||
ltx2_vae_spatial_tile_overlap_in_pixels: generator.pipeline.profile_overrides.ltx2.vae.spatial_tile_overlap_in_pixels
|
||||
ltx2_vae_temporal_tile_size_in_frames: generator.pipeline.profile_overrides.ltx2.vae.temporal_tile_size_in_frames
|
||||
ltx2_vae_temporal_tile_overlap_in_frames: generator.pipeline.profile_overrides.ltx2.vae.temporal_tile_overlap_in_frames
|
||||
ltx2_initial_latent_path: request.extensions.ltx2.initial_latent_path
|
||||
compatibility_only:
|
||||
mode: "Legacy multi-mode FastVideoArgs switch; typed inference config should not expose execution mode."
|
||||
inference_mode: "Legacy boolean mirror of mode; kept only through adapters while FastVideoArgs remains."
|
||||
lora_nickname: "Legacy adapter-selection surface pending LoRA API cleanup."
|
||||
lora_target_modules: "Legacy LoRA configuration surface pending dedicated component API."
|
||||
output_type: "Legacy output formatting surface pending GenerationResult cleanup."
|
||||
VSA_sparsity: "Model-specific inference optimization not yet represented in the typed public schema."
|
||||
moba_config_path: "Model-specific MoBA optimization surface not yet represented in the typed public schema."
|
||||
master_port: "Executor/bootstrap compatibility field; not part of the canonical inference schema."
|
||||
private_only:
|
||||
ray_placement_group: "Ray deployment-only field."
|
||||
ray_runtime_env: "Ray deployment-only field."
|
||||
internal_only:
|
||||
pipeline_config: "Legacy internal carrier object."
|
||||
preprocess_config: "Legacy preprocess carrier object."
|
||||
moba_config: "Derived runtime config loaded from moba_config_path."
|
||||
model_paths: "Runtime bookkeeping."
|
||||
model_loaded: "Runtime bookkeeping."
|
||||
|
||||
pipeline_config_base:
|
||||
moved:
|
||||
pipeline_config_path: generator.pipeline.components.pipeline_config_path
|
||||
profile_owned:
|
||||
embedded_cfg_scale: generator.pipeline.profile_overrides.embedded_cfg_scale
|
||||
flow_shift: generator.pipeline.profile_overrides.flow_shift
|
||||
flow_shift_sr: generator.pipeline.profile_overrides.flow_shift_sr
|
||||
is_causal: generator.pipeline.profile_overrides.is_causal
|
||||
vae_tiling: generator.pipeline.profile_overrides.vae_tiling
|
||||
vae_sp: generator.pipeline.profile_overrides.vae_sp
|
||||
dmd_denoising_steps: generator.pipeline.profile_overrides.dmd_denoising_steps
|
||||
ti2v_task: generator.pipeline.profile_overrides.ti2v_task
|
||||
boundary_ratio: generator.pipeline.profile_overrides.boundary_ratio
|
||||
compatibility_only:
|
||||
model_path: "Redundant with generator.model_path."
|
||||
disable_autocast: "Duplicated by generator.engine.disable_autocast during migration."
|
||||
dit_precision: "Precision override pending dedicated typed component precision design."
|
||||
upsampler_precision: "Precision override pending dedicated typed component precision design."
|
||||
vae_precision: "Precision override pending dedicated typed component precision design."
|
||||
image_encoder_precision: "Precision override pending dedicated typed component precision design."
|
||||
text_encoder_precisions: "Precision override pending dedicated typed component precision design."
|
||||
internal_only:
|
||||
dit_config: "Legacy internal component config object."
|
||||
upsampler_config: "Legacy internal component config object."
|
||||
vae_config: "Legacy internal component config object."
|
||||
image_encoder_config: "Legacy internal component config object."
|
||||
text_encoder_configs: "Legacy internal component config object."
|
||||
preprocess_text_funcs: "Internal text preprocessing hooks."
|
||||
postprocess_text_funcs: "Internal text postprocessing hooks."
|
||||
|
||||
pipeline_config_extensions:
|
||||
profile_owned:
|
||||
conditioning_strategy:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
max_num_conditional_frames:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
min_num_conditional_frames:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
sigma_conditional:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
sigma_data:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
state_ch:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
state_t:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
text_encoder_class:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.cosmos.CosmosConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
autoregressive_chunk_frames:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
autoregressive_overlap_frames:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
cfg_behavior:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
default_camera_rotation:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
default_movement_distance:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
default_negative_prompt:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
default_trajectory_type:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
filter_points_threshold:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
fps:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
frame_buffer_max:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
moge_model_name:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
noise_aug_strength:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
num_frames:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
offload_moge_after_depth:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
use_moge_depth:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
video_resolution:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
text_encoder_crop_start:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V480PStepDistilledConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V720PConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15SR1080PConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V480PConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V720PConfig
|
||||
- fastvideo.configs.pipelines.hyworld.HYWorldConfig
|
||||
- fastvideo.configs.pipelines.hyworld.Hunyuan15T2V480PConfig
|
||||
text_encoder_max_lengths:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V480PStepDistilledConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15I2V720PConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15SR1080PConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V480PConfig
|
||||
- fastvideo.configs.pipelines.hunyuan15.Hunyuan15T2V720PConfig
|
||||
- fastvideo.configs.pipelines.hyworld.HYWorldConfig
|
||||
- fastvideo.configs.pipelines.hyworld.Hunyuan15T2V480PConfig
|
||||
precision:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
|
||||
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.wan.MatrixGameBaseI2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.SelfForcingWan2_2_T2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.SelfForcingWanT2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WANV2VConfig
|
||||
- fastvideo.configs.pipelines.wan.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.wan.Wan2_2_T2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.wan.Wan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.wan.WanI2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanI2V720PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V720PConfig
|
||||
warp_denoising_step:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
|
||||
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.wan.MatrixGameBaseI2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.SelfForcingWan2_2_T2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.SelfForcingWanT2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WANV2VConfig
|
||||
- fastvideo.configs.pipelines.wan.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.wan.Wan2_2_T2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.wan.Wan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.wan.WanI2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanI2V720PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V720PConfig
|
||||
bsa_cdf_threshold:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_chunk_k:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_chunk_q:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_params:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_sparsity:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enable_bsa:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enable_kv_cache:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enhance_hf:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
offload_kv_cache:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
t_thresh:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
use_distill:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
scheduler_arch:
|
||||
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
|
||||
text_encoder_archs:
|
||||
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
|
||||
tokenizer_archs:
|
||||
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
|
||||
transformer_arch:
|
||||
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
|
||||
vae_arch:
|
||||
sources: [fastvideo.configs.pipelines.sd35.SD35Config]
|
||||
expand_timesteps:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.wan.Wan2_2_TI2V_5B_Config
|
||||
context_noise:
|
||||
sources: [fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig]
|
||||
num_frames_per_block:
|
||||
sources: [fastvideo.configs.pipelines.wan.MatrixGameI2V480PConfig]
|
||||
compatibility_only:
|
||||
batch_size: "Gen3C inference-only tuning field pending typed batching design."
|
||||
gradient_checkpointing: "Gen3C inference-only compatibility field pending typed batching design."
|
||||
guidance_scale: "Gen3C pipeline-level default pending profile/default-request cleanup."
|
||||
num_inference_steps: "Gen3C pipeline-level default pending profile/default-request cleanup."
|
||||
internal_only:
|
||||
audio_decoder_config: "Legacy internal component config object."
|
||||
audio_decoder_precision: "Precision override pending dedicated component precision design."
|
||||
vocoder_config: "Legacy internal component config object."
|
||||
vocoder_precision: "Precision override pending dedicated component precision design."
|
||||
|
||||
sampling_param_base:
|
||||
moved:
|
||||
image_path: request.inputs.image_path
|
||||
pil_image: request.inputs.pil_image
|
||||
video_path: request.inputs.video_path
|
||||
mouse_cond: request.inputs.mouse_cond
|
||||
keyboard_cond: request.inputs.keyboard_cond
|
||||
grid_sizes: request.inputs.grid_sizes
|
||||
pose: request.inputs.pose
|
||||
c2ws_plucker_emb: request.inputs.c2ws_plucker_emb
|
||||
refine_from: request.inputs.refine_from
|
||||
stage1_video: request.inputs.stage1_video
|
||||
prompt: request.prompt
|
||||
negative_prompt: request.negative_prompt
|
||||
prompt_path: request.inputs.prompt_path
|
||||
output_path: request.output.output_path
|
||||
output_video_name: request.output.output_video_name
|
||||
num_videos_per_prompt: request.sampling.num_videos_per_prompt
|
||||
seed: request.sampling.seed
|
||||
num_frames: request.sampling.num_frames
|
||||
height: request.sampling.height
|
||||
width: request.sampling.width
|
||||
height_sr: request.sampling.height_sr
|
||||
width_sr: request.sampling.width_sr
|
||||
fps: request.sampling.fps
|
||||
num_inference_steps: request.sampling.num_inference_steps
|
||||
num_inference_steps_sr: request.sampling.num_inference_steps_sr
|
||||
guidance_scale: request.sampling.guidance_scale
|
||||
guidance_scale_2: request.sampling.guidance_scale_2
|
||||
guidance_rescale: request.sampling.guidance_rescale
|
||||
boundary_ratio: request.sampling.boundary_ratio
|
||||
sigmas: request.sampling.sigmas
|
||||
enable_teacache: request.runtime.enable_teacache
|
||||
save_video: request.output.save_video
|
||||
return_frames: request.output.return_frames
|
||||
return_trajectory_latents: request.runtime.return_trajectory_latents
|
||||
return_trajectory_decoded: request.runtime.return_trajectory_decoded
|
||||
profile_owned:
|
||||
t_thresh: request.stage_overrides.refine.t_thresh
|
||||
spatial_refine_only: request.stage_overrides.refine.spatial_refine_only
|
||||
num_cond_frames: request.stage_overrides.refine.num_cond_frames
|
||||
trajectory_type: request.extensions.gen3c.trajectory_type
|
||||
movement_distance: request.extensions.gen3c.movement_distance
|
||||
camera_rotation: request.extensions.gen3c.camera_rotation
|
||||
internal_only:
|
||||
data_type: "Derived from the request shape and not a public input."
|
||||
|
||||
sampling_param_extensions:
|
||||
moved: {}
|
||||
profile_owned:
|
||||
action_list:
|
||||
target: request.extensions.hunyuangamecraft.action_list
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
action_speed_list:
|
||||
target: request.extensions.hunyuangamecraft.action_speed_list
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
camera_states:
|
||||
target: request.extensions.hunyuangamecraft.camera_states
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
camera_trajectory:
|
||||
target: request.extensions.hunyuangamecraft.camera_trajectory
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
conditioning_mask:
|
||||
target: request.extensions.hunyuangamecraft.conditioning_mask
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
gt_latents:
|
||||
target: request.extensions.hunyuangamecraft.gt_latents
|
||||
sources:
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraftSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft65FrameSamplingParam
|
||||
- fastvideo.configs.sample.hunyuangamecraft.HunyuanGameCraft129FrameSamplingParam
|
||||
prompt_attention_mask:
|
||||
target: request.extensions.hyworld.prompt_attention_mask
|
||||
sources: [fastvideo.configs.sample.hyworld.HYWorld_SamplingParam]
|
||||
negative_attention_mask:
|
||||
target: request.extensions.hyworld.negative_attention_mask
|
||||
sources: [fastvideo.configs.sample.hyworld.HYWorld_SamplingParam]
|
||||
ltx2_cfg_scale_audio:
|
||||
target: request.extensions.ltx2.cfg_scale_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_cfg_scale_video:
|
||||
target: request.extensions.ltx2.cfg_scale_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_modality_scale_audio:
|
||||
target: request.extensions.ltx2.modality_scale_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_modality_scale_video:
|
||||
target: request.extensions.ltx2.modality_scale_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_rescale_scale:
|
||||
target: request.extensions.ltx2.rescale_scale
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_blocks_audio:
|
||||
target: request.extensions.ltx2.stg_blocks_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_blocks_video:
|
||||
target: request.extensions.ltx2.stg_blocks_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_scale_audio:
|
||||
target: request.extensions.ltx2.stg_scale_audio
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
ltx2_stg_scale_video:
|
||||
target: request.extensions.ltx2.stg_scale_video
|
||||
sources: [fastvideo.configs.sample.ltx2.LTX2BaseSamplingParam]
|
||||
|
||||
openai_image_request:
|
||||
kept:
|
||||
model: "HTTP adapter model-routing field."
|
||||
response_format: "HTTP adapter response formatting field."
|
||||
output_format: "HTTP adapter output-format field."
|
||||
background: "HTTP adapter output-format field."
|
||||
quality: "Compatibility field currently accepted by the adapter."
|
||||
style: "Compatibility field currently accepted by the adapter."
|
||||
user: "Compatibility field currently accepted by the adapter."
|
||||
moved:
|
||||
prompt: request.prompt
|
||||
n: request.sampling.num_videos_per_prompt
|
||||
size:
|
||||
target: request.sampling.width,height
|
||||
note: "Adapter parses OpenAI size strings as WIDTHxHEIGHT and forwards width then height."
|
||||
num_inference_steps: request.sampling.num_inference_steps
|
||||
guidance_scale: request.sampling.guidance_scale
|
||||
true_cfg_scale: request.sampling.true_cfg_scale
|
||||
seed: request.sampling.seed
|
||||
negative_prompt: request.negative_prompt
|
||||
enable_teacache: request.runtime.enable_teacache
|
||||
|
||||
openai_video_request:
|
||||
kept:
|
||||
model: "HTTP adapter model-routing field."
|
||||
moved:
|
||||
prompt: request.prompt
|
||||
input_reference: request.inputs.image_path
|
||||
reference_url: request.inputs.image_path
|
||||
size:
|
||||
target: request.sampling.width,height
|
||||
note: "Adapter parses OpenAI size strings as WIDTHxHEIGHT and forwards width then height."
|
||||
fps: request.sampling.fps
|
||||
num_frames: request.sampling.num_frames
|
||||
seed: request.sampling.seed
|
||||
num_inference_steps: request.sampling.num_inference_steps
|
||||
guidance_scale: request.sampling.guidance_scale
|
||||
guidance_scale_2: request.sampling.guidance_scale_2
|
||||
true_cfg_scale: request.sampling.true_cfg_scale
|
||||
negative_prompt: request.negative_prompt
|
||||
enable_teacache: request.runtime.enable_teacache
|
||||
output_path: request.output.output_path
|
||||
compatibility_only:
|
||||
seconds:
|
||||
target: request.sampling.num_frames
|
||||
note: "HTTP adapter duration convenience field. If num_frames is omitted, the adapter computes num_frames = fps * seconds."
|
||||
|
||||
cli:
|
||||
notes:
|
||||
- "CLI parity is checked against the actual generate/serve parser dest sets."
|
||||
- "The inventory tracks parser dest names, excluding argparse's implicit help action."
|
||||
generate:
|
||||
explicit_local_fields:
|
||||
- config
|
||||
expected_dests:
|
||||
- VSA_sparsity
|
||||
- boundary_ratio
|
||||
- bsa_cdf_threshold
|
||||
- bsa_chunk_k
|
||||
- bsa_chunk_q
|
||||
- bsa_sparsity
|
||||
- config
|
||||
- disable_autocast
|
||||
- dist_timeout
|
||||
- distributed_executor_backend
|
||||
- dit_config.prefix
|
||||
- dit_config.quant_config
|
||||
- dit_cpu_offload
|
||||
- dit_layerwise_offload
|
||||
- dit_precision
|
||||
- dmd_denoising_steps
|
||||
- embedded_cfg_scale
|
||||
- enable_bsa
|
||||
- enable_stage_verification
|
||||
- enable_torch_compile
|
||||
- flow_shift
|
||||
- fps
|
||||
- guidance_rescale
|
||||
- guidance_scale
|
||||
- height
|
||||
- hsdp_replicate_dim
|
||||
- hsdp_shard_dim
|
||||
- image_encoder_cpu_offload
|
||||
- image_encoder_precision
|
||||
- image_path
|
||||
- inference_mode
|
||||
- init_weights_from_safetensors
|
||||
- init_weights_from_safetensors_2
|
||||
- lora_nickname
|
||||
- lora_path
|
||||
- lora_target_modules
|
||||
- ltx2_initial_latent_path
|
||||
- ltx2_vae_spatial_tile_overlap_in_pixels
|
||||
- ltx2_vae_spatial_tile_size_in_pixels
|
||||
- ltx2_vae_temporal_tile_overlap_in_frames
|
||||
- ltx2_vae_temporal_tile_size_in_frames
|
||||
- ltx2_vae_tiling
|
||||
- master_port
|
||||
- moba_config_path
|
||||
- mode
|
||||
- model_path
|
||||
- negative_prompt
|
||||
- num_cond_frames
|
||||
- num_frames
|
||||
- num_gpus
|
||||
- num_inference_steps
|
||||
- num_videos_per_prompt
|
||||
- output_path
|
||||
- output_type
|
||||
- output_video_name
|
||||
- override_pipeline_cls_name
|
||||
- override_text_encoder_quant
|
||||
- override_text_encoder_safetensors
|
||||
- override_transformer_cls_name
|
||||
- pin_cpu_memory
|
||||
- pipeline_config_path
|
||||
- preprocess.dataloader_num_workers
|
||||
- preprocess.dataset_output_dir
|
||||
- preprocess.dataset_path
|
||||
- preprocess.dataset_type
|
||||
- preprocess.do_temporal_sample
|
||||
- preprocess.drop_short_ratio
|
||||
- preprocess.flush_frequency
|
||||
- preprocess.max_height
|
||||
- preprocess.max_width
|
||||
- preprocess.model_path
|
||||
- preprocess.num_frames
|
||||
- preprocess.preprocess_video_batch_size
|
||||
- preprocess.samples_per_file
|
||||
- preprocess.seed
|
||||
- preprocess.speed_factor
|
||||
- preprocess.train_fps
|
||||
- preprocess.training_cfg_rate
|
||||
- preprocess.video_length_tolerance_range
|
||||
- preprocess.video_loader_type
|
||||
- preprocess.with_audio
|
||||
- prompt
|
||||
- prompt_path
|
||||
- prompt_txt
|
||||
- refine_from
|
||||
- return_frames
|
||||
- return_trajectory_decoded
|
||||
- return_trajectory_latents
|
||||
- revision
|
||||
- save_video
|
||||
- seed
|
||||
- sp_size
|
||||
- spatial_refine_only
|
||||
- t_thresh
|
||||
- text_encoder_configs
|
||||
- text_encoder_cpu_offload
|
||||
- text_encoder_precisions
|
||||
- torch_compile_kwargs
|
||||
- transformer_quant
|
||||
- tp_size
|
||||
- trust_remote_code
|
||||
- use_fsdp_inference
|
||||
- vae_config.blend_num_frames
|
||||
- vae_config.load_decoder
|
||||
- vae_config.load_encoder
|
||||
- vae_config.tile_sample_min_height
|
||||
- vae_config.tile_sample_min_num_frames
|
||||
- vae_config.tile_sample_min_width
|
||||
- vae_config.tile_sample_stride_height
|
||||
- vae_config.tile_sample_stride_num_frames
|
||||
- vae_config.tile_sample_stride_width
|
||||
- vae_config.use_parallel_tiling
|
||||
- vae_config.use_temporal_tiling
|
||||
- vae_config.use_tiling
|
||||
- vae_cpu_offload
|
||||
- vae_precision
|
||||
- vae_sp
|
||||
- vae_tiling
|
||||
- video_path
|
||||
- width
|
||||
- workload_type
|
||||
serve:
|
||||
explicit_local_fields:
|
||||
- config
|
||||
- host
|
||||
- output_dir
|
||||
- port
|
||||
expected_dests:
|
||||
- VSA_sparsity
|
||||
- bsa_cdf_threshold
|
||||
- bsa_chunk_k
|
||||
- bsa_chunk_q
|
||||
- bsa_sparsity
|
||||
- config
|
||||
- disable_autocast
|
||||
- dist_timeout
|
||||
- distributed_executor_backend
|
||||
- dit_config.prefix
|
||||
- dit_config.quant_config
|
||||
- dit_cpu_offload
|
||||
- dit_layerwise_offload
|
||||
- dit_precision
|
||||
- dmd_denoising_steps
|
||||
- embedded_cfg_scale
|
||||
- enable_bsa
|
||||
- enable_stage_verification
|
||||
- enable_torch_compile
|
||||
- flow_shift
|
||||
- host
|
||||
- hsdp_replicate_dim
|
||||
- hsdp_shard_dim
|
||||
- image_encoder_cpu_offload
|
||||
- image_encoder_precision
|
||||
- inference_mode
|
||||
- init_weights_from_safetensors
|
||||
- init_weights_from_safetensors_2
|
||||
- lora_nickname
|
||||
- lora_path
|
||||
- lora_target_modules
|
||||
- ltx2_initial_latent_path
|
||||
- ltx2_vae_spatial_tile_overlap_in_pixels
|
||||
- ltx2_vae_spatial_tile_size_in_pixels
|
||||
- ltx2_vae_temporal_tile_overlap_in_frames
|
||||
- ltx2_vae_temporal_tile_size_in_frames
|
||||
- ltx2_vae_tiling
|
||||
- master_port
|
||||
- mode
|
||||
- model_path
|
||||
- num_gpus
|
||||
- output_dir
|
||||
- output_type
|
||||
- override_pipeline_cls_name
|
||||
- override_text_encoder_quant
|
||||
- override_text_encoder_safetensors
|
||||
- override_transformer_cls_name
|
||||
- pin_cpu_memory
|
||||
- pipeline_config_path
|
||||
- port
|
||||
- preprocess.dataloader_num_workers
|
||||
- preprocess.dataset_output_dir
|
||||
- preprocess.dataset_path
|
||||
- preprocess.dataset_type
|
||||
- preprocess.do_temporal_sample
|
||||
- preprocess.drop_short_ratio
|
||||
- preprocess.flush_frequency
|
||||
- preprocess.max_height
|
||||
- preprocess.max_width
|
||||
- preprocess.model_path
|
||||
- preprocess.num_frames
|
||||
- preprocess.preprocess_video_batch_size
|
||||
- preprocess.samples_per_file
|
||||
- preprocess.seed
|
||||
- preprocess.speed_factor
|
||||
- preprocess.train_fps
|
||||
- preprocess.training_cfg_rate
|
||||
- preprocess.video_length_tolerance_range
|
||||
- preprocess.video_loader_type
|
||||
- preprocess.with_audio
|
||||
- prompt_txt
|
||||
- revision
|
||||
- sp_size
|
||||
- text_encoder_cpu_offload
|
||||
- text_encoder_precisions
|
||||
- torch_compile_kwargs
|
||||
- transformer_quant
|
||||
- tp_size
|
||||
- trust_remote_code
|
||||
- use_fsdp_inference
|
||||
- vae_config.blend_num_frames
|
||||
- vae_config.load_decoder
|
||||
- vae_config.load_encoder
|
||||
- vae_config.tile_sample_min_height
|
||||
- vae_config.tile_sample_min_num_frames
|
||||
- vae_config.tile_sample_min_width
|
||||
- vae_config.tile_sample_stride_height
|
||||
- vae_config.tile_sample_stride_num_frames
|
||||
- vae_config.tile_sample_stride_width
|
||||
- vae_config.use_parallel_tiling
|
||||
- vae_config.use_temporal_tiling
|
||||
- vae_config.use_tiling
|
||||
- vae_cpu_offload
|
||||
- vae_precision
|
||||
- vae_sp
|
||||
- vae_tiling
|
||||
- workload_type
|
||||
@@ -167,11 +167,6 @@ How this maps to FastVideo:
|
||||
|
||||
- Attention backends live in `fastvideo/attention/` and can be selected via
|
||||
`FASTVIDEO_ATTENTION_BACKEND`.
|
||||
- SageAttention3 is split into two selectable backends:
|
||||
`SAGE_ATTN_THREE` for the regular upstream package and
|
||||
`ATTN_QAT_INFER` for the FastVideoKernel-backed inference variant.
|
||||
- `ATTN_QAT_TRAIN` is a separate FastVideoKernel Triton backend for the QAT attention
|
||||
path.
|
||||
- `LocalAttention` is used for cross-attention and most attention layers.
|
||||
- `DistributedAttention` is used for full-sequence self-attention in the DiT.
|
||||
- Tensor-parallel layers live in `fastvideo/layers/`.
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
# GEN3C: 3D-Informed Camera-Controlled Video Generation
|
||||
|
||||
[GEN3C](https://arxiv.org/abs/2503.03751) is NVIDIA's Cosmos-7B-based video model for camera-controlled generation from a single image. The FastVideo integration supports the GEN3C I2V workflow, including 3D cache conditioning and tokenizer-based conditioning latents.
|
||||
|
||||
## Key Features
|
||||
|
||||
- **Camera trajectory control**: `left/right/up/down/zoom_in/zoom_out/clockwise/counterclockwise`
|
||||
- **3D cache conditioning**: depth prediction -> point cloud cache -> forward warping -> latent conditioning
|
||||
- **Single-image to video generation**: 121-frame generation with camera motion
|
||||
- **Official raw checkpoint conversion**: `model.pt` -> Diffusers/FastVideo layout
|
||||
|
||||
## Model Sources
|
||||
|
||||
- Official raw checkpoint (not Diffusers): `nvidia/GEN3C-Cosmos-7B`
|
||||
- Diffusers-format checkpoint: `FastVideo/GEN3C-Cosmos-7B-Diffusers`
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Install MoGe:
|
||||
|
||||
```bash
|
||||
pip install git+https://github.com/microsoft/MoGe.git
|
||||
```
|
||||
|
||||
- If you hit `ImportError: libGL.so.1` (common on Ubuntu/headless nodes), you can try installing OpenCV runtime libs:
|
||||
|
||||
```bash
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libgl1 libglib2.0-0 libsm6 libxext6 libxrender1
|
||||
```
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Option A: Use Diffusers-format weights directly
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_gen3c.py \
|
||||
--model_path FastVideo/GEN3C-Cosmos-7B-Diffusers \
|
||||
--image_path /path/to/input.png \
|
||||
--prompt "" \
|
||||
--trajectory left \
|
||||
--movement_distance 0.3 \
|
||||
--camera_rotation center_facing \
|
||||
--num_inference_steps 35 \
|
||||
--guidance_scale 1.0 \
|
||||
--output_path outputs_video/gen3c_output.mp4
|
||||
```
|
||||
|
||||
### Option B: Convert official raw checkpoint locally
|
||||
|
||||
1. Download:
|
||||
|
||||
```bash
|
||||
huggingface-cli download nvidia/GEN3C-Cosmos-7B --local-dir official_weights/GEN3C-Cosmos-7B
|
||||
```
|
||||
|
||||
1. Convert:
|
||||
|
||||
```bash
|
||||
python scripts/checkpoint_conversion/convert_gen3c_to_fastvideo.py \
|
||||
--source official_weights/GEN3C-Cosmos-7B/model.pt \
|
||||
--output converted_weights/GEN3C-Cosmos-7B
|
||||
```
|
||||
|
||||
1. Run:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_gen3c.py \
|
||||
--model_path converted_weights/GEN3C-Cosmos-7B \
|
||||
--image_path /path/to/input.png \
|
||||
--prompt "" \
|
||||
--trajectory left \
|
||||
--movement_distance 0.3 \
|
||||
--camera_rotation center_facing \
|
||||
--num_inference_steps 35 \
|
||||
--guidance_scale 1.0 \
|
||||
--output_path outputs_video/gen3c_output.mp4
|
||||
```
|
||||
|
||||
## FastVideo Defaults
|
||||
|
||||
GEN3C defaults in FastVideo:
|
||||
|
||||
- `height=704`, `width=1280`
|
||||
- `num_frames=121`
|
||||
- `num_inference_steps=35`
|
||||
- `guidance_scale=1.0`
|
||||
- `fps=24`
|
||||
|
||||
These values are defined in:
|
||||
|
||||
- `fastvideo/configs/sample/gen3c.py`
|
||||
- `fastvideo/configs/pipelines/gen3c.py`
|
||||
|
||||
and align with the official GEN3C inference defaults in:
|
||||
|
||||
- `tmp/GEN3C/cosmos_predict1/diffusion/inference/inference_utils.py`
|
||||
|
||||
## Scheduler Note
|
||||
|
||||
The converted GEN3C Diffusers layout may include a FlowMatch scheduler config, but GEN3C denoising uses EDM preconditioning behavior. FastVideo's GEN3C pipeline enforces an EDM scheduler at runtime for parity with official inference behavior.
|
||||
|
||||
Implementation path:
|
||||
|
||||
- `fastvideo/pipelines/basic/gen3c/gen3c_pipeline.py`
|
||||
|
||||
## 3D Cache Conditioning Path
|
||||
|
||||
FastVideo GEN3C conditioning stage performs:
|
||||
|
||||
1. MoGe depth estimation from input image
|
||||
2. 3D cache initialization
|
||||
3. Camera trajectory generation
|
||||
4. Forward rendering of warped frames + masks
|
||||
5. VAE/tokenizer encoding of conditioning buffers
|
||||
6. Denoising with condition mask + condition pose channels
|
||||
|
||||
Main implementation:
|
||||
|
||||
- `fastvideo/pipelines/basic/gen3c/gen3c_pipeline.py`
|
||||
- `fastvideo/pipelines/basic/gen3c/cache_3d.py`
|
||||
- `fastvideo/pipelines/basic/gen3c/depth_estimation.py`
|
||||
- `fastvideo/models/vaes/gen3c_tokenizer_vae.py`
|
||||
|
||||
## References
|
||||
|
||||
- [GEN3C Paper](https://arxiv.org/abs/2503.03751)
|
||||
- [Official Repository](https://github.com/nv-tlabs/GEN3C)
|
||||
- [Official Checkpoint (raw)](https://huggingface.co/nvidia/GEN3C-Cosmos-7B)
|
||||
@@ -107,8 +107,6 @@ If you encounter CUDA out of memory errors:
|
||||
(single GPU) or `use_fsdp_inference=True` (multi-GPU)
|
||||
- Try a smaller model or use distilled versions
|
||||
- Use `num_gpus` > 1 if multiple GPUs are available
|
||||
- Try enabling FSDP inference with `use_fsdp_inference=True` (may slow down generation)
|
||||
- Try enabling DiT layerwise offload with `dit_layerwise_offload=True` (now only a few models support this, but may introduce less overhead than FSDP)
|
||||
|
||||
### Slow Generation
|
||||
|
||||
|
||||
@@ -21,8 +21,6 @@ This page describes the various options for speeding up generation times in Fast
|
||||
- Video Sparse Attention: `FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN`
|
||||
- Sage Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN`
|
||||
- Sage Attention 3: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN_THREE`
|
||||
- Attn QAT Infer: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER`
|
||||
- Attn QAT Train: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN`
|
||||
- Video MoBA Attention: `FASTVIDEO_ATTENTION_BACKEND=VMOBA_ATTN`
|
||||
- Sparse Linear Attention: `FASTVIDEO_ATTENTION_BACKEND=SLA_ATTN`
|
||||
- SageSLA Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_SLA_ATTN`
|
||||
@@ -105,14 +103,6 @@ python setup.py install # or pip install -e .
|
||||
|
||||
### Sage Attention 3
|
||||
|
||||
FastVideo now exposes two SageAttention3-compatible backends with distinct
|
||||
environment variable values:
|
||||
|
||||
- `SAGE_ATTN_THREE`: the regular upstream SageAttention3 backend imported from
|
||||
the `sageattn3` package.
|
||||
- `ATTN_QAT_INFER`: the inference CUDA-kernel backend imported from the
|
||||
in-repo `attn_qat_infer` package.
|
||||
|
||||
**`SAGE_ATTN_THREE`**
|
||||
|
||||
[SageAttention 3](https://github.com/thu-ml/SageAttention/tree/main/sageattention3_blackwell) is an advanced attention mechanism that leverages FP4 quantization and Blackwell GPU Tensor Cores for significant performance improvements.
|
||||
@@ -127,53 +117,6 @@ Note that Sage Attention 3 requires `python>=3.13`, `torch>=2.8.0`, `CUDA >=12.8
|
||||
|
||||
To use Sage Attention 3 in FastVideo, follow the `README.md` in the linked repository to install the package from source.
|
||||
|
||||
### Attn QAT Infer
|
||||
|
||||
**`ATTN_QAT_INFER`**
|
||||
|
||||
This backend uses the `attn_qat_infer` implementation that lives in the
|
||||
`fastvideo-kernel` repository alongside the `fastvideo_kernel` Triton kernels.
|
||||
Use this backend when you want to run the dedicated FP4 inference CUDA kernel
|
||||
directly during inference.
|
||||
|
||||
For the full Attention QAT guide, including Wan 2.1 14B checkpoint download,
|
||||
example editing steps, training launchers, and troubleshooting, see
|
||||
[Attention QAT](../attention/attn_qat/index.md).
|
||||
|
||||
This backend currently assumes access to the in-repo `fastvideo-kernel`
|
||||
checkout or an equivalent editable/source install that exposes:
|
||||
|
||||
- `attn_qat_infer`
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
```
|
||||
|
||||
### QAT Attention
|
||||
|
||||
**`ATTN_QAT_TRAIN`**
|
||||
|
||||
This backend uses the FastVideoKernel Triton attention implementation from
|
||||
`fastvideo_kernel.triton_kernels.attn_qat_train`. Use it when you specifically
|
||||
want the training-oriented Triton attention path rather than the
|
||||
`attn_qat_infer` CUDA kernel path.
|
||||
|
||||
The dedicated [Attention QAT](../attention/attn_qat/index.md) page covers when
|
||||
to use `ATTN_QAT_TRAIN` versus `ATTN_QAT_INFER`, the ready-made training
|
||||
launchers, and the end-to-end Wan 2.1 14B inference workflow.
|
||||
|
||||
This backend currently assumes access to an install that exposes:
|
||||
|
||||
- `fastvideo_kernel`
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_TRAIN"
|
||||
```
|
||||
|
||||
### V-MoBA / SLA / SageSLA
|
||||
|
||||
These backends are model-specific and require the corresponding kernels and
|
||||
|
||||
@@ -73,7 +73,6 @@ pipeline initialization and sampling.
|
||||
| Matrix Game 2.0 Base | `FastVideo/Matrix-Game-2.0-Base-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 GTA | `FastVideo/Matrix-Game-2.0-GTA-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 TempleRun | `FastVideo/Matrix-Game-2.0-TempleRun-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| GEN3C Cosmos 7B | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | 704px1280p | ❌ | ❌ | ❌ | ⭕ | ⭕ |
|
||||
|
||||
**Note**: Wan2.2 TI2V 5B has some quality issues when performing I2V generation. We are working on fixing this issue.
|
||||
|
||||
@@ -86,11 +85,6 @@ The authoritative source for model-ID recognition is
|
||||
`fastvideo/registry.py`. If a model ID is registered there, FastVideo can
|
||||
resolve default pipeline and sampling configuration for it.
|
||||
|
||||
**Note (GEN3C)**: The official `nvidia/GEN3C-Cosmos-7B` repo provides a raw
|
||||
`model.pt` checkpoint. Use a Diffusers-format repo (for example,
|
||||
`FastVideo/GEN3C-Cosmos-7B-Diffusers`) or convert locally with
|
||||
`scripts/checkpoint_conversion/convert_gen3c_to_fastvideo.py`.
|
||||
|
||||
## Special requirements
|
||||
|
||||
### Sliding Tile Attention
|
||||
|
||||
@@ -27,8 +27,7 @@ Useful variables:
|
||||
- `FASTVIDEO_LOGGING_LEVEL`: `DEBUG`, `INFO`, `WARNING`, `ERROR`
|
||||
- `FASTVIDEO_STAGE_LOGGING`: print per-stage timings during pipeline execution
|
||||
- `FASTVIDEO_ATTENTION_BACKEND`: force an attention backend (for example
|
||||
`TORCH_SDPA`, `FLASH_ATTN`, `SAGE_ATTN_THREE`, or
|
||||
`ATTN_QAT_INFER`, or `ATTN_QAT_TRAIN`)
|
||||
`TORCH_SDPA` or `FLASH_ATTN`)
|
||||
|
||||
## Common Failure Modes
|
||||
|
||||
@@ -53,11 +52,7 @@ If forcing a backend fails, verify optional dependencies are installed:
|
||||
- `VIDEO_SPARSE_ATTN`: `fastvideo-kernel`
|
||||
- `SLIDING_TILE_ATTN`: STA legacy workflow in
|
||||
`sta_do_not_delete` + `fastvideo-kernel`
|
||||
- `SAGE_ATTN`: SageAttention package
|
||||
- `SAGE_ATTN_THREE`: upstream `sageattn3` package
|
||||
- `ATTN_QAT_INFER`: `fastvideo-kernel` checkout/source install that exposes
|
||||
`attn_qat_infer`
|
||||
- `ATTN_QAT_TRAIN`: `fastvideo-kernel` install exposing `fastvideo_kernel`
|
||||
- `SAGE_ATTN` / `SAGE_ATTN_THREE`: SageAttention packages
|
||||
|
||||
As a fallback, use:
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@ export NODE_RANK=$SLURM_PROCID
|
||||
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
|
||||
export MASTER_ADDR=${nodes[0]}
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
export WANDB_API_KEY="2f25ad37933894dbf0966c838c0b8494987f9f2f"
|
||||
# export WANDB_API_KEY='your_wandb_api_key_here'
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
|
||||
@@ -9,7 +9,7 @@ pip install vsa
|
||||
|
||||
### 1. Download dataset:
|
||||
```bash
|
||||
bash examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/download_dataset.sh
|
||||
bash examples/distill/Wan-Syn-480P/download_dataset.sh
|
||||
```
|
||||
|
||||
### 2. Configure and run distillation:
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
#!/bin/bash
|
||||
mkdir -p data
|
||||
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "data/Wan-Syn_77x448x832_600k" --repo_type "dataset"
|
||||
|
||||
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "FastVideo/Wan-Syn_77x448x832_600k" --repo_type "dataset"
|
||||
|
||||
@@ -28,11 +28,6 @@ For an example running DMD+VSA inference:
|
||||
python examples/inference/basic/basic_dmd.py
|
||||
```
|
||||
|
||||
For the typed config/request path added during the inference API refactor:
|
||||
```
|
||||
python examples/inference/basic/basic_dmd_new_api.py
|
||||
```
|
||||
|
||||
## Basic Walkthrough
|
||||
|
||||
All you need to generate videos using multi-gpus from state-of-the-art diffusion pipelines is the following few lines!
|
||||
|
||||
@@ -1,98 +0,0 @@
|
||||
import os
|
||||
import time
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
EngineConfig,
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
OffloadConfig,
|
||||
OutputConfig,
|
||||
PipelineSelection,
|
||||
)
|
||||
|
||||
OUTPUT_PATH = "video_samples_dmd2_typed"
|
||||
|
||||
|
||||
def main():
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
|
||||
|
||||
model_name = "FastVideo/FastWan2.1-T2V-1.3B-Diffusers"
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=model_name,
|
||||
engine=EngineConfig(
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
offload=OffloadConfig(
|
||||
text_encoder=True,
|
||||
pin_cpu_memory=True,
|
||||
dit=False,
|
||||
vae=False,
|
||||
),
|
||||
),
|
||||
# PR 2 still routes a few advanced inference knobs through the
|
||||
# compatibility bridge until they get first-class typed fields.
|
||||
pipeline=PipelineSelection(
|
||||
experimental={
|
||||
"VSA_sparsity": 0.8,
|
||||
},
|
||||
),
|
||||
)
|
||||
|
||||
load_start_time = time.perf_counter()
|
||||
generator = VideoGenerator.from_config(generator_config)
|
||||
load_end_time = time.perf_counter()
|
||||
load_time = load_end_time - load_start_time
|
||||
|
||||
prompt = (
|
||||
"A neon-lit alley in futuristic Tokyo during a heavy rainstorm at night. "
|
||||
"The puddles reflect glowing signs in kanji, advertising ramen, karaoke, "
|
||||
"and VR arcades. A woman in a translucent raincoat walks briskly with an "
|
||||
"LED umbrella. Steam rises from a street food cart, and a cat darts "
|
||||
"across the screen. Raindrops are visible on the camera lens, creating "
|
||||
"a cinematic bokeh effect."
|
||||
)
|
||||
request = GenerationRequest(
|
||||
prompt=prompt,
|
||||
output=OutputConfig(
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
return_frames=False,
|
||||
),
|
||||
)
|
||||
|
||||
start_time = time.perf_counter()
|
||||
result = generator.generate(request)
|
||||
end_time = time.perf_counter()
|
||||
gen_time = end_time - start_time
|
||||
|
||||
prompt2 = (
|
||||
"A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently "
|
||||
"in the breeze, enhancing the lion's commanding presence. The tone is "
|
||||
"vibrant, embodying the raw energy of the wild. Low angle, steady "
|
||||
"tracking shot, cinematic."
|
||||
)
|
||||
request2 = GenerationRequest(
|
||||
prompt=prompt2,
|
||||
output=OutputConfig(
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
return_frames=False,
|
||||
),
|
||||
)
|
||||
|
||||
start_time = time.perf_counter()
|
||||
result2 = generator.generate(request2)
|
||||
end_time = time.perf_counter()
|
||||
gen_time2 = end_time - start_time
|
||||
|
||||
print(f"Time taken to load model: {load_time} seconds")
|
||||
print(f"Time taken to generate video: {gen_time} seconds")
|
||||
print(f"First output written to: {result.video_path}")
|
||||
print(f"Time taken to generate video2: {gen_time2} seconds")
|
||||
print(f"Second output written to: {result2.video_path}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,109 +0,0 @@
|
||||
"""
|
||||
GEN3C: 3D-aware camera-controlled video generation.
|
||||
|
||||
This example generates a video from a single input image with camera control.
|
||||
The pipeline uses MoGe depth estimation, 3D point cloud forward warping,
|
||||
and the GEN3C diffusion model.
|
||||
|
||||
Requirements:
|
||||
1. Install MoGe:
|
||||
pip install git+https://github.com/microsoft/MoGe.git
|
||||
If you hit `ImportError: libGL.so.1`, install:
|
||||
sudo apt-get update && sudo apt-get install -y libgl1 libglib2.0-0 libsm6 libxext6 libxrender1
|
||||
2. Download and convert weights:
|
||||
huggingface-cli download nvidia/GEN3C-Cosmos-7B --local-dir official_weights/GEN3C-Cosmos-7B
|
||||
python scripts/checkpoint_conversion/convert_gen3c_to_fastvideo.py \
|
||||
--source ./official_weights/GEN3C-Cosmos-7B/model.pt \
|
||||
--output ./converted_weights/GEN3C-Cosmos-7B \
|
||||
--components-source nvidia/Cosmos-Predict2-2B-Video2World
|
||||
3. Provide an input image for 3D-conditioned generation.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="GEN3C video generation")
|
||||
parser.add_argument("--model_path",
|
||||
type=str,
|
||||
default="converted_weights/GEN3C-Cosmos-7B")
|
||||
parser.add_argument("--image_path",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Input image for 3D cache conditioning")
|
||||
parser.add_argument("--prompt",
|
||||
type=str,
|
||||
default="A slow camera pan over a sunlit landscape.")
|
||||
parser.add_argument(
|
||||
"--negative_prompt",
|
||||
type=str,
|
||||
default=(
|
||||
"The video captures a series of frames showing ugly scenes, static with no motion, motion blur, "
|
||||
"over-saturation, shaky footage, low resolution, grainy texture, pixelated images, poorly lit areas, "
|
||||
"underexposed and overexposed scenes, poor color balance, washed out colors, choppy sequences, "
|
||||
"jerky movements, low frame rate, artifacting, color banding, unnatural transitions, outdated special "
|
||||
"effects, fake elements, unconvincing visuals, poorly edited content, jump cuts, visual noise, and "
|
||||
"flickering. Overall, the video is of poor quality."
|
||||
),
|
||||
)
|
||||
parser.add_argument("--trajectory",
|
||||
type=str,
|
||||
default="left",
|
||||
choices=[
|
||||
"left", "right", "up", "down", "zoom_in",
|
||||
"zoom_out", "clockwise", "counterclockwise", "none"
|
||||
])
|
||||
parser.add_argument("--movement_distance", type=float, default=0.3)
|
||||
parser.add_argument("--camera_rotation",
|
||||
type=str,
|
||||
default="center_facing",
|
||||
choices=[
|
||||
"center_facing", "no_rotation",
|
||||
"trajectory_aligned"
|
||||
])
|
||||
parser.add_argument("--height", type=int, default=704)
|
||||
parser.add_argument("--width", type=int, default=1280)
|
||||
parser.add_argument("--num_frames", type=int, default=121)
|
||||
parser.add_argument("--num_inference_steps", type=int, default=35)
|
||||
parser.add_argument("--guidance_scale", type=float, default=1.0)
|
||||
parser.add_argument("--output_path",
|
||||
type=str,
|
||||
default="outputs_video/gen3c.mp4")
|
||||
parser.add_argument("--seed", type=int, default=42)
|
||||
args = parser.parse_args()
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
args.model_path,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True,
|
||||
)
|
||||
|
||||
video = generator.generate_video(
|
||||
args.prompt,
|
||||
negative_prompt=args.negative_prompt,
|
||||
image_path=args.image_path,
|
||||
trajectory_type=args.trajectory,
|
||||
movement_distance=args.movement_distance,
|
||||
camera_rotation=args.camera_rotation,
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
num_frames=args.num_frames,
|
||||
num_inference_steps=args.num_inference_steps,
|
||||
guidance_scale=args.guidance_scale,
|
||||
fps=24,
|
||||
seed=args.seed,
|
||||
output_path=args.output_path,
|
||||
save_video=True,
|
||||
)
|
||||
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,79 +1,5 @@
|
||||
# Optimization Examples
|
||||
|
||||
## Wan 2.1 QAT Attention 14B Inference
|
||||
|
||||
Use these files for Wan 2.1 14B inference with the `ATTN_QAT_INFER` backend:
|
||||
|
||||
- `examples/inference/optimizations/download_14B_qat.sh`
|
||||
- `examples/inference/optimizations/attn_qat_inference_example.py`
|
||||
|
||||
### 1. Download the 14B QAT checkpoint
|
||||
|
||||
The helper script downloads the QAT safetensors from
|
||||
`FastVideo/14B_qat_400` into `checkpoints/14B_qat_400` by default.
|
||||
|
||||
Prerequisites:
|
||||
|
||||
- `huggingface_hub` installed, for example: `uv pip install huggingface_hub`
|
||||
- access to the model repo if it is private or gated: `huggingface-cli login`
|
||||
|
||||
Run:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh
|
||||
python examples/inference/optimizations/attention_example.py
|
||||
```
|
||||
|
||||
To download into a custom directory, pass it as the first argument:
|
||||
|
||||
```bash
|
||||
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
|
||||
```
|
||||
|
||||
### 2. Edit the inference example for Wan 2.1 14B
|
||||
|
||||
Open `examples/inference/optimizations/attn_qat_inference_example.py` and
|
||||
update these two values:
|
||||
|
||||
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
|
||||
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
|
||||
2. Replace the placeholder
|
||||
`init_weights_from_safetensors="safetensors_path"` with the directory that
|
||||
contains the downloaded `.safetensors` files.
|
||||
|
||||
Example:
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
init_weights_from_safetensors="checkpoints/14B_qat_400",
|
||||
)
|
||||
```
|
||||
|
||||
The script already sets:
|
||||
|
||||
```python
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
```
|
||||
|
||||
### 3. Run the example
|
||||
|
||||
```bash
|
||||
python examples/inference/optimizations/attn_qat_inference_example.py
|
||||
```
|
||||
|
||||
The generated videos are written to `video_samples/` by default.
|
||||
|
||||
### Notes
|
||||
|
||||
- `ATTN_QAT_INFER` requires the in-repo `fastvideo-kernel` build to expose the
|
||||
`attn_qat_infer` package.
|
||||
- If you have not built the kernel yet, run `cd fastvideo-kernel && ./build.sh`
|
||||
first.
|
||||
- If you keep the example on the `1.3B` base model while loading the 14B QAT
|
||||
weights, the model/config will not match.
|
||||
|
||||
@@ -1,54 +0,0 @@
|
||||
from fastvideo import VideoGenerator
|
||||
import os
|
||||
from pathlib import Path
|
||||
# from fastvideo.configs.sample import SamplingParam
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
CHECKPOINT_PATH = Path(__file__).parent.parent.parent
|
||||
|
||||
def main():
|
||||
# FastVideo will automatically use the optimal default arguments for the
|
||||
# model.
|
||||
# If a local path is provided, FastVideo will make a best effort
|
||||
# attempt to identify the optimal arguments.
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
|
||||
# image_encoder_cpu_offload=False,
|
||||
# Load custom weights from checkpoint
|
||||
init_weights_from_safetensors="safetensors_path"
|
||||
)
|
||||
|
||||
# sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
# sampling_param.num_frames = 45
|
||||
# sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
|
||||
# Generate videos with the same simple API, regardless of GPU count
|
||||
prompt = (
|
||||
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
)
|
||||
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True)
|
||||
# video = generator.generate_video(prompt, sampling_param=sampling_param, output_path="wan_t2v_videos/")
|
||||
|
||||
# Generate another video with a different prompt, without reloading the
|
||||
# model!
|
||||
prompt2 = (
|
||||
"A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently in "
|
||||
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic.")
|
||||
video2 = generator.generate_video(prompt2, output_path=OUTPUT_PATH, save_video=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,58 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
|
||||
REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." && pwd)"
|
||||
|
||||
HF_REPO_ID="${HF_REPO_ID:-FastVideo/14B_qat_400}"
|
||||
HF_REVISION="${HF_REVISION:-main}"
|
||||
LOCAL_DIR="${1:-${REPO_ROOT}/checkpoints/14B_qat_400}"
|
||||
PYTHON_BIN="${PYTHON:-python}"
|
||||
|
||||
if ! command -v "${PYTHON_BIN}" >/dev/null 2>&1; then
|
||||
echo "Python executable not found: ${PYTHON_BIN}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! "${PYTHON_BIN}" -c "import huggingface_hub" >/dev/null 2>&1; then
|
||||
echo "Missing dependency: huggingface_hub" >&2
|
||||
echo "Install it with: uv pip install huggingface_hub" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p "${LOCAL_DIR}"
|
||||
|
||||
echo "Downloading ${HF_REPO_ID}@${HF_REVISION}"
|
||||
echo "Local directory: ${LOCAL_DIR}"
|
||||
|
||||
"${PYTHON_BIN}" -c '
|
||||
import argparse
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--repo-id", required=True)
|
||||
parser.add_argument("--revision", required=True)
|
||||
parser.add_argument("--local-dir", required=True)
|
||||
args = parser.parse_args()
|
||||
|
||||
snapshot_download(
|
||||
repo_id=args.repo_id,
|
||||
revision=args.revision,
|
||||
repo_type="model",
|
||||
local_dir=args.local_dir,
|
||||
local_dir_use_symlinks=False,
|
||||
resume_download=True,
|
||||
)
|
||||
' \
|
||||
--repo-id "${HF_REPO_ID}" \
|
||||
--revision "${HF_REVISION}" \
|
||||
--local-dir "${LOCAL_DIR}"
|
||||
|
||||
echo
|
||||
echo "Download complete."
|
||||
echo "Use this in your inference script:"
|
||||
echo "init_weights_from_safetensors=\"${LOCAL_DIR}\""
|
||||
echo
|
||||
echo "If the repo is private or gated, make sure you are logged in with:"
|
||||
echo "huggingface-cli login"
|
||||
@@ -1,88 +0,0 @@
|
||||
import torch
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.pipelines.base import PipelineConfig
|
||||
|
||||
OUTPUT_PATH = "video_samples"
|
||||
|
||||
|
||||
def main():
|
||||
print("=== FP4 Quantization Video Generation Example ===")
|
||||
|
||||
if not torch.cuda.is_available():
|
||||
print("Warning: CUDA not available. FP4 quantization requires GPU.")
|
||||
return
|
||||
|
||||
gpu_capability = torch.cuda.get_device_capability()
|
||||
if gpu_capability[0] < 9: # H100 and newer
|
||||
print(f"Warning: GPU capability {gpu_capability} may not support FP4. Recommended: 9.0+")
|
||||
|
||||
print(f"GPU: {torch.cuda.get_device_name()}")
|
||||
print(f"GPU Capability: {gpu_capability}")
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
# model_id = "Wan-AI/Wan2.1-T2V-14B-Diffusers"
|
||||
pipeline_config = PipelineConfig.from_pretrained(model_id)
|
||||
pipeline_config.dit_precision = "bf16"
|
||||
|
||||
print("\nLoading model with FP4 quantization...")
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_id,
|
||||
pipeline_config=pipeline_config,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
transformer_quant="fp4",
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
)
|
||||
|
||||
print("FP4 configuration applied. Generating videos...")
|
||||
|
||||
print("\n=== Generating Video with FP4 Quantization ===")
|
||||
|
||||
prompt1 = (
|
||||
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
|
||||
)
|
||||
|
||||
print(f"Prompt: {prompt1}")
|
||||
print("Generating video...")
|
||||
|
||||
try:
|
||||
video1 = generator.generate_video(
|
||||
prompt1,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
)
|
||||
print("✓ First video generated successfully with FP4 quantization!")
|
||||
|
||||
# # Generate a second video to show the model can be reused
|
||||
prompt2 = (
|
||||
"A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently in "
|
||||
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic."
|
||||
)
|
||||
|
||||
print(f"\nGenerating second video...")
|
||||
print(f"Prompt: {prompt2}")
|
||||
|
||||
video2 = generator.generate_video(
|
||||
prompt2,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
)
|
||||
print("✓ Second video generated successfully with FP4 quantization!")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during video generation: {e}")
|
||||
return
|
||||
|
||||
print(f"Videos saved to: {OUTPUT_PATH}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,47 +1,27 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=wan_t2v_1.3B_finetune
|
||||
#SBATCH --partition=all
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --gres=gpu:4
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --output=logs/wan_t2v_1.3B_finetune.out
|
||||
#SBATCH --error=logs/wan_t2v_1.3B_finetune.err
|
||||
|
||||
source .venv/bin/activate
|
||||
|
||||
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
|
||||
|
||||
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATA_DIR=data/Wan-Syn_77x448x832_600k
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
|
||||
NUM_GPUS=1
|
||||
DATA_DIR="data/crush-smol_processed_t2v/combined_parquet_dataset/"
|
||||
VALIDATION_DATASET_FILE="$(dirname "$0")/validation.json"
|
||||
NUM_GPUS=4
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ---- torchrun rendezvous (multi-node) ----
|
||||
# Launch ONE torchrun per node (via srun) and let torchrun spawn 4 workers per node.
|
||||
MASTER_ADDR="$(scontrol show hostnames "$SLURM_JOB_NODELIST" | head -n 1)"
|
||||
MASTER_PORT="${MASTER_PORT:-29500}"
|
||||
export MASTER_ADDR MASTER_PORT
|
||||
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name "wan_t2v_finetune_qat"
|
||||
--output_dir "checkpoints/wan_t2v_finetune_1.3B_77"
|
||||
--max_train_steps 4000
|
||||
--tracker_project_name "wan_t2v_finetune"
|
||||
--output_dir "checkpoints/wan_t2v_finetune"
|
||||
--max_train_steps 5000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--gradient_accumulation_steps 8
|
||||
--num_latent_t 20
|
||||
--num_height 448
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--num_frames 77
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
@@ -50,7 +30,7 @@ training_args=(
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus $NUM_GPUS
|
||||
--sp_size 1
|
||||
--sp_size $NUM_GPUS
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1
|
||||
--hsdp_shard_dim $NUM_GPUS
|
||||
@@ -65,7 +45,7 @@ model_args=(
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path $DATA_DIR
|
||||
--dataloader_num_workers 4
|
||||
--dataloader_num_workers 1
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
@@ -74,16 +54,16 @@ validation_args=(
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "5.0"
|
||||
--validation_guidance_scale "3.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-6
|
||||
--learning_rate 5e-5
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 1000
|
||||
--training_state_checkpointing_steps 1000
|
||||
--weight_decay 0.01
|
||||
--weight_decay 1e-4
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
@@ -92,24 +72,23 @@ miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--multi_phased_distill_schedule "4000-1"
|
||||
--not_apply_cfg_solver
|
||||
--dit_precision "fp32"
|
||||
--num_euler_timesteps 50
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
# --resume_from_checkpoint "checkpoints/wan_t2v_finetune/checkpoint-2500"
|
||||
)
|
||||
|
||||
srun --nodes="$SLURM_NNODES" --ntasks="$SLURM_NNODES" --ntasks-per-node=1 \
|
||||
torchrun \
|
||||
--nnodes "$SLURM_NNODES" \
|
||||
--nproc_per_node 4 \
|
||||
--rdzv_backend c10d \
|
||||
--rdzv_endpoint "${MASTER_ADDR}:${MASTER_PORT}" \
|
||||
--rdzv_id "$SLURM_JOB_ID" \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
torchrun \
|
||||
--nnodes 1 \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
|
||||
@@ -1,124 +0,0 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
|
||||
#SBATCH --partition=all
|
||||
#SBATCH --nodes=4
|
||||
#SBATCH --gres=gpu:4
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
|
||||
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
|
||||
|
||||
source .venv/bin/activate
|
||||
|
||||
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
|
||||
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
# Use node-local Triton cache to avoid stale file handle errors on shared filesystems
|
||||
export TRITON_CACHE_DIR="/tmp/triton_cache_${SLURM_JOB_ID}_${SLURM_NODEID}"
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATA_DIR=YOUR_DATA_DIR
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
|
||||
NUM_GPUS=16
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ---- torchrun rendezvous (multi-node) ----
|
||||
# 1. Get the hostname of the first node (Master)
|
||||
nodes=( $( scontrol show hostnames $SLURM_JOB_NODELIST ) )
|
||||
nodes_array=($nodes)
|
||||
head_node=${nodes_array[0]}
|
||||
MASTER_ADDR=$(srun --nodes=1 --ntasks=1 -w "$head_node" hostname --ip-address)
|
||||
MASTER_PORT=29500
|
||||
|
||||
# 2. Get the node count automatically
|
||||
NNODES=$SLURM_NNODES
|
||||
GPUS_PER_NODE=$SLURM_GPUS_ON_NODE
|
||||
NUM_GPUS=$((NNODES * GPUS_PER_NODE))
|
||||
|
||||
echo "MASTER_ADDR=$MASTER_ADDR MASTER_PORT=$MASTER_PORT NNODES=$NNODES"
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name "wan_t2v_finetune_qat"
|
||||
--output_dir "checkpoints/wan_1.3B_t2v_finetune_qat"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 20
|
||||
--num_height 448
|
||||
--num_width 832
|
||||
--num_frames 77
|
||||
--enable_gradient_checkpointing_type "full" # if OOM enable this
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus $NUM_GPUS
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim $NUM_GPUS
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "5.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-6
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 200
|
||||
--training_state_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 1
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $NNODES \
|
||||
--nproc_per_node $GPUS_PER_NODE \
|
||||
--node_rank $SLURM_PROCID \
|
||||
--rdzv_backend c10d \
|
||||
--rdzv_endpoint $MASTER_ADDR:$MASTER_PORT \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
@@ -1,131 +1,31 @@
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
|
||||
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"caption": "A large metal cylinder is seen pressing down on a pile of Oreo cookies, flattening them as if they were under a hydraulic press.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
|
||||
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"caption": "A large metal cylinder is seen compressing colorful clay into a compact shape, demonstrating the power of a hydraulic press.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
|
||||
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"caption": "A large metal cylinder is seen pressing down on a pile of colorful candies, flattening them as if they were under a hydraulic press. The candies are crushed and broken into small pieces, creating a mess on the table.",
|
||||
"image_path": null,
|
||||
"video_path": null,
|
||||
"num_inference_steps": 50,
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
|
||||
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
|
||||
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
|
||||
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
|
||||
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
|
||||
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
|
||||
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
|
||||
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
|
||||
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
|
||||
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
|
||||
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
|
||||
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
|
||||
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
|
||||
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 448,
|
||||
"width": 832,
|
||||
"num_frames": 77
|
||||
} ]
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,123 +0,0 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
|
||||
#SBATCH --partition=main
|
||||
#SBATCH --nodes=4
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --cpus-per-task=128
|
||||
#SBATCH --mem=1440G
|
||||
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
|
||||
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
|
||||
#SBATCH --exclusive
|
||||
|
||||
source ~/conda/miniconda/bin/activate
|
||||
conda activate matthew-fv
|
||||
|
||||
# Basic Info
|
||||
export WANDB_MODE="online"
|
||||
export NCCL_P2P_DISABLE=1
|
||||
export TORCH_NCCL_ENABLE_MONITORING=0
|
||||
# different cache dir for different processes
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
|
||||
export MASTER_PORT=29500
|
||||
export NODE_RANK=$SLURM_PROCID
|
||||
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
|
||||
export MASTER_ADDR=${nodes[0]}
|
||||
export CUDA_VISIBLE_DEVICES=$SLURM_LOCALID
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export WANDB_MODE=online
|
||||
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
|
||||
|
||||
echo "MASTER_ADDR: $MASTER_ADDR"
|
||||
echo "NODE_RANK: $NODE_RANK"
|
||||
|
||||
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
|
||||
export WANDB_API_KEY=YOUR_WANDB_API_KEY
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-14B-Diffusers"
|
||||
DATA_DIR=YOUR_DATA_DIR
|
||||
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
|
||||
NUM_GPUS_PER_NODE=8
|
||||
TOTAL_GPUS=$((NUM_GPUS_PER_NODE * SLURM_JOB_NUM_NODES))
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name "wan_t2v_finetune_qat"
|
||||
--output_dir "checkpoints/wan_14B_t2v_finetune_qat"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 20
|
||||
--num_height 768
|
||||
--num_width 1280
|
||||
--num_frames 77
|
||||
--enable_gradient_checkpointing_type "full" # if OOM enable this
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus $TOTAL_GPUS
|
||||
--sp_size 4
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 4
|
||||
--hsdp_shard_dim 8
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
# --log_validation
|
||||
--validation_dataset_file $VALIDATION_DATASET_FILE
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "50"
|
||||
--validation_guidance_scale "5.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-6
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 200
|
||||
--training_state_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.1
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS_PER_NODE \
|
||||
--node_rank $SLURM_PROCID \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}"
|
||||
@@ -1,131 +0,0 @@
|
||||
{
|
||||
"data": [
|
||||
{
|
||||
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
|
||||
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
|
||||
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
|
||||
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
|
||||
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
|
||||
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
|
||||
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
|
||||
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
|
||||
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
|
||||
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
|
||||
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
|
||||
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
|
||||
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
|
||||
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
|
||||
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
|
||||
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
},
|
||||
{
|
||||
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
|
||||
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
|
||||
"num_inference_steps": 50,
|
||||
"height": 768,
|
||||
"width": 1280,
|
||||
"num_frames": 77
|
||||
} ]
|
||||
}
|
||||
@@ -12,18 +12,6 @@ else()
|
||||
enable_language(CUDA)
|
||||
# Ensure CUDA toolkit targets (CUDA::cudart, CUDA::cuda_driver, etc.) are available.
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
if(NOT DEFINED CUDA_TOOLKIT_ROOT_DIR)
|
||||
if(DEFINED CUDAToolkit_ROOT)
|
||||
set(CUDA_TOOLKIT_ROOT_DIR "${CUDAToolkit_ROOT}" CACHE PATH
|
||||
"CUDA toolkit root directory" FORCE)
|
||||
elseif(DEFINED ENV{CUDAToolkit_ROOT})
|
||||
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDAToolkit_ROOT}" CACHE PATH
|
||||
"CUDA toolkit root directory" FORCE)
|
||||
elseif(DEFINED ENV{CUDA_HOME})
|
||||
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDA_HOME}" CACHE PATH
|
||||
"CUDA toolkit root directory" FORCE)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Import common utils if needed, but we keep it simple for now
|
||||
@@ -31,46 +19,13 @@ endif()
|
||||
# Find Python and Torch
|
||||
find_package(Python COMPONENTS Interpreter Development.Module REQUIRED)
|
||||
|
||||
# Locate the installed torch package without importing it. This keeps CMake
|
||||
# configure working even on nodes where CUDA runtime libraries are not yet on
|
||||
# the dynamic loader path.
|
||||
# Robustly find Torch include paths using Python
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('platlib'))"
|
||||
OUTPUT_VARIABLE PYTHON_PLATLIB
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import torch; from torch.utils.cpp_extension import include_paths; print(';'.join(include_paths()))"
|
||||
OUTPUT_VARIABLE TORCH_INCLUDE_PATHS
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('purelib'))"
|
||||
OUTPUT_VARIABLE PYTHON_PURELIB
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
|
||||
set(TORCH_PYTHON_PACKAGE_DIR "")
|
||||
foreach(_candidate
|
||||
"${PYTHON_PLATLIB}/torch"
|
||||
"${PYTHON_PURELIB}/torch"
|
||||
)
|
||||
if(EXISTS "${_candidate}")
|
||||
set(TORCH_PYTHON_PACKAGE_DIR "${_candidate}")
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if(NOT TORCH_PYTHON_PACKAGE_DIR)
|
||||
message(FATAL_ERROR "Could not locate the installed torch Python package.")
|
||||
endif()
|
||||
|
||||
list(APPEND TORCH_INCLUDE_DIRS
|
||||
"${TORCH_PYTHON_PACKAGE_DIR}/include"
|
||||
"${TORCH_PYTHON_PACKAGE_DIR}/include/torch/csrc/api/include"
|
||||
)
|
||||
|
||||
if(NOT Torch_DIR)
|
||||
set(_TORCH_CONFIG_DIR "${TORCH_PYTHON_PACKAGE_DIR}/share/cmake/Torch")
|
||||
if(EXISTS "${_TORCH_CONFIG_DIR}/TorchConfig.cmake")
|
||||
set(Torch_DIR "${_TORCH_CONFIG_DIR}" CACHE PATH "Path to Torch CMake config" FORCE)
|
||||
endif()
|
||||
endif()
|
||||
list(APPEND TORCH_INCLUDE_DIRS ${TORCH_INCLUDE_PATHS})
|
||||
|
||||
# Find Torch package (still useful for libraries)
|
||||
find_package(Torch REQUIRED)
|
||||
@@ -95,21 +50,6 @@ include_directories(
|
||||
set(FASTVIDEO_KERNEL_BUILD_TK "AUTO" CACHE STRING "Build ThunderKittens kernels: AUTO/ON/OFF")
|
||||
set_property(CACHE FASTVIDEO_KERNEL_BUILD_TK PROPERTY STRINGS AUTO ON OFF)
|
||||
|
||||
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "AUTO")
|
||||
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 AND NOT DEFINED CACHE{FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER})
|
||||
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "${FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3}")
|
||||
endif()
|
||||
|
||||
set(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER "${_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT}" CACHE STRING
|
||||
"Build attn_qat_infer Blackwell inference kernels: AUTO/ON/OFF")
|
||||
set_property(CACHE FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER PROPERTY STRINGS AUTO ON OFF)
|
||||
|
||||
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3)
|
||||
message(DEPRECATION
|
||||
"FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 is deprecated. "
|
||||
"Use FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER instead.")
|
||||
endif()
|
||||
|
||||
# Prefer environment variable (used by CI) if CMake var is not explicitly set.
|
||||
if(NOT DEFINED TORCH_CUDA_ARCH_LIST AND DEFINED ENV{TORCH_CUDA_ARCH_LIST})
|
||||
set(TORCH_CUDA_ARCH_LIST "$ENV{TORCH_CUDA_ARCH_LIST}")
|
||||
@@ -117,7 +57,6 @@ endif()
|
||||
|
||||
message(STATUS "TORCH_CUDA_ARCH_LIST (cmake/env): ${TORCH_CUDA_ARCH_LIST}")
|
||||
message(STATUS "FASTVIDEO_KERNEL_BUILD_TK: ${FASTVIDEO_KERNEL_BUILD_TK}")
|
||||
message(STATUS "FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER: ${FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER}")
|
||||
|
||||
set(ENABLE_TK_KERNELS OFF)
|
||||
if(FASTVIDEO_KERNEL_BUILD_TK STREQUAL "ON")
|
||||
@@ -152,54 +91,6 @@ else()
|
||||
message(STATUS "ThunderKittens kernels: DISABLED (will use Triton fallbacks at runtime)")
|
||||
endif()
|
||||
|
||||
set(ENABLE_ATTN_QAT_INFER OFF)
|
||||
if(GPU_BACKEND STREQUAL "ROCM")
|
||||
message(STATUS "attn_qat_infer kernels: DISABLED (ROCm build)")
|
||||
else()
|
||||
set(_WANTS_ATTN_QAT_INFER OFF)
|
||||
if(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "ON")
|
||||
set(_WANTS_ATTN_QAT_INFER ON)
|
||||
elseif(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "AUTO")
|
||||
if(TORCH_CUDA_ARCH_LIST)
|
||||
string(REGEX MATCH
|
||||
"(^|[; ,])((12\\.0a)|(120a)|(sm_120a))([; ,]|$)"
|
||||
_HAS_120A "${TORCH_CUDA_ARCH_LIST}")
|
||||
if(_HAS_120A)
|
||||
set(_WANTS_ATTN_QAT_INFER ON)
|
||||
endif()
|
||||
else()
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c
|
||||
"import torch; print('1' if (torch.cuda.is_available() and torch.version.cuda and torch.cuda.get_device_capability()[0] >= 12) else '0')"
|
||||
OUTPUT_VARIABLE _LOCAL_HAS_BLACKWELL
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_QUIET
|
||||
)
|
||||
if(_LOCAL_HAS_BLACKWELL STREQUAL "1")
|
||||
set(_WANTS_ATTN_QAT_INFER ON)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(_WANTS_ATTN_QAT_INFER)
|
||||
if(CUDAToolkit_VERSION VERSION_LESS 12.8)
|
||||
message(WARNING
|
||||
"attn_qat_infer kernels require CUDA Toolkit 12.8+. "
|
||||
"Skipping because CUDAToolkit_VERSION=${CUDAToolkit_VERSION}.")
|
||||
else()
|
||||
set(ENABLE_ATTN_QAT_INFER ON)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(ENABLE_ATTN_QAT_INFER)
|
||||
message(STATUS "attn_qat_infer kernels: ENABLED")
|
||||
else()
|
||||
message(STATUS
|
||||
"attn_qat_infer kernels: DISABLED "
|
||||
"(requires CUDA 12.8+ and Blackwell sm_120a)")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Always try to build the extension if CUDA is available, but conditionally add sources/flags
|
||||
set(BUILD_CXX_KERNELS ON)
|
||||
|
||||
@@ -270,15 +161,12 @@ if(BUILD_CXX_KERNELS)
|
||||
|
||||
# Also link against libtorch_python to satisfy Python-binding symbols
|
||||
# (e.g., torch::PyWarningHandler) required by torch/extension.h.
|
||||
file(GLOB TORCH_PYTHON_LIBRARY_CANDIDATES
|
||||
"${TORCH_PYTHON_PACKAGE_DIR}/lib/libtorch_python*"
|
||||
execute_process(
|
||||
COMMAND "${Python_EXECUTABLE}" -c "import torch; from pathlib import Path; p=Path(torch.__file__).parent/'lib'; m=sorted(p.glob('libtorch_python*')); print(str(m[0]) if m else '')"
|
||||
OUTPUT_VARIABLE TORCH_PYTHON_LIBRARY_PATH
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_QUIET
|
||||
)
|
||||
list(LENGTH TORCH_PYTHON_LIBRARY_CANDIDATES _TORCH_PYTHON_LIBRARY_COUNT)
|
||||
if(_TORCH_PYTHON_LIBRARY_COUNT GREATER 0)
|
||||
list(GET TORCH_PYTHON_LIBRARY_CANDIDATES 0 TORCH_PYTHON_LIBRARY_PATH)
|
||||
else()
|
||||
set(TORCH_PYTHON_LIBRARY_PATH "")
|
||||
endif()
|
||||
if(TORCH_PYTHON_LIBRARY_PATH)
|
||||
message(STATUS "TORCH_PYTHON_LIBRARY_PATH: ${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
target_link_libraries(fastvideo_kernel_ops PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
@@ -295,73 +183,3 @@ if(BUILD_CXX_KERNELS)
|
||||
install(TARGETS fastvideo_kernel_ops LIBRARY DESTINATION fastvideo_kernel/_C)
|
||||
endif()
|
||||
|
||||
if(ENABLE_ATTN_QAT_INFER)
|
||||
set(ATTN_QAT_INFER_DIR ${CMAKE_SOURCE_DIR}/attn_qat_infer)
|
||||
set(ATTN_QAT_INFER_INCLUDE_DIRS
|
||||
${ATTN_QAT_INFER_DIR}
|
||||
${CMAKE_SOURCE_DIR}/include/cutlass/include
|
||||
${CMAKE_SOURCE_DIR}/include/cutlass/tools/util/include
|
||||
${TORCH_INCLUDE_DIRS}
|
||||
)
|
||||
set(ATTN_QAT_INFER_CUDA_FLAGS
|
||||
"-O3"
|
||||
"-std=c++17"
|
||||
"-U__CUDA_NO_HALF_OPERATORS__"
|
||||
"-U__CUDA_NO_HALF_CONVERSIONS__"
|
||||
"-U__CUDA_NO_BFLOAT16_OPERATORS__"
|
||||
"-U__CUDA_NO_BFLOAT16_CONVERSIONS__"
|
||||
"-U__CUDA_NO_BFLOAT162_OPERATORS__"
|
||||
"-U__CUDA_NO_BFLOAT162_CONVERSIONS__"
|
||||
"--expt-relaxed-constexpr"
|
||||
"--expt-extended-lambda"
|
||||
"--use_fast_math"
|
||||
"--ptxas-options=--verbose,--warn-on-local-memory-usage"
|
||||
"-lineinfo"
|
||||
"-DCUTLASS_DEBUG_TRACE_LEVEL=0"
|
||||
"-DNDEBUG"
|
||||
"-DQBLKSIZE=128"
|
||||
"-DKBLKSIZE=128"
|
||||
"-DCTA256"
|
||||
"-DDQINRMEM"
|
||||
)
|
||||
|
||||
Python_add_library(fp4attn_cuda MODULE WITH_SOABI
|
||||
attn_qat_infer/blackwell/api.cu
|
||||
)
|
||||
target_include_directories(fp4attn_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
|
||||
target_compile_definitions(fp4attn_cuda PRIVATE TORCH_EXTENSION_NAME=fp4attn_cuda)
|
||||
target_compile_options(fp4attn_cuda PRIVATE
|
||||
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
|
||||
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
|
||||
)
|
||||
set_target_properties(fp4attn_cuda PROPERTIES
|
||||
CUDA_ARCHITECTURES "120a"
|
||||
CXX_STANDARD 17
|
||||
CUDA_STANDARD 17
|
||||
)
|
||||
target_link_libraries(fp4attn_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
|
||||
|
||||
Python_add_library(fp4quant_cuda MODULE WITH_SOABI
|
||||
attn_qat_infer/quantization/fp4_quantization_4d.cu
|
||||
)
|
||||
target_include_directories(fp4quant_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
|
||||
target_compile_definitions(fp4quant_cuda PRIVATE TORCH_EXTENSION_NAME=fp4quant_cuda)
|
||||
target_compile_options(fp4quant_cuda PRIVATE
|
||||
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
|
||||
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
|
||||
)
|
||||
set_target_properties(fp4quant_cuda PROPERTIES
|
||||
CUDA_ARCHITECTURES "120a"
|
||||
CXX_STANDARD 17
|
||||
CUDA_STANDARD 17
|
||||
)
|
||||
target_link_libraries(fp4quant_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
|
||||
|
||||
if(TORCH_PYTHON_LIBRARY_PATH)
|
||||
target_link_libraries(fp4attn_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
target_link_libraries(fp4quant_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
|
||||
endif()
|
||||
|
||||
install(TARGETS fp4attn_cuda LIBRARY DESTINATION .)
|
||||
install(TARGETS fp4quant_cuda LIBRARY DESTINATION .)
|
||||
endif()
|
||||
|
||||
@@ -2,6 +2,5 @@ include LICENSE
|
||||
include README.md
|
||||
include pyproject.toml
|
||||
recursive-include python/fastvideo_kernel *.py
|
||||
recursive-include attn_qat_infer *.py *.cu *.cuh *.cpp *.h
|
||||
recursive-include csrc *.cu *.cuh *.cpp *.h
|
||||
recursive-include include/tk *.cu *.cuh *.cpp *.h *.src
|
||||
|
||||
@@ -20,11 +20,6 @@ cd fastvideo-kernel
|
||||
./build.sh
|
||||
```
|
||||
|
||||
On supported Blackwell environments, the same install also packages
|
||||
`attn_qat_infer` and builds its `fp4attn_cuda` / `fp4quant_cuda`
|
||||
extensions directly from `fastvideo-kernel/attn_qat_infer/`. This path
|
||||
requires CUDA Toolkit 12.8+ and targets `sm_120a`.
|
||||
|
||||
### Rocm Build
|
||||
If you are in a rocm environment without the compilation toolchaine of CUDA.
|
||||
|
||||
@@ -63,18 +58,6 @@ cd fastvideo-kernel
|
||||
python benchmarks/bench_vsa.py --batch_size 1 --num_heads 16 --head_dim 128 --q_seq_lens 49152 --topk 64
|
||||
```
|
||||
|
||||
### Attn QAT Attention Benchmarks
|
||||
|
||||
The Attn QAT microbenchmarks now live alongside the kernel package:
|
||||
|
||||
```bash
|
||||
cd fastvideo-kernel
|
||||
python benchmarks/benchmark_flashattn2.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
|
||||
python benchmarks/benchmark_sageattn3.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
|
||||
python benchmarks/benchmark_blockscaled_fp4_attn.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
|
||||
python benchmarks/benchmark_combined.py --output benchmark_attention.png
|
||||
```
|
||||
|
||||
### TurboDiffusion Kernels
|
||||
|
||||
This package also includes kernels from [TurboDiffusion](https://github.com/thu-ml/TurboDiffusion), including INT8 GEMM, Quantization, RMSNorm and LayerNorm.
|
||||
|
||||
@@ -1,16 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
from .api import sageattn_blackwell
|
||||
@@ -1,185 +0,0 @@
|
||||
# Modified from the original SageATtention3 code
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import triton
|
||||
import triton.language as tl
|
||||
import torch.nn.functional as F
|
||||
from typing import Tuple
|
||||
from torch.nn.functional import scaled_dot_product_attention as sdpa
|
||||
import fp4attn_cuda
|
||||
import fp4quant_cuda
|
||||
|
||||
# Centralized block size configuration for sageattn_blackwell kernels
|
||||
# These should match the values in fastvideo/attention/backends/sageattn/blackwell/block_config.h
|
||||
BLOCK_M = 128 # Block size for M dimension (query sequence length)
|
||||
BLOCK_N = 128 # Block size for N dimension (key/value sequence length)
|
||||
|
||||
|
||||
@triton.jit
|
||||
def group_mean_kernel(
|
||||
q_ptr,
|
||||
q_out_ptr,
|
||||
qm_out_ptr,
|
||||
B, H, L, D: tl.constexpr,
|
||||
stride_qb, stride_qh, stride_ql, stride_qd,
|
||||
stride_qmb, stride_qmh, stride_qml, stride_qmd,
|
||||
GROUP_SIZE: tl.constexpr
|
||||
):
|
||||
pid_b = tl.program_id(0)
|
||||
pid_h = tl.program_id(1)
|
||||
pid_group = tl.program_id(2)
|
||||
|
||||
group_start = pid_group * GROUP_SIZE
|
||||
offsets = group_start + tl.arange(0, GROUP_SIZE)
|
||||
|
||||
q_offsets = pid_b * stride_qb + pid_h * stride_qh + offsets[:, None] * stride_ql + tl.arange(0, D)[None, :] * stride_qd
|
||||
q_group = tl.load(q_ptr + q_offsets)
|
||||
|
||||
qm_group = tl.sum(q_group, axis=0) / GROUP_SIZE
|
||||
|
||||
q_group = q_group - qm_group
|
||||
tl.store(q_out_ptr + q_offsets, q_group)
|
||||
|
||||
qm_offset = pid_b * stride_qmb + pid_h * stride_qmh + pid_group * stride_qml + tl.arange(0, D) * stride_qmd
|
||||
tl.store(qm_out_ptr + qm_offset, qm_group)
|
||||
|
||||
|
||||
def triton_group_mean(q: torch.Tensor):
|
||||
B, H, L, D = q.shape
|
||||
GROUP_SIZE = BLOCK_M
|
||||
num_groups = L // GROUP_SIZE
|
||||
|
||||
q_out = torch.empty_like(q) # [B, H, L, D]
|
||||
qm = torch.empty(B, H, num_groups, D, device=q.device, dtype=q.dtype)
|
||||
|
||||
grid = (B, H, num_groups)
|
||||
|
||||
group_mean_kernel[grid](
|
||||
q, q_out, qm,
|
||||
B, H, L, D,
|
||||
q.stride(0), q.stride(1), q.stride(2), q.stride(3),
|
||||
qm.stride(0), qm.stride(1), qm.stride(2), qm.stride(3),
|
||||
GROUP_SIZE=GROUP_SIZE
|
||||
)
|
||||
return q_out, qm
|
||||
|
||||
|
||||
def preprocess_qkv(q: torch.Tensor, k: torch.Tensor, v: torch.Tensor, per_block_mean: bool = True, enable_smoothing_q: bool = False, enable_smoothing_k: bool = False):
|
||||
|
||||
def pad_to_block_size(x):
|
||||
L = x.size(2)
|
||||
pad_len = (BLOCK_M - L % BLOCK_M) % BLOCK_M
|
||||
if pad_len == 0:
|
||||
return x.contiguous()
|
||||
return F.pad(x, (0, 0, 0, pad_len), value=0).contiguous()
|
||||
|
||||
if enable_smoothing_k:
|
||||
k -= k.mean(dim=-2, keepdim=True)
|
||||
q, k, v = map(lambda x: pad_to_block_size(x), [q, k, v])
|
||||
if per_block_mean and enable_smoothing_q:
|
||||
q, qm = triton_group_mean(q)
|
||||
elif enable_smoothing_q:
|
||||
qm = q.mean(dim=-2, keepdim=True)
|
||||
q = q - qm
|
||||
if enable_smoothing_q:
|
||||
delta_s = torch.matmul(qm, k.transpose(-2, -1)).to(torch.float32).contiguous()
|
||||
else: # used to disable q smoothing
|
||||
B, H, L, D = q.shape
|
||||
delta_s = torch.zeros((B, H, L // BLOCK_M, k.shape[2]), device=q.device, dtype=torch.float32)
|
||||
|
||||
return q, k, v, delta_s
|
||||
|
||||
def scale_and_quant_fp4(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
assert x.ndim == 4
|
||||
B, H, N, D = x.shape
|
||||
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
|
||||
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
|
||||
fp4quant_cuda.scaled_fp4_quant(x, packed_fp4, fp8_scale, 1)
|
||||
return packed_fp4, fp8_scale
|
||||
|
||||
def scale_and_quant_fp4_permute(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
assert x.ndim == 4
|
||||
B, H, N, D = x.shape
|
||||
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
|
||||
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
|
||||
fp4quant_cuda.scaled_fp4_quant_permute(x, packed_fp4, fp8_scale, 1)
|
||||
return packed_fp4, fp8_scale
|
||||
|
||||
def scale_and_quant_fp4_transpose(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
assert x.ndim == 4
|
||||
B, H, N, D = x.shape
|
||||
packed_fp4 = torch.empty((B, H, D, N // 2), device=x.device, dtype=torch.uint8)
|
||||
fp8_scale = torch.empty((B, H, D, N // 16), device=x.device, dtype=torch.float8_e4m3fn)
|
||||
fp4quant_cuda.scaled_fp4_quant_trans(x, packed_fp4, fp8_scale, 1)
|
||||
return packed_fp4, fp8_scale
|
||||
|
||||
def blockscaled_fp4_attn(qlist: Tuple,
|
||||
klist: Tuple,
|
||||
vlist: Tuple,
|
||||
delta_s: torch.Tensor,
|
||||
KL: int,
|
||||
is_causal: bool = False,
|
||||
per_block_mean: bool = True,
|
||||
is_bf16: bool = True,
|
||||
single_level_p_quant: bool = False
|
||||
):
|
||||
softmax_scale = (qlist[0].shape[-1] * 2) ** (-0.5)
|
||||
return fp4attn_cuda.fwd(qlist[0], klist[0], vlist[0], qlist[1], klist[1], vlist[1], delta_s, KL, None, softmax_scale, is_causal, per_block_mean, is_bf16, single_level_p_quant)
|
||||
|
||||
|
||||
def sageattn_blackwell(q, k, v, attn_mask = None, is_causal = False, per_block_mean = True, single_level_p_quant = True, **kwargs):
|
||||
"""
|
||||
SageAttention3 Blackwell kernel for FP4 attention.
|
||||
|
||||
Args:
|
||||
q: Query tensor [B, H, L, D]
|
||||
k: Key tensor [B, H, L, D]
|
||||
v: Value tensor [B, H, L, D]
|
||||
attn_mask: Attention mask (not used)
|
||||
is_causal: Whether to use causal masking
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly
|
||||
(standard per-block FP4 quantization like V, no s_P1).
|
||||
If False (default), use two-level quantization:
|
||||
s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1).
|
||||
**kwargs: Additional arguments (ignored)
|
||||
|
||||
Returns:
|
||||
Output tensor [B, H, L, D]
|
||||
"""
|
||||
if q.size(-1) >= 256:
|
||||
print(f"Unsupported Headdim {q.size(-1)}")
|
||||
return sdpa(q, k, v, is_causal = is_causal)
|
||||
QL = q.size(2)
|
||||
KL = k.size(2)
|
||||
is_bf16 = q.dtype == torch.bfloat16
|
||||
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean)
|
||||
qlist_from_cuda = scale_and_quant_fp4(q)
|
||||
klist_from_cuda = scale_and_quant_fp4_permute(k)
|
||||
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
|
||||
o_fp4 = blockscaled_fp4_attn(
|
||||
qlist_from_cuda,
|
||||
klist_from_cuda,
|
||||
vlist_from_cuda,
|
||||
delta_s,
|
||||
KL,
|
||||
is_causal,
|
||||
per_block_mean,
|
||||
is_bf16,
|
||||
single_level_p_quant
|
||||
)[0][:, :, :QL, :].contiguous()
|
||||
return o_fp4
|
||||
@@ -1 +0,0 @@
|
||||
__version__ = "3.0.0.b1"
|
||||
@@ -1,346 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
// Include these 2 headers instead of torch/extension.h since we don't need all of the torch headers.
|
||||
#include <torch/python.h>
|
||||
#include <torch/nn/functional.h>
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <c10/cuda/CUDAGuard.h>
|
||||
|
||||
#include <cutlass/numeric_types.h>
|
||||
|
||||
#include "params.h"
|
||||
#include "launch.h"
|
||||
#include "static_switch.h"
|
||||
#include "block_config.h"
|
||||
|
||||
#define CHECK_DEVICE(x) TORCH_CHECK(x.is_cuda(), #x " must be on CUDA")
|
||||
#define CHECK_SHAPE(x, ...) TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), #x " must have shape (" #__VA_ARGS__ ")")
|
||||
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
|
||||
|
||||
|
||||
void set_params_fprop(Flash_fwd_params ¶ms,
|
||||
// sizes
|
||||
const size_t b,
|
||||
const size_t seqlen_q,
|
||||
const size_t seqlen_k,
|
||||
const size_t unpadded_seqlen_k,
|
||||
const size_t seqlen_q_rounded,
|
||||
const size_t seqlen_k_rounded,
|
||||
const size_t h,
|
||||
const size_t h_k,
|
||||
const size_t d,
|
||||
const size_t d_rounded,
|
||||
// device pointers
|
||||
const at::Tensor q,
|
||||
const at::Tensor k,
|
||||
const at::Tensor v,
|
||||
const at::Tensor delta_s,
|
||||
at::Tensor out,
|
||||
const at::Tensor sfq,
|
||||
const at::Tensor sfk,
|
||||
const at::Tensor sfv,
|
||||
void *cu_seqlens_q_d,
|
||||
void *cu_seqlens_k_d,
|
||||
void *seqused_k,
|
||||
void *p_d,
|
||||
void *softmax_lse_d,
|
||||
float p_dropout,
|
||||
float softmax_scale,
|
||||
int window_size_left,
|
||||
int window_size_right,
|
||||
bool per_block_mean,
|
||||
bool is_bf16,
|
||||
bool single_level_p_quant=false,
|
||||
bool seqlenq_ngroups_swapped=false) {
|
||||
|
||||
// Reset the parameters
|
||||
params = {};
|
||||
// Set the pointers and strides.
|
||||
params.q_ptr = q.data_ptr();
|
||||
params.k_ptr = k.data_ptr();
|
||||
params.v_ptr = v.data_ptr();
|
||||
params.delta_s_ptr = delta_s.data_ptr();
|
||||
params.sfq_ptr = sfq.data_ptr();
|
||||
params.sfk_ptr = sfk.data_ptr();
|
||||
params.sfv_ptr = sfv.data_ptr();
|
||||
|
||||
// All stride are in elements, not bytes.
|
||||
params.q_row_stride = q.stride(-2) * 2;
|
||||
params.k_row_stride = k.stride(-2) * 2;
|
||||
params.v_row_stride = v.stride(-2) * 2;;
|
||||
params.q_head_stride = q.stride(-3) * 2;
|
||||
params.k_head_stride = k.stride(-3) * 2;
|
||||
params.v_head_stride = v.stride(-3) * 2; // for packed q k v
|
||||
|
||||
params.ds_row_stride = delta_s.stride(-2);
|
||||
params.ds_head_stride = delta_s.stride(-3);
|
||||
|
||||
params.sfq_row_stride = sfq.stride(-2);
|
||||
params.sfk_row_stride = sfk.stride(-2);
|
||||
params.sfv_row_stride = sfv.stride(-2);
|
||||
params.sfq_head_stride = sfq.stride(-3);
|
||||
params.sfk_head_stride = sfk.stride(-3);
|
||||
params.sfv_head_stride = sfv.stride(-3);
|
||||
params.o_ptr = out.data_ptr();
|
||||
params.o_row_stride = out.stride(-2);
|
||||
params.o_head_stride = out.stride(-3);
|
||||
|
||||
if (cu_seqlens_q_d == nullptr) {
|
||||
params.q_batch_stride = q.stride(0) * 2;
|
||||
params.k_batch_stride = k.stride(0) * 2;
|
||||
params.v_batch_stride = v.stride(0) * 2;
|
||||
params.ds_batch_stride = delta_s.stride(0);
|
||||
params.sfq_batch_stride = sfq.stride(0);
|
||||
params.sfk_batch_stride = sfk.stride(0);
|
||||
params.sfv_batch_stride = sfv.stride(0);
|
||||
params.o_batch_stride = out.stride(0);
|
||||
if (seqlenq_ngroups_swapped) {
|
||||
params.q_batch_stride *= seqlen_q;
|
||||
params.o_batch_stride *= seqlen_q;
|
||||
}
|
||||
}
|
||||
|
||||
params.cu_seqlens_q = static_cast<int *>(cu_seqlens_q_d);
|
||||
params.cu_seqlens_k = static_cast<int *>(cu_seqlens_k_d);
|
||||
params.seqused_k = static_cast<int *>(seqused_k);
|
||||
|
||||
// P = softmax(QK^T)
|
||||
params.p_ptr = p_d;
|
||||
|
||||
// Softmax sum
|
||||
params.softmax_lse_ptr = softmax_lse_d;
|
||||
|
||||
// Set the dimensions.
|
||||
params.b = b;
|
||||
params.h = h;
|
||||
params.h_k = h_k;
|
||||
params.h_h_k_ratio = h / h_k;
|
||||
params.seqlen_q = seqlen_q;
|
||||
params.seqlen_k = seqlen_k;
|
||||
params.unpadded_seqlen_k = unpadded_seqlen_k;
|
||||
params.seqlen_q_rounded = seqlen_q_rounded;
|
||||
params.seqlen_k_rounded = seqlen_k_rounded;
|
||||
params.d = d;
|
||||
params.d_rounded = d_rounded;
|
||||
|
||||
params.head_divmod = cutlass::FastDivmod(int(h));
|
||||
|
||||
// Set the different scale values.
|
||||
params.scale_softmax = softmax_scale;
|
||||
params.scale_softmax_log2 = softmax_scale * M_LOG2E;
|
||||
__half scale_softmax_log2_half = __float2half(params.scale_softmax_log2);
|
||||
__half2 scale_softmax_log2_half2 = __half2(scale_softmax_log2_half, scale_softmax_log2_half);
|
||||
params.scale_softmax_log2_half2 = reinterpret_cast<uint32_t&>(scale_softmax_log2_half2);
|
||||
|
||||
// Set this to probability of keeping an element to simplify things.
|
||||
params.p_dropout = 1.f - p_dropout;
|
||||
// Convert p from float to int so we don't have to convert the random uint to float to compare.
|
||||
// [Minor] We want to round down since when we do the comparison we use <= instead of <
|
||||
// params.p_dropout_in_uint = uint32_t(std::floor(params.p_dropout * 4294967295.0));
|
||||
// params.p_dropout_in_uint16_t = uint16_t(std::floor(params.p_dropout * 65535.0));
|
||||
params.p_dropout_in_uint8_t = uint8_t(std::floor(params.p_dropout * 255.0));
|
||||
params.rp_dropout = 1.f / params.p_dropout;
|
||||
params.scale_softmax_rp_dropout = params.rp_dropout * params.scale_softmax;
|
||||
TORCH_CHECK(p_dropout < 1.f);
|
||||
#ifdef FLASHATTENTION_DISABLE_DROPOUT
|
||||
TORCH_CHECK(p_dropout == 0.0f, "This flash attention build does not support dropout.");
|
||||
#endif
|
||||
|
||||
// Causal is the special case where window_size_right == 0 and window_size_left < 0.
|
||||
// Local is the more general case where window_size_right >= 0 or window_size_left >= 0.
|
||||
params.is_causal = window_size_left < 0 && window_size_right == 0;
|
||||
params.per_block_mean = per_block_mean;
|
||||
if (per_block_mean) {
|
||||
params.seqlen_s = seqlen_q;
|
||||
} else {
|
||||
params.seqlen_s = flash::BLOCK_M; // size of BLOCK_M
|
||||
}
|
||||
if (window_size_left < 0 && window_size_right >= 0) { window_size_left = seqlen_k; }
|
||||
if (window_size_left >= 0 && window_size_right < 0) { window_size_right = seqlen_k; }
|
||||
params.window_size_left = window_size_left;
|
||||
params.window_size_right = window_size_right;
|
||||
|
||||
#ifdef FLASHATTENTION_DISABLE_LOCAL
|
||||
TORCH_CHECK(params.is_causal || (window_size_left < 0 && window_size_right < 0),
|
||||
"This flash attention build does not support local attention.");
|
||||
#endif
|
||||
|
||||
params.is_seqlens_k_cumulative = true;
|
||||
params.is_bf16 = is_bf16;
|
||||
params.single_level_p_quant = single_level_p_quant;
|
||||
#ifdef FLASHATTENTION_DISABLE_UNEVEN_K
|
||||
TORCH_CHECK(d == d_rounded, "This flash attention build does not support headdim not being a multiple of 32.");
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool IsBF16>
|
||||
void run_mha_fwd_dispatch_dtype(Flash_fwd_params ¶ms, cudaStream_t stream) {
|
||||
using OType = std::conditional_t<IsBF16, cutlass::bfloat16_t, cutlass::half_t>;
|
||||
if (params.d == 64) {
|
||||
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 64, OType>(params, stream);
|
||||
} else if (params.d == 128) {
|
||||
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 128, OType>(params, stream);
|
||||
}
|
||||
}
|
||||
|
||||
void run_mha_fwd(Flash_fwd_params ¶ms, cudaStream_t stream, bool force_split_kernel = false) {
|
||||
BOOL_SWITCH(params.is_bf16, IsBF16, ([&] {
|
||||
run_mha_fwd_dispatch_dtype<IsBF16>(params, stream);
|
||||
}));
|
||||
}
|
||||
|
||||
std::vector<at::Tensor>
|
||||
mha_fwd(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head_size // 2)
|
||||
const at::Tensor &k, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
|
||||
const at::Tensor &v, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
|
||||
const at::Tensor &sfq,
|
||||
const at::Tensor &sfk,
|
||||
const at::Tensor &sfv,
|
||||
const at::Tensor &delta_s,
|
||||
int unpadded_k,
|
||||
c10::optional<at::Tensor> &out_, // batch_size x seqlen_q x num_heads x head_size
|
||||
const float softmax_scale,
|
||||
bool is_causal,
|
||||
bool per_block_mean,
|
||||
bool is_bf16,
|
||||
bool single_level_p_quant=false // If true, use only per-row scale s_P2 (no per-block s_P1)
|
||||
) {
|
||||
|
||||
auto dprops = at::cuda::getCurrentDeviceProperties();
|
||||
bool is_sm120 = dprops->major == 12 && dprops->minor == 0;
|
||||
TORCH_CHECK(is_sm120, "only supports Blackwell GPUs or newer.");
|
||||
|
||||
auto q_dtype = q.dtype();
|
||||
auto sfq_dtype = sfq.dtype();
|
||||
TORCH_CHECK(q_dtype == torch::kUInt8, "q dtype must be uint8");
|
||||
TORCH_CHECK(k.dtype() == q_dtype, "query and key must have the same dtype");
|
||||
TORCH_CHECK(v.dtype() == q_dtype, "query and value must have the same dtype");
|
||||
CHECK_DEVICE(q); CHECK_DEVICE(k); CHECK_DEVICE(v);
|
||||
|
||||
TORCH_CHECK(sfq_dtype == torch::kFloat8_e4m3fn, "q dtype must be uint8");
|
||||
TORCH_CHECK(sfk.dtype() == sfq_dtype, "query and key must have the same dtype");
|
||||
TORCH_CHECK(sfv.dtype() == sfq_dtype, "query and value must have the same dtype");
|
||||
CHECK_DEVICE(sfq); CHECK_DEVICE(sfk); CHECK_DEVICE(sfv);
|
||||
|
||||
TORCH_CHECK(q.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
TORCH_CHECK(k.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
TORCH_CHECK(v.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
TORCH_CHECK(delta_s.stride(-1) == 1, "Input tensor must have contiguous last dimension");
|
||||
|
||||
TORCH_CHECK(q.is_contiguous(), "Input tensor must be contiguous");
|
||||
TORCH_CHECK(k.is_contiguous(), "Input tensor must be contiguous");
|
||||
TORCH_CHECK(v.is_contiguous(), "Input tensor must be contiguous");
|
||||
|
||||
const auto sizes = q.sizes();
|
||||
auto opts = q.options();
|
||||
const int batch_size = sizes[0];
|
||||
int seqlen_q = sizes[2];
|
||||
int num_heads = sizes[1];
|
||||
const int head_size_og = sizes[3];
|
||||
const int unpacked_head_size = head_size_og * 2;
|
||||
const int seqlen_k = k.size(2);
|
||||
const int num_heads_k = k.size(1);
|
||||
|
||||
TORCH_CHECK(batch_size > 0, "batch size must be postive");
|
||||
TORCH_CHECK(unpacked_head_size <= 256, "FlashAttention forward only supports head dimension at most 256");
|
||||
TORCH_CHECK(num_heads % num_heads_k == 0, "Number of heads in key/value must divide number of heads in query");
|
||||
TORCH_CHECK(num_heads == num_heads_k, "We do not support MQA/GQA yet");
|
||||
|
||||
TORCH_CHECK(unpacked_head_size == 64 || unpacked_head_size == 128 || unpacked_head_size == 256, "Only support head size 64, 128, and 256 for now");
|
||||
|
||||
CHECK_SHAPE(q, batch_size, num_heads, seqlen_q, head_size_og);
|
||||
CHECK_SHAPE(k, batch_size, num_heads_k, seqlen_k, head_size_og);
|
||||
CHECK_SHAPE(v, batch_size, num_heads_k, unpacked_head_size, seqlen_k/2);
|
||||
// CHECK_SHAPE(delta_s, batch_size, num_heads, seqlen_q / 128, seqlen_k);
|
||||
// CHECK_SHAPE(sfq, batch_size, seqlen_q, num_heads, unpacked_head_size);
|
||||
// CHECK_SHAPE(sfk, batch_size, seqlen_k, num_heads_k, unpacked_head_size);
|
||||
// CHECK_SHAPE(sfv, batch_size, unpacked_head_size, num_heads_k, seqlen_k);
|
||||
TORCH_CHECK(unpacked_head_size % 8 == 0, "head_size must be a multiple of 8");
|
||||
|
||||
auto dtype = is_bf16 ? at::ScalarType::BFloat16 : at::ScalarType::Half;
|
||||
at::Tensor out = torch::empty({batch_size, num_heads, seqlen_q, unpacked_head_size}, opts.dtype(dtype));
|
||||
|
||||
auto round_multiple = [](int x, int m) { return (x + m - 1) / m * m; };
|
||||
// const int head_size = round_multiple(head_size_og, 8);
|
||||
// const int head_size_rounded = round_multiple(head_size, 32);
|
||||
const int seqlen_q_rounded = round_multiple(seqlen_q, flash::BLOCK_M);
|
||||
const int seqlen_k_rounded = round_multiple(seqlen_k, flash::BLOCK_N);
|
||||
|
||||
// Otherwise the kernel will be launched from cuda:0 device
|
||||
// Cast to char to avoid compiler warning about narrowing
|
||||
at::cuda::CUDAGuard device_guard{(char)q.get_device()};
|
||||
|
||||
|
||||
|
||||
auto softmax_lse = torch::empty({batch_size, num_heads, seqlen_q}, opts.dtype(at::kFloat));
|
||||
at::Tensor p;
|
||||
|
||||
Flash_fwd_params params;
|
||||
set_params_fprop(params,
|
||||
batch_size,
|
||||
seqlen_q, seqlen_k, unpadded_k,
|
||||
seqlen_q_rounded, seqlen_k_rounded,
|
||||
num_heads, num_heads_k,
|
||||
unpacked_head_size, unpacked_head_size,
|
||||
q, k, v, delta_s, out,
|
||||
sfq, sfk, sfv,
|
||||
/*cu_seqlens_q_d=*/nullptr,
|
||||
/*cu_seqlens_k_d=*/nullptr,
|
||||
/*seqused_k=*/nullptr,
|
||||
nullptr,
|
||||
softmax_lse.data_ptr(),
|
||||
/*p_dropout=*/0.f,
|
||||
softmax_scale,
|
||||
/*window_size_left=*/-1,
|
||||
/*window_size_right=*/is_causal ? 0 : -1,
|
||||
per_block_mean,
|
||||
is_bf16,
|
||||
single_level_p_quant
|
||||
);
|
||||
// TODO: 132 sm count?
|
||||
auto tile_count_semaphore = is_causal ? torch::full({1}, 132, opts.dtype(torch::kInt32)) : torch::empty({1}, opts.dtype(torch::kInt32));
|
||||
params.tile_count_semaphore = tile_count_semaphore.data_ptr<int>();
|
||||
|
||||
if (seqlen_k > 0) {
|
||||
auto stream = at::cuda::getCurrentCUDAStream().stream();
|
||||
run_mha_fwd(params, stream);
|
||||
} else {
|
||||
// If seqlen_k == 0, then we have an empty tensor. We need to set the output to 0.
|
||||
out.zero_();
|
||||
softmax_lse.fill_(std::numeric_limits<float>::infinity());
|
||||
}
|
||||
|
||||
// at::Tensor out_padded = out;
|
||||
// if (head_size_og % 8 != 0) {
|
||||
// out = out.index({"...", torch::indexing::Slice(torch::indexing::None, head_size_og)});
|
||||
// if (out_.has_value()) { out_.value().copy_(out); }
|
||||
// }
|
||||
|
||||
// return {out, q_padded, k_padded, v_padded, out_padded, softmax_lse, p};
|
||||
// cudaDeviceSynchronize();
|
||||
// auto err = cudaGetLastError();
|
||||
// printf("%s\n", cudaGetErrorString(err));
|
||||
return {out, softmax_lse};
|
||||
}
|
||||
|
||||
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
m.doc() = "FlashAttention";
|
||||
m.def("fwd", &mha_fwd, "Forward pass");
|
||||
}
|
||||
@@ -1,28 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
// Centralized block size configuration for sageattn_blackwell kernels
|
||||
// Block sizes for M and N dimensions
|
||||
namespace flash {
|
||||
// Block size for M dimension (query sequence length)
|
||||
static constexpr int BLOCK_M = 128;
|
||||
|
||||
// Block size for N dimension (key/value sequence length)
|
||||
static constexpr int BLOCK_N = 128;
|
||||
}
|
||||
|
||||
@@ -1,60 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
|
||||
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
namespace flash {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<bool Varlen=true>
|
||||
struct BlockInfo {
|
||||
|
||||
template<typename Params>
|
||||
__device__ BlockInfo(const Params ¶ms, const int bidb)
|
||||
: sum_s_q(!Varlen || params.cu_seqlens_q == nullptr ? -1 : params.cu_seqlens_q[bidb])
|
||||
, sum_s_k(!Varlen || params.cu_seqlens_k == nullptr || !params.is_seqlens_k_cumulative ? -1 : params.cu_seqlens_k[bidb])
|
||||
, actual_seqlen_q(!Varlen || params.cu_seqlens_q == nullptr ? params.seqlen_q : params.cu_seqlens_q[bidb + 1] - sum_s_q)
|
||||
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
|
||||
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
|
||||
, seqlen_k_cache(!Varlen || params.cu_seqlens_k == nullptr ? params.seqlen_k : (params.is_seqlens_k_cumulative ? params.cu_seqlens_k[bidb + 1] - sum_s_k : params.cu_seqlens_k[bidb]))
|
||||
, actual_seqlen_k(params.seqused_k ? params.seqused_k[bidb] : seqlen_k_cache + (params.knew_ptr == nullptr ? 0 : params.seqlen_knew))
|
||||
{
|
||||
}
|
||||
|
||||
template <typename index_t>
|
||||
__forceinline__ __device__ index_t q_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
|
||||
return sum_s_q == -1 ? bidb * batch_stride : uint32_t(sum_s_q) * row_stride;
|
||||
}
|
||||
|
||||
template <typename index_t>
|
||||
__forceinline__ __device__ index_t k_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
|
||||
return sum_s_k == -1 ? bidb * batch_stride : uint32_t(sum_s_k) * row_stride;
|
||||
}
|
||||
|
||||
const int sum_s_q;
|
||||
const int sum_s_k;
|
||||
const int actual_seqlen_q;
|
||||
// We have to have seqlen_k_cache declared before actual_seqlen_k, otherwise actual_seqlen_k is set to 0.
|
||||
const int seqlen_k_cache;
|
||||
const int actual_seqlen_k;
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,149 +0,0 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2023 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Blocked Scale configs specific for SM100 BlockScaled MMA
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/layout/matrix.h"
|
||||
|
||||
#include "cute/int_tuple.hpp"
|
||||
#include "cute/atom/mma_traits_sm100.hpp"
|
||||
|
||||
namespace flash {
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
using namespace cute;
|
||||
|
||||
template<int SFVecSize, UMMA::Major major = UMMA::Major::K>
|
||||
struct BlockScaledBasicChunk {
|
||||
|
||||
using Blk_MN = _64;
|
||||
using Blk_SF = _4;
|
||||
|
||||
using SfAtom = Layout< Shape< Shape<_16,_4>, Shape<Int<SFVecSize>, _4>>,
|
||||
Stride<Stride<_16,_4>, Stride< _0, _1>>>;
|
||||
};
|
||||
|
||||
template<int SFVecSize_>
|
||||
struct BlockScaledConfig {
|
||||
// We are creating the SFA and SFB tensors' layouts in the collective since they always have the same layout.
|
||||
// k-major order
|
||||
static constexpr int SFVecSize = SFVecSize_;
|
||||
static constexpr int MMA_NSF = 4; // SFVecSize, MMA_NSF
|
||||
using BlkScaledChunk = BlockScaledBasicChunk<SFVecSize>;
|
||||
using Blk_MN = _64;
|
||||
using Blk_SF = _4;
|
||||
using mnBasicBlockShape = Shape<_16,_4>;
|
||||
using mnBasicBlockStride = Stride<_16,_4>;
|
||||
using kBasicBlockShape = Shape<Int<SFVecSize>, Int<MMA_NSF>>; // SFVecSize, MMA_NSF
|
||||
using kBasicBlockStride = Stride<_0, _1>;
|
||||
using SfAtom = Layout< Shape< mnBasicBlockShape, kBasicBlockShape>,
|
||||
Stride<mnBasicBlockStride, kBasicBlockStride>>;
|
||||
|
||||
using LayoutSF = decltype(blocked_product(SfAtom{},
|
||||
make_layout(
|
||||
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
|
||||
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))));
|
||||
// A single indivisible block will hold 4 scale factors of 64 rows/columns (A/B matrix).
|
||||
// 4 is chosen to make consecutive 32bits of data to have scale factors for only a single row (col). 32bits corresponds to the TMEM word size
|
||||
using Blk_Elems = decltype(Blk_MN{} * Blk_SF{});
|
||||
using sSF_strideMN = decltype(prepend(Blk_Elems{}, mnBasicBlockStride{}));
|
||||
|
||||
|
||||
// The following function is provided for user fill dynamic problem size to the layout_SFA.
|
||||
template < class ProblemShape>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
tile_atom_to_shape_SFQKV(ProblemShape problem_shape) {
|
||||
auto [Seqlen, Dim, HeadNum, Batch] = problem_shape;
|
||||
return tile_to_shape(SfAtom{}, make_shape(Seqlen, Dim, HeadNum, Batch), Step<_2,_1,_3,_4>{});
|
||||
}
|
||||
|
||||
// The following function is provided for user fill dynamic problem size to the layout_SFB.
|
||||
template <class ProblemShape>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
tile_atom_to_shape_SFVt(ProblemShape problem_shape) {
|
||||
auto [Dim, Seqlen, HeadNum, Batch] = problem_shape;
|
||||
return tile_to_shape(SfAtom{}, make_shape(Dim, Seqlen, HeadNum, Batch), Step<_2,_1,_3,_4>{});
|
||||
}
|
||||
|
||||
template<class TiledMma, class TileShape_MNK>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
deduce_smem_layoutSFQ(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
|
||||
|
||||
using sSFQ_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
|
||||
using sSFQ_shapeM = decltype(prepend(size<0>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
|
||||
using sSFQ_strideM = sSF_strideMN;
|
||||
using sSFQ_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<0>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
|
||||
using sSFQ_shape = decltype(make_shape(sSFQ_shapeM{}, sSFQ_shapeK{}));
|
||||
using sSFQ_stride = decltype(make_stride(sSFQ_strideM{}, sSFQ_strideK{}));
|
||||
using SmemLayoutAtomSFQ = decltype(make_layout(sSFQ_shape{}, sSFQ_stride{}));
|
||||
return SmemLayoutAtomSFQ{};
|
||||
}
|
||||
|
||||
template<class TiledMma, class TileShape_MNK>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
deduce_smem_layoutSFKV(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
|
||||
|
||||
using sSFK_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
|
||||
using sSFK_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
|
||||
using sSFK_strideN = sSF_strideMN;
|
||||
using sSFK_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
|
||||
using sSFK_shape = decltype(make_shape(sSFK_shapeN{}, sSFK_shapeK{}));
|
||||
using sSFK_stride = decltype(make_stride(sSFK_strideN{}, sSFK_strideK{}));
|
||||
using SmemLayoutAtomSFK = decltype(make_layout(sSFK_shape{}, sSFK_stride{}));
|
||||
return SmemLayoutAtomSFK{};
|
||||
}
|
||||
|
||||
template<class TiledMma, class TileShape_MNK>
|
||||
CUTE_HOST_DEVICE
|
||||
static constexpr auto
|
||||
deduce_smem_layoutSFVt(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
|
||||
|
||||
using sSFVt_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
|
||||
using sSFVt_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
|
||||
using sSFVt_strideN = sSF_strideMN;
|
||||
using sSFVt_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
|
||||
using sSFVt_shape = decltype(make_shape(sSFVt_shapeN{}, sSFVt_shapeK{}));
|
||||
using sSFVt_stride = decltype(make_stride(sSFVt_strideN{}, sSFVt_strideK{}));
|
||||
using SmemLayoutAtomSFVt = decltype(make_layout(sSFVt_shape{}, sSFVt_stride{}));
|
||||
return SmemLayoutAtomSFVt{};
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,327 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#include "cute/arch/mma_sm120.hpp"
|
||||
#include "cute/atom/mma_traits_sm120.hpp"
|
||||
#include "cute/atom/mma_atom.hpp"
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/float8.h"
|
||||
#include "cutlass/float_subbyte.h"
|
||||
|
||||
namespace cute::SM120::BLOCKSCALED {
|
||||
|
||||
using cutlass::float_e2m1_t;
|
||||
using cutlass::float_ue4m3_t;
|
||||
|
||||
// MMA.SF 16x32x64 TN E2M1 x E2M1 with SF E4M3
|
||||
struct SM120_16x32x64_TN_VS_NVFP4 {
|
||||
using DRegisters = float[16];
|
||||
using ARegisters = uint32_t[4];
|
||||
using BRegisters = uint32_t[8];
|
||||
using CRegisters = float[16];
|
||||
|
||||
static constexpr int SFBits = 32;
|
||||
using RegTypeSF = cute::uint_bit_t<SFBits>;
|
||||
|
||||
using SFARegisters = RegTypeSF[1];
|
||||
using SFBRegisters = RegTypeSF[1];
|
||||
|
||||
CUTE_HOST_DEVICE static void
|
||||
fma(float & d0 , float & d1 , float & d2 , float & d3 ,
|
||||
float & d4 , float & d5 , float & d6 , float & d7 ,
|
||||
float & d8 , float & d9 , float & d10, float & d11,
|
||||
float & d12, float & d13, float & d14, float & d15,
|
||||
uint32_t const& a0 , uint32_t const& a1 , uint32_t const& a2 , uint32_t const& a3 ,
|
||||
uint32_t const& b0 , uint32_t const& b1 , uint32_t const& b2 , uint32_t const& b3 ,
|
||||
uint32_t const& b4 , uint32_t const& b5 , uint32_t const& b6 , uint32_t const& b7 ,
|
||||
float const & c0 , float const & c1 , float const & c2 , float const & c3 ,
|
||||
float const & c4 , float const & c5 , float const & c6 , float const & c7 ,
|
||||
float const & c8 , float const & c9 , float const & c10 , float const & c11,
|
||||
float const & c12, float const & c13, float const & c14, float const & c15,
|
||||
RegTypeSF const& sfa0,
|
||||
RegTypeSF const& sfb0)
|
||||
{
|
||||
static constexpr uint16_t tidA = 0;
|
||||
static constexpr uint16_t bidA = 0;
|
||||
static constexpr uint16_t bidB = 0;
|
||||
static constexpr uint16_t tidB0 = 0;
|
||||
static constexpr uint16_t tidB1 = 1;
|
||||
static constexpr uint16_t tidB2 = 2;
|
||||
static constexpr uint16_t tidB3 = 3;
|
||||
|
||||
#if defined(CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED)
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d0), "=f"(d1), "=f"(d8), "=f"(d9)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b0), "r"(b1),
|
||||
"f"(c0), "f"(c1), "f"(c8), "f"(c9),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB0));
|
||||
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d2), "=f"(d3), "=f"(d10), "=f"(d11)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b2), "r"(b3),
|
||||
"f"(c2), "f"(c3), "f"(c10), "f"(c11),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB1));
|
||||
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d4), "=f"(d5), "=f"(d12), "=f"(d13)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b4), "r"(b5),
|
||||
"f"(c4), "f"(c5), "f"(c12), "f"(c13),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB2));
|
||||
|
||||
asm volatile(
|
||||
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
|
||||
"{%0, %1, %2, %3},"
|
||||
"{%4, %5, %6, %7},"
|
||||
"{%8, %9},"
|
||||
"{%10, %11, %12, %13},"
|
||||
"{%14},"
|
||||
"{%15, %16},"
|
||||
"{%17},"
|
||||
"{%18, %19};\n"
|
||||
: "=f"(d6), "=f"(d7), "=f"(d14), "=f"(d15)
|
||||
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
|
||||
"r"(b6), "r"(b7),
|
||||
"f"(c6), "f"(c7), "f"(c14), "f"(c15),
|
||||
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
|
||||
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB3));
|
||||
#else
|
||||
CUTE_INVALID_CONTROL_PATH("Attempting to use SM120::BLOCKSCALED::SM120_16x8x64_TN_VS without CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED");
|
||||
#endif
|
||||
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace cute::SM120::BLOCKSCALED
|
||||
|
||||
namespace cute {
|
||||
|
||||
// MMA NVFP4 16x32x64 TN
|
||||
template <>
|
||||
struct MMA_Traits<SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4>
|
||||
{
|
||||
// The MMA accepts 4-bit inputs regardless of the types for A and B
|
||||
using ValTypeA = uint4_t;
|
||||
using ValTypeB = uint4_t;
|
||||
|
||||
using ValTypeD = float;
|
||||
using ValTypeC = float;
|
||||
|
||||
using ValTypeSF = cutlass::float_ue4m3_t;
|
||||
constexpr static int SFVecSize = 16;
|
||||
|
||||
using Shape_MNK = Shape<_16,_32,_64>;
|
||||
using ThrID = Layout<_32>;
|
||||
|
||||
// (T32,V32) -> (M16,K64)
|
||||
using ALayout = Layout<Shape <Shape < _4,_8>,Shape < _8,_2, _2>>,
|
||||
Stride<Stride<_128,_1>,Stride<_16,_8,_512>>>;
|
||||
// (T32,V64) -> (N32,K64)
|
||||
using BLayout = Layout<Shape <Shape < _4,_8>,Shape <_8, _2, _4>>,
|
||||
Stride<Stride<_256,_1>,Stride<_32,_1024, _8>>>;
|
||||
// (T32,V64) -> (M16,K64)
|
||||
using SFALayout = Layout<Shape <Shape <_2,_2,_8>,_64>,
|
||||
Stride<Stride<_8,_0,_1>,_16>>;
|
||||
// (T32,V64) -> (N32,K64)
|
||||
using SFBLayout = Layout<Shape <Shape <_4,_8>,_64>,
|
||||
Stride<Stride<_8,_1>, _32>>;
|
||||
// (T32,V16) -> (M16,N32)
|
||||
using CLayout = Layout<Shape <Shape < _4,_8>,Shape < Shape<_2, _4>,_2>>,
|
||||
Stride<Stride<_32,_1>,Stride<Stride<_16, _128>,_8>>>;
|
||||
};
|
||||
|
||||
|
||||
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<0>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<1>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
|
||||
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<1>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<2>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFATensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
|
||||
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
return thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
}
|
||||
|
||||
template <class SFATensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
return make_fragment_like<ValTypeSF>(partition_SFA(sfatensor, thread_mma));
|
||||
}
|
||||
|
||||
template <class SFBTensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
|
||||
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
return thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
}
|
||||
|
||||
template <class SFBTensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
return make_fragment_like<ValTypeSF>(partition_SFB(sfbtensor, thread_mma));
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFA_TV(TiledMma& mma)
|
||||
{
|
||||
// (M,K) -> (M,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto atile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<1>{} , Int<0>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFB_TV(TiledMma& mma)
|
||||
{
|
||||
// (N,K) -> (N,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto btile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<0>{} , Int<1>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
} // namespace cute
|
||||
@@ -1,222 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cutlass/cutlass.h>
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include "cutlass/gemm/collective/collective_builder.hpp"
|
||||
#include "named_barrier.h"
|
||||
#include "utils.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <typename Ktraits>
|
||||
struct CollectiveEpilogueFwd{
|
||||
|
||||
using Element = typename Ktraits::ElementOut;
|
||||
static constexpr int kBlockM = Ktraits::kBlockM;
|
||||
static constexpr int kBlockN = Ktraits::kBlockN;
|
||||
static constexpr int kHeadDim = Ktraits::kHeadDim;
|
||||
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
|
||||
static constexpr int kNWarps = Ktraits::kNWarps;
|
||||
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
|
||||
static constexpr int NumMmaThreads = kNThreads - cutlass::NumThreadsPerWarpGroup;
|
||||
|
||||
using GmemTiledCopyOTMA = cute::SM90_TMA_STORE;
|
||||
|
||||
// These are for storing the output tensor without TMA (e.g., for setting output to zero)
|
||||
static constexpr int kGmemElemsPerLoad = sizeof(cute::uint128_t) / sizeof(Element);
|
||||
static_assert(kHeadDim % kGmemElemsPerLoad == 0, "kHeadDim must be a multiple of kGmemElemsPerLoad");
|
||||
static constexpr int kGmemThreadsPerRow = kHeadDim / kGmemElemsPerLoad;
|
||||
static_assert(NumMmaThreads % kGmemThreadsPerRow == 0, "NumMmaThreads must be a multiple of kGmemThreadsPerRow");
|
||||
using GmemLayoutAtom = Layout<Shape <Int<NumMmaThreads / kGmemThreadsPerRow>, Int<kGmemThreadsPerRow>>,
|
||||
Stride<Int<kGmemThreadsPerRow>, _1>>;
|
||||
using GmemTiledCopyO = decltype(
|
||||
make_tiled_copy(Copy_Atom<DefaultCopy, Element>{},
|
||||
GmemLayoutAtom{},
|
||||
Layout<Shape<_1, Int<kGmemElemsPerLoad>>>{})); // Val layout, 8 or 16 vals per store
|
||||
|
||||
using SmemLayoutO = typename Ktraits::SmemLayoutO;
|
||||
|
||||
using SmemCopyAtomO = Copy_Atom<SM90_U32x2_STSM_N, Element>;
|
||||
using SharedStorage = cute::array_aligned<Element, cute::cosize_v<SmemLayoutO>>;
|
||||
|
||||
using ShapeO = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen_q, d, head, batch)
|
||||
using StrideO = cute::Stride<int64_t, _1, int64_t, int64_t>;
|
||||
using StrideLSE = cute::Stride<_1, int64_t, int64_t>; // (seqlen_q, head, batch)
|
||||
|
||||
using TMA_O = decltype(make_tma_copy(
|
||||
GmemTiledCopyOTMA{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element*>(nullptr)), repeat_like(StrideO{}, int32_t(0)), StrideO{}),
|
||||
SmemLayoutO{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{})); // no mcast for O
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
Element* ptr_O;
|
||||
ShapeO const shape_O;
|
||||
StrideO const stride_O;
|
||||
float* ptr_LSE;
|
||||
StrideLSE const stride_LSE;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
Element* ptr_O;
|
||||
ShapeO const shape_O;
|
||||
StrideO const stride_O;
|
||||
float* ptr_LSE;
|
||||
StrideLSE const stride_LSE;
|
||||
TMA_O tma_store_O;
|
||||
};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
Tensor mO = make_tensor(make_gmem_ptr(args.ptr_O), args.shape_O, args.stride_O);
|
||||
TMA_O tma_store_O = make_tma_copy(
|
||||
GmemTiledCopyOTMA{},
|
||||
mO,
|
||||
SmemLayoutO{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{}); // no mcast for O
|
||||
return {args.ptr_O, args.shape_O, args.stride_O, args.ptr_LSE, args.stride_LSE, tma_store_O};
|
||||
}
|
||||
|
||||
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
|
||||
CUTLASS_DEVICE
|
||||
static void prefetch_tma_descriptors(Params const& epilogue_params) {
|
||||
cute::prefetch_tma_descriptor(epilogue_params.tma_store_O.get_tma_descriptor());
|
||||
}
|
||||
|
||||
template <typename SharedStorage, typename FrgTensorO, typename TiledMma>
|
||||
CUTLASS_DEVICE void
|
||||
mma_store(
|
||||
SharedStorage& shared_storage,
|
||||
TiledMma tiled_mma,
|
||||
FrgTensorO const& tOrO,
|
||||
int thread_idx
|
||||
){
|
||||
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
|
||||
auto smem_tiled_copy_O = make_tiled_copy_C(SmemCopyAtomO{}, tiled_mma);
|
||||
auto smem_thr_copy_O = smem_tiled_copy_O.get_thread_slice(thread_idx);
|
||||
constexpr int numel = decltype(size(tOrO))::value;
|
||||
cutlass::NumericArrayConverter<Element, float, numel> convert_op;
|
||||
// HACK: this requires tensor to be "contiguous"
|
||||
auto frag = convert_op(*reinterpret_cast<const cutlass::Array<float, numel> *>(tOrO.data()));
|
||||
auto tOrO_out = make_tensor(make_rmem_ptr<Element>(&frag), tOrO.layout());
|
||||
Tensor taccOrO = smem_thr_copy_O.retile_S(tOrO_out); // ((Atom,AtomNum), MMA_M, MMA_N)
|
||||
Tensor taccOsO = smem_thr_copy_O.partition_D(sO); // ((Atom,AtomNum),PIPE_M,PIPE_N)
|
||||
cute::copy(smem_tiled_copy_O, taccOrO, taccOsO);
|
||||
cutlass::arch::fence_view_async_shared(); // ensure smem writes are visible to TMA
|
||||
}
|
||||
|
||||
template<typename SharedStorage, typename Params, typename WorkTileInfo, typename SchedulerParams>
|
||||
CUTLASS_DEVICE void
|
||||
tma_store(
|
||||
SharedStorage& shared_storage,
|
||||
Params const& epilogue_params,
|
||||
WorkTileInfo work_tile_info,
|
||||
SchedulerParams const& scheduler_params,
|
||||
int thread_idx
|
||||
) {
|
||||
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
|
||||
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
|
||||
Tensor mO = epilogue_params.tma_store_O.get_tma_tensor(epilogue_params.shape_O);
|
||||
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
|
||||
auto block_tma_O = epilogue_params.tma_store_O.get_slice(_0{});
|
||||
Tensor tOgO = block_tma_O.partition_D(gO); // (TMA, TMA_M, TMA_K)
|
||||
Tensor tOsO = block_tma_O.partition_S(sO); // (TMA, TMA_M, TMA_K)
|
||||
|
||||
// auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
|
||||
// Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
|
||||
// Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
|
||||
|
||||
// Tensor caccO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{}));
|
||||
// auto thread_mma = tiled_mma.get_thread_slice(thread_idx);
|
||||
// Tensor taccOcO = thread_mma.partition_C(caccO); // (MMA,MMA_M,MMA_K)
|
||||
// static_assert(decltype(size<0, 0>(taccOcO))::value == 2);
|
||||
// static_assert(decltype(size<0, 1>(taccOcO))::value == 2);
|
||||
// // // // taccOcO has shape ((2, 2, V), MMA_M, MMA_K), we only take only the row indices.
|
||||
// Tensor taccOcO_row = taccOcO(make_coord(_0{}, _), _, _0{});
|
||||
// CUTE_STATIC_ASSERT_V(size(lse) == size(taccOcO_row)); // MMA_M
|
||||
// if (get<1>(taccOcO_row(_0{})) == 0) {
|
||||
// #pragma unroll
|
||||
// for (int mi = 0; mi < size(lse); ++mi) {
|
||||
// const int row = get<0>(taccOcO_row(mi));
|
||||
// if (row < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(row) = lse(mi); }
|
||||
// }
|
||||
// }
|
||||
|
||||
// if (cutlass::canonical_warp_idx_sync() == kNWarps - 1) {
|
||||
// cutlass::arch::NamedBarrier::sync(NumMmaThreads + cutlass::NumThreadsPerWarp,
|
||||
// static_cast<uint32_t>(FP4NamedBarriers::EpilogueBarrier));
|
||||
// int const lane_predicate = cute::elect_one_sync();
|
||||
// if (lane_predicate) {
|
||||
// cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
|
||||
// tma_store_arrive();
|
||||
// }
|
||||
// }
|
||||
cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
|
||||
tma_store_arrive();
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
store_tail() {
|
||||
tma_store_wait<0>();
|
||||
}
|
||||
|
||||
// Write 0 to output and -inf to LSE
|
||||
CUTLASS_DEVICE void
|
||||
store_zero(
|
||||
Params const& epilogue_params,
|
||||
int thread_idx,
|
||||
cute::tuple<int32_t, int32_t, int32_t> const& block_coord
|
||||
) {
|
||||
auto [m_block, bidh, bidb] = block_coord;
|
||||
Tensor mO = make_tensor(make_gmem_ptr(epilogue_params.ptr_O), epilogue_params.shape_O, epilogue_params.stride_O);
|
||||
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
|
||||
auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
|
||||
Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
|
||||
Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
|
||||
|
||||
GmemTiledCopyO gmem_tiled_copy_O;
|
||||
auto gmem_thr_copy_O = gmem_tiled_copy_O.get_thread_slice(thread_idx);
|
||||
Tensor tOgO = gmem_thr_copy_O.partition_D(gO);
|
||||
Tensor tOrO = make_fragment_like(tOgO);
|
||||
clear(tOrO);
|
||||
// Construct identity layout for sO
|
||||
Tensor cO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{})); // (BLK_M,BLK_K) -> (blk_m,blk_k)
|
||||
// Repeat the partitioning with identity layouts
|
||||
Tensor tOcO = gmem_thr_copy_O.partition_D(cO);
|
||||
Tensor tOpO = make_tensor<bool>(make_shape(size<2>(tOgO)));
|
||||
#pragma unroll
|
||||
for (int k = 0; k < size(tOpO); ++k) { tOpO(k) = get<1>(tOcO(_0{}, _0{}, k)) < get<1>(epilogue_params.shape_O); }
|
||||
// Clear_OOB_K must be false since we don't want to write zeros to gmem
|
||||
flash::copy</*Is_even_MN=*/false, /*Is_even_K=*/false, /*Clear_OOB_MN=*/false, /*Clear_OOB_K=*/false>(
|
||||
gmem_tiled_copy_O, tOrO, tOgO, tOcO, tOpO, get<0>(epilogue_params.shape_O) - m_block * kBlockM
|
||||
);
|
||||
static_assert(kBlockM <= NumMmaThreads);
|
||||
if (thread_idx < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(thread_idx) = INFINITY; }
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,202 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cute/algorithm/copy.hpp"
|
||||
#include "cute/atom/mma_atom.hpp"
|
||||
#include "cutlass/gemm/collective/collective_builder.hpp"
|
||||
#include "cute/tensor.hpp"
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/layout/layout.h"
|
||||
#include "cutlass/numeric_types.h"
|
||||
#include "cutlass/pipeline/pipeline.hpp"
|
||||
|
||||
#include "blockscaled_layout.h"
|
||||
#include "cute_extension.h"
|
||||
#include "named_barrier.h"
|
||||
using namespace cute;
|
||||
|
||||
template <
|
||||
int kStages,
|
||||
int EpiStages,
|
||||
typename Element,
|
||||
typename ElementSF,
|
||||
typename OutputType,
|
||||
typename SmemLayoutQ,
|
||||
typename SmemLayoutK,
|
||||
typename SmemLayoutV,
|
||||
typename SmemLayoutDS,
|
||||
typename SmemLayoutO,
|
||||
typename SmemLayoutSFQ,
|
||||
typename SmemLayoutSFK,
|
||||
typename SmemLayoutSFV
|
||||
>
|
||||
struct SharedStorageQKVOwithSF : cute::aligned_struct<128, _0>{
|
||||
|
||||
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutQ>> smem_q;
|
||||
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutK>> smem_k;
|
||||
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFQ>> smem_SFQ;
|
||||
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFK>> smem_SFK;
|
||||
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFV>> smem_SFV;
|
||||
alignas(1024) cute::ArrayEngine<float, cute::cosize_v<SmemLayoutDS>> smem_ds;
|
||||
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutV>> smem_v;
|
||||
alignas(1024) cute::ArrayEngine<OutputType, cute::cosize_v<SmemLayoutO>> smem_o;
|
||||
|
||||
struct {
|
||||
alignas(16) typename cutlass::PipelineTmaAsync<1>::SharedStorage pipeline_q;
|
||||
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_k;
|
||||
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_v;
|
||||
alignas(16) typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>::SharedStorage barrier_o;
|
||||
int tile_count_semaphore;
|
||||
};
|
||||
};
|
||||
|
||||
template <
|
||||
int kHeadDim_,
|
||||
int kBlockM_,
|
||||
int kBlockN_,
|
||||
int kStages_,
|
||||
int kClusterM_,
|
||||
bool BlockMean_,
|
||||
typename ElementPairType_ = cutlass::nv_float4_t<cutlass::float_e2m1_t>,
|
||||
typename ElementOut_ = cutlass::bfloat16_t
|
||||
>
|
||||
struct Flash_fwd_kernel_traits {
|
||||
static constexpr int kBlockM = kBlockM_;
|
||||
static constexpr int kBlockN = kBlockN_;
|
||||
static constexpr int kHeadDim = kHeadDim_;
|
||||
static constexpr bool BlockMean = BlockMean_;
|
||||
static constexpr bool SmoothQ = true;
|
||||
static_assert(kHeadDim % 32 == 0);
|
||||
static_assert(kBlockM == 64 || kBlockM == 128);
|
||||
static constexpr int kNWarps = kBlockM == 128 ? 12 : 8;
|
||||
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
|
||||
static constexpr int kClusterM = kClusterM_;
|
||||
static constexpr int kStages = kStages_;
|
||||
static constexpr int EpiStages = 1;
|
||||
static constexpr int NumSFQK = kHeadDim / 16;
|
||||
static constexpr int NumSFPV = kBlockN / 16;
|
||||
using ElementSF = cutlass::float_ue4m3_t;
|
||||
using Element = cutlass::float_e2m1_t;
|
||||
using ElementAccum = float;
|
||||
using ElementOut = ElementOut_;
|
||||
using index_t = int64_t;
|
||||
static constexpr auto SFVectorSize = 16;
|
||||
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
|
||||
using ClusterShape_MNK = Shape<_1, _1, _1>;
|
||||
using PermTileM = decltype(cute::min(size<0>(TileShape_MNK{}), _128{}));
|
||||
using PermTileN = _32;
|
||||
using PermTileK = Int<kHeadDim>;
|
||||
|
||||
using ElementQMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
|
||||
using ElementKMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
|
||||
|
||||
using AtomLayoutMNK = std::conditional_t<kBlockM == 128,
|
||||
Layout<Shape<_8, _1, _1>>,
|
||||
Layout<Shape<_4, _1, _1>>
|
||||
>;
|
||||
using TiledMmaQK = decltype(cute::make_tiled_mma(
|
||||
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
|
||||
AtomLayoutMNK{},
|
||||
Tile<PermTileM, PermTileN, PermTileK>{}
|
||||
));
|
||||
|
||||
using TiledMmaPV = decltype(cute::make_tiled_mma(
|
||||
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
|
||||
AtomLayoutMNK{},
|
||||
Tile<PermTileM, _32, PermTileK>{}
|
||||
));
|
||||
|
||||
static constexpr int MMA_NSF = size<2>(typename TiledMmaQK::AtomShape_MNK{}) / SFVectorSize;
|
||||
|
||||
using GmemTiledCopy = SM90_TMA_LOAD;
|
||||
using GmemTiledCopySF = SM90_TMA_LOAD;
|
||||
|
||||
using SmemLayoutAtomQ = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutAtomK = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutAtomV = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutAtomVt = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<1>(TileShape_MNK{}))>());
|
||||
using SmemLayoutQ = decltype(tile_to_shape(SmemLayoutAtomQ{}, select<0, 2>(TileShape_MNK{})));
|
||||
using SmemLayoutK =
|
||||
decltype(tile_to_shape(SmemLayoutAtomK{},
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
|
||||
using SmemLayoutV =
|
||||
decltype(tile_to_shape(SmemLayoutAtomV{},
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
|
||||
using SmemLayoutVt =
|
||||
decltype(tile_to_shape(SmemLayoutAtomVt{},
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
|
||||
using SmemLayoutAtomDS = Layout<Shape<Int<kBlockM>, Int<kBlockN>>, Stride<_0, _1>>;
|
||||
using SmemLayoutDS =
|
||||
decltype(tile_to_shape(SmemLayoutAtomDS{},
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
|
||||
|
||||
using SmemCopyAtomQ = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
|
||||
using SmemCopyAtomKV = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
|
||||
using SmemCopyAtomSF = Copy_Atom<UniversalCopy<ElementSF>, ElementSF>;
|
||||
using SmemCopyAtomDS = Copy_Atom<UniversalCopy<float>, float>;
|
||||
|
||||
using BlkScaledConfig = flash::BlockScaledConfig<SFVectorSize>;
|
||||
using LayoutSF = typename BlkScaledConfig::LayoutSF;
|
||||
using SfAtom = typename BlkScaledConfig::SfAtom;
|
||||
using SmemLayoutAtomSFQ = decltype(BlkScaledConfig::deduce_smem_layoutSFQ(TiledMmaQK{}, TileShape_MNK{}));
|
||||
using SmemLayoutAtomSFK = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaQK{}, TileShape_MNK{}));
|
||||
using SmemLayoutAtomSFV = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaPV{}, TileShape_MNK{}));
|
||||
using SmemLayoutAtomSFVt = decltype(BlkScaledConfig::deduce_smem_layoutSFVt(TiledMmaPV{}, Shape<Int<kBlockM>, Int<kHeadDim>, Int<kBlockN>>{}));
|
||||
using LayoutSFP = decltype(
|
||||
make_layout(
|
||||
make_shape(make_shape(_16{}, _4{}), _1{}, Int<kBlockN / 64>{}),
|
||||
make_stride(make_stride(_0{}, _1{}), _0{}, _4{})
|
||||
)
|
||||
);
|
||||
using LayoutP = decltype(
|
||||
make_layout(
|
||||
make_shape(make_shape(_8{}, _2{}, _2{}), _1{}, Int<kBlockN / 64>{}),
|
||||
make_stride(make_stride(_1{}, _8{}, _16{}), _0{}, _32{})
|
||||
)
|
||||
);
|
||||
using SmemLayoutSFQ = decltype(make_layout(
|
||||
shape(SmemLayoutAtomSFQ{}),
|
||||
stride(SmemLayoutAtomSFQ{})
|
||||
));
|
||||
using SmemLayoutSFK = decltype(make_layout(
|
||||
append(shape(SmemLayoutAtomSFK{}), Int<kStages>{}),
|
||||
append(stride(SmemLayoutAtomSFK{}), size(filter_zeros(SmemLayoutAtomSFK{})))
|
||||
));
|
||||
using SmemLayoutSFV = decltype(make_layout(
|
||||
append(shape(SmemLayoutAtomSFV{}), Int<kStages>{}),
|
||||
append(stride(SmemLayoutAtomSFV{}), size(filter_zeros(SmemLayoutAtomSFV{})))
|
||||
));
|
||||
using SmemLayoutSFVt = decltype(make_layout(
|
||||
append(shape(SmemLayoutAtomSFVt{}), Int<kStages>{}),
|
||||
append(stride(SmemLayoutAtomSFVt{}), size(filter_zeros(SmemLayoutAtomSFVt{})))
|
||||
));
|
||||
|
||||
using SmemLayoutAtomO = decltype(cutlass::gemm::collective::detail::ss_smem_selector<GMMA::Major::K, ElementOut,
|
||||
decltype(cute::get<0>(TileShape_MNK{})), decltype(cute::get<2>(TileShape_MNK{}))>());
|
||||
using SmemLayoutO = decltype(tile_to_shape(SmemLayoutAtomO{}, select<0, 2>(TileShape_MNK{}), Step<_1, _2>{}));
|
||||
using SharedStorage = SharedStorageQKVOwithSF<kStages, EpiStages, Element, ElementSF, ElementOut,
|
||||
SmemLayoutQ, SmemLayoutK, SmemLayoutV, SmemLayoutDS,
|
||||
SmemLayoutO, SmemLayoutSFQ, SmemLayoutSFK, SmemLayoutSFVt>;
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<kStages>;
|
||||
using PipelineState = typename cutlass::PipelineState<kStages>;
|
||||
using MainloopPipelineQ = cutlass::PipelineTmaAsync<1>;
|
||||
using PipelineParamsQ = typename MainloopPipelineQ::Params;
|
||||
using PipelineStateQ = typename cutlass::PipelineState<1>;
|
||||
using EpilogueBarrier = typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>;
|
||||
};
|
||||
|
||||
@@ -1,204 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include <cutlass/cutlass.h>
|
||||
#include <cutlass/arch/reg_reconfig.h>
|
||||
#include <cutlass/array.h>
|
||||
#include <cutlass/numeric_types.h>
|
||||
#include <cutlass/numeric_conversion.h>
|
||||
#include "cutlass/pipeline/pipeline.hpp"
|
||||
|
||||
#include "params.h"
|
||||
#include "utils.h"
|
||||
#include "tile_scheduler.h"
|
||||
#include "mainloop_tma_ws.h"
|
||||
#include "epilogue_tma_ws.h"
|
||||
#include "named_barrier.h"
|
||||
#include "softmax_fused.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <typename Ktraits, bool Is_causal, typename TileScheduler>
|
||||
__global__ void __launch_bounds__(Ktraits::kNWarps * cutlass::NumThreadsPerWarp, 1)
|
||||
compute_attn_ws(CUTE_GRID_CONSTANT Flash_fwd_params const params,
|
||||
CUTE_GRID_CONSTANT typename CollectiveMainloopFwd<Ktraits, Is_causal>::Params const mainloop_params,
|
||||
CUTE_GRID_CONSTANT typename CollectiveEpilogueFwd<Ktraits>::Params const epilogue_params,
|
||||
CUTE_GRID_CONSTANT typename TileScheduler::Params const scheduler_params
|
||||
) {
|
||||
|
||||
using Element = typename Ktraits::Element;
|
||||
using ElementAccum = typename Ktraits::ElementAccum;
|
||||
using SoftType = ElementAccum;
|
||||
using TileShape_MNK = typename Ktraits::TileShape_MNK;
|
||||
using ClusterShape = typename Ktraits::ClusterShape_MNK;
|
||||
|
||||
static constexpr int NumMmaThreads = size(typename Ktraits::TiledMmaQK{});
|
||||
static constexpr int NumCopyThreads = cutlass::NumThreadsPerWarpGroup;
|
||||
static constexpr int kBlockM = Ktraits::kBlockM;
|
||||
|
||||
using CollectiveMainloop = CollectiveMainloopFwd<Ktraits, Is_causal>;
|
||||
using CollectiveEpilogue = CollectiveEpilogueFwd<Ktraits>;
|
||||
|
||||
using MainloopPipeline = typename Ktraits::MainloopPipeline;
|
||||
using PipelineParams = typename MainloopPipeline::Params;
|
||||
using PipelineState = typename MainloopPipeline::PipelineState;
|
||||
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
|
||||
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
|
||||
using PipelineStateQ = typename Ktraits::PipelineStateQ;
|
||||
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
|
||||
|
||||
|
||||
enum class WarpGroupRole {
|
||||
Producer = 0,
|
||||
Consumer0 = 1,
|
||||
Consumer1 = 2
|
||||
};
|
||||
enum class ProducerWarpRole {
|
||||
Mainloop = 0,
|
||||
Epilogue = 1,
|
||||
Warp2 = 2,
|
||||
Warp3 = 3
|
||||
};
|
||||
|
||||
extern __shared__ char shared_memory[];
|
||||
auto &shared_storage = *reinterpret_cast<typename Ktraits::SharedStorage*>(shared_memory);
|
||||
|
||||
int const lane_predicate = cute::elect_one_sync();
|
||||
int const warp_idx = cutlass::canonical_warp_idx_sync();
|
||||
int warp_group_idx = cutlass::canonical_warp_group_idx();
|
||||
int const warp_group_thread_idx = threadIdx.x % cutlass::NumThreadsPerWarpGroup;
|
||||
int warp_idx_in_warp_group = warp_idx % cutlass::NumWarpsPerWarpGroup;
|
||||
auto warp_group_role = WarpGroupRole(warp_group_idx);
|
||||
auto producer_warp_role = ProducerWarpRole(warp_idx_in_warp_group);
|
||||
|
||||
// Issue Tma Descriptor Prefetch from a single thread
|
||||
if (warp_idx == 0 && lane_predicate) {
|
||||
CollectiveMainloop::prefetch_tma_descriptors(mainloop_params);
|
||||
CollectiveEpilogue::prefetch_tma_descriptors(epilogue_params);
|
||||
}
|
||||
|
||||
// Obtain warp index
|
||||
|
||||
PipelineParams pipeline_params_v;
|
||||
pipeline_params_v.transaction_bytes = CollectiveMainloop::TmaTransactionBytesV;
|
||||
pipeline_params_v.role = warp_group_role == WarpGroupRole::Producer
|
||||
? MainloopPipeline::ThreadCategory::Producer
|
||||
: MainloopPipeline::ThreadCategory::Consumer;
|
||||
pipeline_params_v.is_leader = warp_group_thread_idx == 0;
|
||||
pipeline_params_v.num_consumers = NumMmaThreads;
|
||||
|
||||
PipelineParams pipeline_params_k;
|
||||
pipeline_params_k.transaction_bytes = CollectiveMainloop::TmaTransactionBytesK;
|
||||
pipeline_params_k.role = warp_group_role == WarpGroupRole::Producer
|
||||
? MainloopPipeline::ThreadCategory::Producer
|
||||
: MainloopPipeline::ThreadCategory::Consumer;
|
||||
pipeline_params_k.is_leader = warp_group_thread_idx == 0;
|
||||
pipeline_params_k.num_consumers = NumMmaThreads;
|
||||
|
||||
PipelineParamsQ pipeline_params_q;
|
||||
pipeline_params_q.transaction_bytes = CollectiveMainloop::TmaTransactionBytesQ;
|
||||
pipeline_params_q.role = warp_group_role == WarpGroupRole::Producer
|
||||
? MainloopPipelineQ::ThreadCategory::Producer
|
||||
: MainloopPipelineQ::ThreadCategory::Consumer;
|
||||
pipeline_params_q.is_leader = warp_group_thread_idx == 0;
|
||||
pipeline_params_q.num_consumers = NumMmaThreads;
|
||||
|
||||
// We're counting on pipeline_k to call cutlass::arch::fence_barrier_init();
|
||||
MainloopPipelineQ pipeline_q(shared_storage.pipeline_q, pipeline_params_q, ClusterShape{});
|
||||
MainloopPipeline pipeline_k(shared_storage.pipeline_k, pipeline_params_k, ClusterShape{});
|
||||
MainloopPipeline pipeline_v(shared_storage.pipeline_v, pipeline_params_v, ClusterShape{});
|
||||
|
||||
uint32_t epilogue_barrier_group_size_list[2] = {cutlass::NumThreadsPerWarp, NumMmaThreads};
|
||||
typename EpilogueBarrier::Params params_epilogue_barrier;
|
||||
params_epilogue_barrier.group_id = (warp_group_role == WarpGroupRole::Producer);
|
||||
params_epilogue_barrier.group_size_list = epilogue_barrier_group_size_list;
|
||||
EpilogueBarrier barrier_o(shared_storage.barrier_o, params_epilogue_barrier);
|
||||
|
||||
CollectiveMainloop collective_mainloop;
|
||||
CollectiveEpilogue collective_epilogue;
|
||||
__syncthreads();
|
||||
|
||||
if (warp_group_role == WarpGroupRole::Producer) {
|
||||
cutlass::arch::warpgroup_reg_dealloc<24>();
|
||||
TileScheduler scheduler;
|
||||
|
||||
if (producer_warp_role == ProducerWarpRole::Mainloop) { // Load Q, K, V
|
||||
PipelineStateQ smem_pipe_write_q = cutlass::make_producer_start_state<MainloopPipelineQ>();
|
||||
PipelineState smem_pipe_write_k = cutlass::make_producer_start_state<MainloopPipeline>();
|
||||
PipelineState smem_pipe_write_v = cutlass::make_producer_start_state<MainloopPipeline>();
|
||||
|
||||
int work_idx = 0;
|
||||
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
|
||||
int tile_count_semaphore = 0;
|
||||
collective_mainloop.load(mainloop_params, scheduler_params,
|
||||
pipeline_q, pipeline_k, pipeline_v,
|
||||
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v,
|
||||
shared_storage, work_tile_info, work_idx, tile_count_semaphore);
|
||||
}
|
||||
collective_mainloop.load_tail(pipeline_q, pipeline_k, pipeline_v,
|
||||
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v);
|
||||
} else if (producer_warp_role == ProducerWarpRole::Epilogue) {
|
||||
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
|
||||
barrier_o.wait();
|
||||
collective_epilogue.tma_store(shared_storage, epilogue_params, work_tile_info, scheduler_params, threadIdx.x);
|
||||
collective_epilogue.store_tail();
|
||||
barrier_o.arrive();
|
||||
}
|
||||
|
||||
}
|
||||
} else if (warp_group_role == WarpGroupRole::Consumer0 || warp_group_role == WarpGroupRole::Consumer1) {
|
||||
cutlass::arch::warpgroup_reg_alloc<232>();
|
||||
typename Ktraits::TiledMmaPV tiled_mma_pv;
|
||||
TileScheduler scheduler{};
|
||||
PipelineState smem_pipe_read_k, smem_pipe_read_v;
|
||||
PipelineStateQ smem_pipe_read_q;
|
||||
|
||||
int work_idx = 0;
|
||||
|
||||
CUTLASS_PRAGMA_NO_UNROLL
|
||||
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
|
||||
// Attention output (GEMM-II) accumulator.
|
||||
Tensor tOrO = partition_fragment_C(tiled_mma_pv, select<0, 2>(TileShape_MNK{}));
|
||||
// flash::Softmax<2 * (2 * kBlockM / NumMmaThreads)> softmax;
|
||||
// Pass single_level_p_quant flag to control P quantization mode
|
||||
flash::SoftmaxFused<2 * (2 * kBlockM / NumMmaThreads)> softmax_fused(params.single_level_p_quant);
|
||||
auto block_coord = work_tile_info.get_block_coord(scheduler_params);
|
||||
auto [m_block, bidh, bidb] = block_coord;
|
||||
|
||||
int n_block_max = collective_mainloop.get_n_block_max(mainloop_params, m_block);
|
||||
if (Is_causal && n_block_max <= 0) { // We exit early and write 0 to gO and -inf to gLSE.
|
||||
collective_epilogue.store_zero(epilogue_params, threadIdx.x - NumCopyThreads, block_coord);
|
||||
continue;
|
||||
}
|
||||
|
||||
collective_mainloop.mma(mainloop_params, pipeline_q, pipeline_k, pipeline_v, smem_pipe_read_q, smem_pipe_read_k, smem_pipe_read_v,
|
||||
tOrO, softmax_fused, n_block_max, threadIdx.x - NumCopyThreads, work_idx, m_block, shared_storage);
|
||||
barrier_o.wait();
|
||||
collective_epilogue.mma_store(shared_storage, tiled_mma_pv, tOrO, threadIdx.x - NumCopyThreads);
|
||||
barrier_o.arrive();
|
||||
++work_idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace flash
|
||||
@@ -1,114 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include "cutlass/cluster_launch.hpp"
|
||||
|
||||
#include "static_switch.h"
|
||||
#include "params.h"
|
||||
#include "tile_scheduler.h"
|
||||
#include "kernel_ws.h"
|
||||
#include "kernel_traits.h"
|
||||
#include "block_config.h"
|
||||
|
||||
|
||||
template<typename Kernel_traits, bool Is_causal>
|
||||
void run_flash_fwd(Flash_fwd_params ¶ms, cudaStream_t stream) {
|
||||
using Element = typename Kernel_traits::Element;
|
||||
using ElementSF = typename Kernel_traits::ElementSF;
|
||||
using ElementOut = typename Kernel_traits::ElementOut;
|
||||
using TileShape_MNK = typename Kernel_traits::TileShape_MNK;
|
||||
using ClusterShape = typename Kernel_traits::ClusterShape_MNK;
|
||||
using CollectiveMainloop = flash::CollectiveMainloopFwd<Kernel_traits, Is_causal>;
|
||||
using CollectiveEpilogue = flash::CollectiveEpilogueFwd<Kernel_traits>;
|
||||
// using Scheduler = flash::SingleTileScheduler;
|
||||
using Scheduler = flash::StaticPersistentTileScheduler;
|
||||
typename CollectiveMainloop::Params mainloop_params =
|
||||
CollectiveMainloop::to_underlying_arguments({
|
||||
static_cast<Element const*>(params.q_ptr),
|
||||
{params.seqlen_q, params.d, params.h, params.b}, // shape_Q
|
||||
{params.q_row_stride, _1{}, params.q_head_stride, params.q_batch_stride}, // stride_Q
|
||||
static_cast<Element const*>(params.k_ptr),
|
||||
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_K
|
||||
{params.k_row_stride, _1{}, params.k_head_stride, params.k_batch_stride}, // stride_K
|
||||
{params.unpadded_seqlen_k, params.d, params.h_k, params.b}, // shape_K
|
||||
static_cast<Element const*>(params.v_ptr),
|
||||
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_Vt
|
||||
{params.v_row_stride, _1{}, params.v_head_stride, params.v_batch_stride}, // stride_Vt
|
||||
static_cast<ElementSF const*>(params.sfq_ptr),
|
||||
{params.seqlen_q, params.d, params.h, params.b}, // shape_SFQ
|
||||
static_cast<ElementSF const*>(params.sfk_ptr),
|
||||
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_SFK
|
||||
static_cast<ElementSF const*>(params.sfv_ptr),
|
||||
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_SFVt
|
||||
static_cast<float const*>(params.delta_s_ptr),
|
||||
{params.seqlen_s, params.seqlen_k, params.h_k, params.b},
|
||||
{params.ds_row_stride, _1{}, params.ds_head_stride, params.ds_batch_stride},
|
||||
params.scale_softmax_log2
|
||||
});
|
||||
typename CollectiveEpilogue::Params epilogue_params =
|
||||
CollectiveEpilogue::to_underlying_arguments({
|
||||
static_cast<ElementOut*>(params.o_ptr),
|
||||
{params.seqlen_q, params.d, params.h, params.b}, // shape_O
|
||||
{params.o_row_stride, _1{}, params.o_head_stride, params.o_batch_stride}, // stride_O
|
||||
static_cast<float*>(params.softmax_lse_ptr),
|
||||
{_1{}, params.seqlen_q, params.h * params.seqlen_q}, // stride_LSE
|
||||
});
|
||||
|
||||
int num_blocks_m = cutlass::ceil_div(params.seqlen_q, Kernel_traits::kBlockM);
|
||||
num_blocks_m = cutlass::ceil_div(num_blocks_m, size<0>(ClusterShape{})) * size<0>(ClusterShape{});
|
||||
typename Scheduler::Arguments scheduler_args = {num_blocks_m, params.h, params.b};
|
||||
typename Scheduler::Params scheduler_params = Scheduler::to_underlying_arguments(scheduler_args);
|
||||
// Get the ptr to kernel function.
|
||||
void *kernel;
|
||||
kernel = (void *)flash::compute_attn_ws<Kernel_traits, Is_causal, Scheduler>;
|
||||
int smem_size = sizeof(typename Kernel_traits::SharedStorage);
|
||||
if (smem_size >= 48 * 1024) {
|
||||
C10_CUDA_CHECK(cudaFuncSetAttribute(kernel, cudaFuncAttributeMaxDynamicSharedMemorySize, smem_size));
|
||||
}
|
||||
static constexpr int ctaSize = Kernel_traits::kNWarps * 32;
|
||||
params.m_block_divmod = cutlass::FastDivmod(num_blocks_m);
|
||||
params.total_blocks = num_blocks_m * params.h * params.b;
|
||||
dim3 grid_dims = Scheduler::get_grid_dim(scheduler_args, 170);
|
||||
dim3 block_dims(ctaSize);
|
||||
dim3 cluster_dims(size<0>(ClusterShape{}), size<1>(ClusterShape{}), size<2>(ClusterShape{}));
|
||||
cutlass::ClusterLaunchParams launch_params{grid_dims, block_dims, cluster_dims, smem_size, stream};
|
||||
cutlass::launch_kernel_on_cluster(launch_params, kernel, params, mainloop_params, epilogue_params, scheduler_params);
|
||||
|
||||
C10_CUDA_KERNEL_LAUNCH_CHECK();
|
||||
}
|
||||
|
||||
|
||||
template<typename T, int Headdim, typename O = cutlass::bfloat16_t>
|
||||
void run_mha_fwd_(Flash_fwd_params ¶ms, cudaStream_t stream) {
|
||||
BOOL_SWITCH(params.is_causal, Is_causal, [&] {
|
||||
BOOL_SWITCH(params.per_block_mean, per_block, [&] {
|
||||
if constexpr (Headdim == 64 || Headdim == 128) {
|
||||
run_flash_fwd<
|
||||
Flash_fwd_kernel_traits<Headdim, flash::BLOCK_M, flash::BLOCK_N, 3, 1, per_block, T, O>,
|
||||
Is_causal
|
||||
>(params, stream);
|
||||
} else {
|
||||
static_assert(Headdim == 64 || Headdim == 128, "Unsupported Headdim");
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
@@ -1,908 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cutlass/cutlass.h>
|
||||
#include <cutlass/array.h>
|
||||
#include <cutlass/numeric_types.h>
|
||||
#include <cutlass/numeric_conversion.h>
|
||||
#include "cutlass/pipeline/pipeline.hpp"
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
#include "cutlass/gemm/collective/collective_builder.hpp"
|
||||
|
||||
#include "utils.h"
|
||||
#include "named_barrier.h"
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <typename Ktraits, bool Is_causal>
|
||||
struct CollectiveMainloopFwd {
|
||||
|
||||
using Element = typename Ktraits::Element;
|
||||
using ElementSF = typename Ktraits::ElementSF;
|
||||
// using TMAElement = Element;
|
||||
// using TMAElementSF = typename Ktraits::ElementSF;
|
||||
using TileShape_MNK = typename Ktraits::TileShape_MNK;
|
||||
using ClusterShape = typename Ktraits::ClusterShape_MNK;
|
||||
|
||||
static constexpr int kStages = Ktraits::kStages;
|
||||
static constexpr int kHeadDim = Ktraits::kHeadDim;
|
||||
static constexpr int BlockMean = Ktraits::BlockMean;
|
||||
using GmemTiledCopy = typename Ktraits::GmemTiledCopy;
|
||||
using SmemLayoutQ = typename Ktraits::SmemLayoutQ;
|
||||
using SmemLayoutK = typename Ktraits::SmemLayoutK;
|
||||
using SmemLayoutV = typename Ktraits::SmemLayoutV;
|
||||
using SmemLayoutVt = typename Ktraits::SmemLayoutVt;
|
||||
using SmemLayoutDS = typename Ktraits::SmemLayoutDS;
|
||||
using SmemLayoutAtomDS = typename Ktraits::SmemLayoutAtomDS;
|
||||
using LayoutDS = decltype(
|
||||
blocked_product(
|
||||
SmemLayoutAtomDS{},
|
||||
make_layout(
|
||||
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
|
||||
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))
|
||||
)
|
||||
);
|
||||
using ShapeQKV = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d, head, batch)
|
||||
using StrideQKV = cute::Stride<int64_t, _1, int64_t, int64_t>;
|
||||
using ShapeSF = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d // 16, head, batch)
|
||||
using LayoutSF = typename Ktraits::LayoutSF;
|
||||
using LayoutP = typename Ktraits::LayoutP;
|
||||
using LayoutSFP = typename Ktraits::LayoutSFP;
|
||||
using SfAtom = typename Ktraits::SfAtom;
|
||||
using TMA_Q = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
|
||||
SmemLayoutQ{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{}));
|
||||
|
||||
using TMA_KV = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
|
||||
take<0, 2>(SmemLayoutK{}),
|
||||
select<1, 2>(TileShape_MNK{}),
|
||||
_1{}));
|
||||
|
||||
using TMA_Vt = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
|
||||
take<0, 2>(SmemLayoutVt{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using TMA_DS = decltype(make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
make_tensor(make_gmem_ptr(static_cast<float const*>(nullptr)), LayoutDS{}),
|
||||
take<0, 2>(SmemLayoutDS{}),
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using BlkScaledConfig = typename Ktraits::BlkScaledConfig;
|
||||
using GmemTiledCopySF = typename Ktraits::GmemTiledCopySF;
|
||||
using SmemLayoutSFQ = typename Ktraits::SmemLayoutSFQ;
|
||||
using SmemLayoutSFK = typename Ktraits::SmemLayoutSFK;
|
||||
using SmemLayoutSFV = typename Ktraits::SmemLayoutSFV;
|
||||
using SmemLayoutSFVt = typename Ktraits::SmemLayoutSFVt;
|
||||
|
||||
using TMA_SFQ = decltype(make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
|
||||
SmemLayoutSFQ{},
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{})); // No programmatic multicast
|
||||
|
||||
|
||||
using TMA_SFKV = decltype(make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
|
||||
SmemLayoutSFK{}(_,_,cute::Int<0>{}),
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using TMA_SFVt = decltype(make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
|
||||
SmemLayoutSFVt{}(_,_,cute::Int<0>{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}));
|
||||
|
||||
using SmemCopyAtomQ = typename Ktraits::SmemCopyAtomQ;
|
||||
using SmemCopyAtomKV = typename Ktraits::SmemCopyAtomKV;
|
||||
using SmemCopyAtomSF = typename Ktraits::SmemCopyAtomSF;
|
||||
using TiledMmaQK = typename Ktraits::TiledMmaQK;
|
||||
using TiledMmaPV = typename Ktraits::TiledMmaPV;
|
||||
static constexpr int NumMmaThreads = size(TiledMmaQK{});
|
||||
using MainloopPipeline = typename Ktraits::MainloopPipeline;
|
||||
using PipelineParams = typename MainloopPipeline::Params;
|
||||
using PipelineState = typename MainloopPipeline::PipelineState;
|
||||
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
|
||||
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
|
||||
using PipelineStateQ = typename Ktraits::PipelineStateQ;
|
||||
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
|
||||
|
||||
// Set the bytes transferred in this TMA transaction (may involve multiple issues)
|
||||
static constexpr uint32_t TmaTransactionBytesQ = static_cast<uint32_t>(
|
||||
cutlass::bits_to_bytes(cosize((SmemLayoutSFQ{})) * cute::sizeof_bits_v<ElementSF>) +
|
||||
cutlass::bits_to_bytes(size((SmemLayoutQ{})) * sizeof_bits<Element>::value));
|
||||
|
||||
static constexpr uint32_t TmaTransactionBytesK = static_cast<uint32_t>(
|
||||
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFK{})) * cute::sizeof_bits_v<ElementSF>) +
|
||||
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutDS{})) * cute::sizeof_bits_v<float>) +
|
||||
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutK{})) * sizeof_bits<Element>::value));
|
||||
|
||||
static constexpr uint32_t TmaTransactionBytesV = static_cast<uint32_t>(
|
||||
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFVt{})) * cute::sizeof_bits_v<ElementSF>) +
|
||||
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutVt{})) * sizeof_bits<Element>::value));
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
Element const* ptr_Q;
|
||||
ShapeQKV const shape_Q;
|
||||
StrideQKV const stride_Q;
|
||||
Element const* ptr_K;
|
||||
ShapeQKV const shape_K;
|
||||
StrideQKV const stride_K;
|
||||
ShapeQKV const unpadded_shape_K;
|
||||
Element const* ptr_Vt;
|
||||
ShapeQKV const shape_Vt;
|
||||
StrideQKV const stride_Vt;
|
||||
ElementSF const* ptr_SFQ{nullptr};
|
||||
ShapeSF const shape_SFQ{};
|
||||
ElementSF const* ptr_SFK{nullptr};
|
||||
ShapeSF const shape_SFK{};
|
||||
ElementSF const* ptr_SFVt{nullptr};
|
||||
ShapeSF const shape_SFVt{};
|
||||
float const* ptr_ds;
|
||||
ShapeQKV const shape_ds;
|
||||
StrideQKV const stride_ds;
|
||||
float const softmax_scale_log2;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
ShapeQKV const shape_Q;
|
||||
LayoutSF const layout_SFQ;
|
||||
ShapeQKV const shape_K;
|
||||
ShapeQKV const unpadded_shape_K;
|
||||
LayoutSF const layout_SFK;
|
||||
ShapeQKV const shape_Vt;
|
||||
LayoutSF const layout_SFVt;
|
||||
LayoutDS const layout_DS;
|
||||
TMA_Q tma_load_Q;
|
||||
TMA_SFQ tma_load_SFQ;
|
||||
TMA_KV tma_load_K;
|
||||
TMA_SFKV tma_load_SFK;
|
||||
TMA_Vt tma_load_Vt;
|
||||
TMA_SFVt tma_load_SFVt;
|
||||
TMA_DS tma_load_DS;
|
||||
float const softmax_scale_log2;
|
||||
};
|
||||
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
Tensor mQ = make_tensor(make_gmem_ptr(args.ptr_Q), args.shape_Q, args.stride_Q);
|
||||
TMA_Q tma_load_Q = make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
mQ,
|
||||
SmemLayoutQ{},
|
||||
select<0, 2>(TileShape_MNK{}),
|
||||
_1{}); // no mcast for Q
|
||||
Tensor mK = make_tensor(make_gmem_ptr(args.ptr_K), args.shape_K, args.stride_K);
|
||||
TMA_KV tma_load_K = make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
mK,
|
||||
SmemLayoutK{}(_, _, _0{}),
|
||||
select<1, 2>(TileShape_MNK{}),
|
||||
_1{}); // mcast along M mode for this N load, if any
|
||||
Tensor mVt = make_tensor(make_gmem_ptr(args.ptr_Vt), args.shape_Vt, args.stride_Vt);
|
||||
TMA_Vt tma_load_Vt = make_tma_copy(
|
||||
GmemTiledCopy{},
|
||||
mVt,
|
||||
SmemLayoutVt{}(_, _, _0{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{}); // mcast along M mode for this N load, if any
|
||||
auto [Seqlen_Q, Seqlen_K, HeadNum, Batch] = args.shape_ds;
|
||||
LayoutDS layout_ds = tile_to_shape(SmemLayoutAtomDS{}, make_shape(Seqlen_Q, Seqlen_K, HeadNum, Batch), Step<_2,_1,_3,_4>{});
|
||||
Tensor mDS = make_tensor(make_gmem_ptr(args.ptr_ds), layout_ds);
|
||||
TMA_DS tma_load_ds = make_tma_copy (
|
||||
GmemTiledCopy{},
|
||||
mDS,
|
||||
SmemLayoutDS{}(_, _, _0{}),
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{});
|
||||
LayoutSF layout_sfq = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFQ);
|
||||
Tensor mSFQ = make_tensor(make_gmem_ptr(args.ptr_SFQ), layout_sfq);
|
||||
TMA_SFQ tma_load_sfq = make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
mSFQ,
|
||||
SmemLayoutSFQ{},
|
||||
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{});
|
||||
LayoutSF layout_sfk = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFK);
|
||||
Tensor mSFK = make_tensor(make_gmem_ptr(args.ptr_SFK), layout_sfk);
|
||||
TMA_SFKV tma_load_sfk = make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
mSFK,
|
||||
SmemLayoutSFK{}(_, _, _0{}),
|
||||
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
|
||||
_1{});
|
||||
LayoutSF layout_sfvt = BlkScaledConfig::tile_atom_to_shape_SFVt(args.shape_SFVt);
|
||||
Tensor mSFVt = make_tensor(make_gmem_ptr(args.ptr_SFVt), layout_sfvt);
|
||||
TMA_SFVt tma_load_sfvt = make_tma_copy<uint16_t>(
|
||||
GmemTiledCopySF{},
|
||||
mSFVt,
|
||||
SmemLayoutSFVt{}(_, _, _0{}),
|
||||
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
|
||||
_1{});
|
||||
return {args.shape_Q, layout_sfq,
|
||||
args.shape_K, args.unpadded_shape_K, layout_sfk,
|
||||
args.shape_Vt, layout_sfvt,
|
||||
layout_ds,
|
||||
tma_load_Q, tma_load_sfq,
|
||||
tma_load_K, tma_load_sfk,
|
||||
tma_load_Vt, tma_load_sfvt,
|
||||
tma_load_ds,
|
||||
args.softmax_scale_log2};
|
||||
}
|
||||
|
||||
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
|
||||
CUTLASS_DEVICE
|
||||
static void prefetch_tma_descriptors(Params const& mainloop_params) {
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Q.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_K.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Vt.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFQ.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFK.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFVt.get_tma_descriptor());
|
||||
cute::prefetch_tma_descriptor(mainloop_params.tma_load_DS.get_tma_descriptor());
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
int get_n_block_max(Params const& mainloop_params, int m_block) {
|
||||
static constexpr int kBlockM = get<0>(TileShape_MNK{});
|
||||
static constexpr int kBlockN = get<1>(TileShape_MNK{});
|
||||
int const seqlen_q = get<0>(mainloop_params.shape_Q);
|
||||
int const seqlen_k = get<0>(mainloop_params.shape_K);
|
||||
int n_block_max = cute::ceil_div(seqlen_k, kBlockN);
|
||||
if constexpr (Is_causal) {
|
||||
n_block_max = std::min(n_block_max,
|
||||
cute::ceil_div((m_block + 1) * kBlockM + seqlen_k - seqlen_q, kBlockN));
|
||||
}
|
||||
return n_block_max;
|
||||
}
|
||||
|
||||
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<0>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<1>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
|
||||
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
|
||||
{
|
||||
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
|
||||
|
||||
using AtomShape_MNK = typename Atom::Shape_MNK;
|
||||
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
|
||||
|
||||
auto permutation_mnk = TiledPerm{};
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// Reorder the tensor for the TiledAtom
|
||||
auto t_tile = make_tile(get<1>(permutation_mnk),
|
||||
get<2>(permutation_mnk));
|
||||
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
|
||||
|
||||
// Tile the tensor for the Atom
|
||||
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
|
||||
make_layout(size<2>(AtomShape_MNK{})));
|
||||
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
|
||||
|
||||
// Transform the Atom mode from (M,K) to (Thr,Val)
|
||||
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
|
||||
|
||||
// Tile the tensor for the Thread
|
||||
auto thr_tile = make_tile(_,
|
||||
make_tile(make_layout(size<2>(thr_layout_vmnk)),
|
||||
make_layout(size<3>(thr_layout_vmnk))));
|
||||
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
|
||||
return thr_tensor;
|
||||
}
|
||||
|
||||
template <class SFATensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma)
|
||||
{
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
auto partition_SFA = thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
return make_fragment_like<ValTypeSF>(partition_SFA);
|
||||
}
|
||||
|
||||
template <class SFBTensor, class ThrMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma)
|
||||
{
|
||||
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
|
||||
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
|
||||
auto thr_vmnk = thread_mma.thr_vmnk_;
|
||||
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
|
||||
auto partition_SFB = thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
|
||||
return make_fragment_like<ValTypeSF>(partition_SFB);
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFA_TV(TiledMma& mma)
|
||||
{
|
||||
// (M,K) -> (M,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto atile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<1>{} , Int<0>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
template<class TiledMma>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
get_layoutSFB_TV(TiledMma& mma)
|
||||
{
|
||||
// (N,K) -> (N,K)
|
||||
auto tile_shape_mnk = tile_shape(mma);
|
||||
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
|
||||
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
|
||||
|
||||
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
|
||||
auto btile = make_tile(_,
|
||||
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
|
||||
make_stride( Int<0>{} , Int<1>{} )),
|
||||
_));
|
||||
|
||||
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
|
||||
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
|
||||
// (thr_idx,val) -> (M,K)
|
||||
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
|
||||
}
|
||||
|
||||
template <typename SchedulerParams, typename SharedStorage, typename WorkTileInfo>
|
||||
CUTLASS_DEVICE void
|
||||
load(Params const& mainloop_params,
|
||||
SchedulerParams const& scheduler_params,
|
||||
MainloopPipelineQ pipeline_q,
|
||||
MainloopPipeline pipeline_k,
|
||||
MainloopPipeline pipeline_v,
|
||||
PipelineStateQ& smem_pipe_write_q,
|
||||
PipelineState& smem_pipe_write_k,
|
||||
PipelineState& smem_pipe_write_v,
|
||||
SharedStorage &shared_storage,
|
||||
WorkTileInfo work_tile_info,
|
||||
int& work_idx,
|
||||
int& tile_count_semaphore
|
||||
) {
|
||||
|
||||
static constexpr int kBlockM = get<0>(TileShape_MNK{});
|
||||
static constexpr int kBlockN = get<1>(TileShape_MNK{});
|
||||
|
||||
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
|
||||
|
||||
int n_block_max = get_n_block_max(mainloop_params, m_block);
|
||||
|
||||
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
|
||||
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
|
||||
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
|
||||
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
|
||||
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
|
||||
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
|
||||
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
|
||||
|
||||
Tensor mQ = mainloop_params.tma_load_Q.get_tma_tensor(mainloop_params.shape_Q);
|
||||
Tensor mK = mainloop_params.tma_load_K.get_tma_tensor(mainloop_params.shape_K);
|
||||
Tensor mVt = mainloop_params.tma_load_Vt.get_tma_tensor(mainloop_params.shape_Vt);
|
||||
Tensor mDS = mainloop_params.tma_load_DS.get_tma_tensor(shape(mainloop_params.layout_DS));
|
||||
Tensor mSFQ = mainloop_params.tma_load_SFQ.get_tma_tensor(shape(mainloop_params.layout_SFQ));
|
||||
Tensor mSFK = mainloop_params.tma_load_SFK.get_tma_tensor(shape(mainloop_params.layout_SFK));
|
||||
Tensor mSFVt = mainloop_params.tma_load_SFVt.get_tma_tensor(shape(mainloop_params.layout_SFVt));
|
||||
uint32_t block_rank_in_cluster = cute::block_rank_in_cluster();
|
||||
constexpr uint32_t cluster_shape_x = get<0>(ClusterShape());
|
||||
uint2 cluster_local_block_id = {block_rank_in_cluster % cluster_shape_x, block_rank_in_cluster / cluster_shape_x};
|
||||
Tensor gQ = local_tile(mQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
|
||||
Tensor gK = local_tile(mK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{})); // (N, K, _)
|
||||
Tensor gVt = local_tile(mVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _)); // (N, K, _)
|
||||
Tensor gDS = [&] {
|
||||
if constexpr (BlockMean) {
|
||||
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(m_block, _));
|
||||
} else {
|
||||
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(_0{}, _));
|
||||
}
|
||||
}();
|
||||
Tensor gSFQ = local_tile(mSFQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{}));
|
||||
Tensor gSFK = local_tile(mSFK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{}));
|
||||
Tensor gSFVt = local_tile(mSFVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _));
|
||||
auto block_tma_q = mainloop_params.tma_load_Q.get_slice(_0{});
|
||||
Tensor tQgQ = block_tma_q.partition_S(gQ);
|
||||
Tensor tQsQ = block_tma_q.partition_D(sQ);
|
||||
auto block_tma_sfq = mainloop_params.tma_load_SFQ.get_slice(_0{});
|
||||
Tensor tQgSFQ = block_tma_sfq.partition_S(gSFQ);
|
||||
Tensor tQsSFQ = block_tma_sfq.partition_D(sSFQ);
|
||||
auto block_tma_k = mainloop_params.tma_load_K.get_slice(cluster_local_block_id.x);
|
||||
Tensor tKgK = group_modes<0, 3>(block_tma_k.partition_S(gK));
|
||||
Tensor tKsK = group_modes<0, 3>(block_tma_k.partition_D(sK));
|
||||
auto block_tma_sfk = mainloop_params.tma_load_SFK.get_slice(cluster_local_block_id.x);
|
||||
Tensor tKgSFK = group_modes<0, 3>(block_tma_sfk.partition_S(gSFK));
|
||||
Tensor tKsSFK = group_modes<0, 3>(block_tma_sfk.partition_D(sSFK));
|
||||
auto block_tma_vt = mainloop_params.tma_load_Vt.get_slice(cluster_local_block_id.x);
|
||||
Tensor tVgVt = group_modes<0, 3>(block_tma_vt.partition_S(gVt));
|
||||
Tensor tVsVt = group_modes<0, 3>(block_tma_vt.partition_D(sVt));
|
||||
auto block_tma_sfvt = mainloop_params.tma_load_SFVt.get_slice(cluster_local_block_id.x);
|
||||
Tensor tVgSFVt = group_modes<0, 3>(block_tma_sfvt.partition_S(gSFVt));
|
||||
Tensor tVsSFVt = group_modes<0, 3>(block_tma_sfvt.partition_D(sSFVt));
|
||||
auto block_tma_ds = mainloop_params.tma_load_DS.get_slice(cluster_local_block_id.x);
|
||||
Tensor tDSgDS = group_modes<0, 3>(block_tma_ds.partition_S(gDS));
|
||||
Tensor tDSsDS = group_modes<0, 3>(block_tma_ds.partition_D(sDS));
|
||||
uint16_t mcast_mask_kv = 0;
|
||||
|
||||
int n_block = n_block_max - 1;
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
if (lane_predicate) {
|
||||
pipeline_q.producer_acquire(smem_pipe_write_q);
|
||||
copy(mainloop_params.tma_load_Q.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgQ, tQsQ);
|
||||
copy(mainloop_params.tma_load_SFQ.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgSFQ, tQsSFQ);
|
||||
++smem_pipe_write_q;
|
||||
pipeline_k.producer_acquire(smem_pipe_write_k);
|
||||
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
|
||||
++smem_pipe_write_k;
|
||||
pipeline_v.producer_acquire(smem_pipe_write_v);
|
||||
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
|
||||
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
|
||||
++smem_pipe_write_v;
|
||||
}
|
||||
|
||||
n_block--;
|
||||
if (lane_predicate) {
|
||||
// CUTLASS_PRAGMA_NO_UNROLL
|
||||
#pragma unroll 2
|
||||
for (; n_block >= 0; --n_block) {
|
||||
pipeline_k.producer_acquire(smem_pipe_write_k);
|
||||
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
|
||||
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
|
||||
++smem_pipe_write_k;
|
||||
pipeline_v.producer_acquire(smem_pipe_write_v);
|
||||
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
|
||||
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
|
||||
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
|
||||
++smem_pipe_write_v;
|
||||
}
|
||||
}
|
||||
++work_idx;
|
||||
}
|
||||
|
||||
/// Perform a Producer Epilogue to prevent early exit of blocks in a Cluster
|
||||
CUTLASS_DEVICE void
|
||||
load_tail(MainloopPipelineQ pipeline_q,
|
||||
MainloopPipeline pipeline_k,
|
||||
MainloopPipeline pipeline_v,
|
||||
PipelineStateQ& smem_pipe_write_q,
|
||||
PipelineState& smem_pipe_write_k,
|
||||
PipelineState& smem_pipe_write_v) {
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
// Issue the epilogue waits
|
||||
if (lane_predicate) {
|
||||
pipeline_q.producer_tail(smem_pipe_write_q);
|
||||
pipeline_k.producer_tail(smem_pipe_write_k);
|
||||
pipeline_v.producer_tail(smem_pipe_write_v);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename SharedStorage, typename FrgTensorO, typename SoftmaxFused>
|
||||
CUTLASS_DEVICE void
|
||||
mma(Params const& mainloop_params,
|
||||
MainloopPipelineQ pipeline_q,
|
||||
MainloopPipeline pipeline_k,
|
||||
MainloopPipeline pipeline_v,
|
||||
PipelineStateQ& smem_pipe_read_q,
|
||||
PipelineState& smem_pipe_read_k,
|
||||
PipelineState& smem_pipe_read_v,
|
||||
FrgTensorO& tOrO_store,
|
||||
SoftmaxFused& softmax_fused,
|
||||
int n_block_count,
|
||||
int thread_idx,
|
||||
int work_idx,
|
||||
int m_block,
|
||||
SharedStorage& shared_storage
|
||||
) {
|
||||
|
||||
static_assert(is_rmem<FrgTensorO>::value, "O tensor must be rmem resident.");
|
||||
|
||||
static constexpr int kBlockM = get<0>(TileShape_MNK{});
|
||||
static constexpr int kBlockN = get<1>(TileShape_MNK{});
|
||||
static constexpr int kBlockK = get<2>(TileShape_MNK{});
|
||||
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
|
||||
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
|
||||
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
|
||||
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
|
||||
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
|
||||
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
|
||||
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
|
||||
|
||||
Tensor cQ = make_identity_tensor(make_shape(size<0>(sQ), size<1>(sQ)));
|
||||
Tensor cKV = make_identity_tensor(make_shape(size<0>(sK), size<1>(sK)));
|
||||
TiledMmaQK tiled_mma_qk;
|
||||
TiledMmaPV tiled_mma_pv;
|
||||
auto thread_mma_qk = tiled_mma_qk.get_thread_slice(thread_idx);
|
||||
auto thread_mma_pv = tiled_mma_pv.get_thread_slice(thread_idx);
|
||||
|
||||
Tensor tSrQ = thread_mma_qk.partition_fragment_A(sQ);
|
||||
Tensor tSrK = thread_mma_qk.partition_fragment_B(sK(_,_,Int<0>{}));
|
||||
Tensor tOrVt = thread_mma_pv.partition_fragment_B(sVt(_,_,Int<0>{}));
|
||||
Tensor tOrP = make_tensor_like<Element>(LayoutP{});
|
||||
Tensor tSrSFQ = partition_fragment_SFA(sSFQ, thread_mma_qk);
|
||||
Tensor tSrSFK = partition_fragment_SFB(sSFK(_,_,Int<0>{}), thread_mma_qk);
|
||||
Tensor tOrSFVt = partition_fragment_SFB(sSFVt(_,_,Int<0>{}), thread_mma_pv);
|
||||
Tensor tOrSFP = make_tensor<ElementSF>(LayoutSFP{});
|
||||
Tensor tOrSFP_flt = filter_zeros(tOrSFP);
|
||||
Tensor tSrDS = make_tensor<float>(make_shape(_8{}, _4{}), make_stride(_1{}, _8{}));
|
||||
// copy qk and sf from smem to rmem
|
||||
auto smem_tiled_copy_Q = make_tiled_copy_A(SmemCopyAtomQ{}, tiled_mma_qk);
|
||||
auto smem_thr_copy_Q = smem_tiled_copy_Q.get_thread_slice(thread_idx);
|
||||
Tensor tSsQ = smem_thr_copy_Q.partition_S(as_position_independent_swizzle_tensor(sQ));
|
||||
Tensor tSrQ_copy_view = smem_thr_copy_Q.retile_D(tSrQ);
|
||||
|
||||
auto smem_tiled_copy_K = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_qk);
|
||||
auto smem_thr_copy_K = smem_tiled_copy_K.get_thread_slice(thread_idx);
|
||||
Tensor tSsK = smem_thr_copy_K.partition_S(as_position_independent_swizzle_tensor(sK));
|
||||
Tensor tSrK_copy_view = smem_thr_copy_K.retile_D(tSrK);
|
||||
|
||||
auto smem_tiled_copy_V = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_pv);
|
||||
auto smem_thr_copy_V = smem_tiled_copy_V.get_thread_slice(thread_idx);
|
||||
Tensor tOsVt = smem_thr_copy_V.partition_S(as_position_independent_swizzle_tensor(sVt));
|
||||
Tensor tOrVt_copy_view = smem_thr_copy_V.retile_D(tOrVt);
|
||||
|
||||
auto tile_shape_mnk = tile_shape(tiled_mma_qk);
|
||||
auto smem_tiled_copy_SFQ = make_tiled_copy_impl(SmemCopyAtomSF{},
|
||||
get_layoutSFA_TV(tiled_mma_qk),
|
||||
make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk))
|
||||
);
|
||||
auto smem_thr_copy_SFQ = smem_tiled_copy_SFQ.get_thread_slice(thread_idx);
|
||||
Tensor tSsSFQ = smem_thr_copy_SFQ.partition_S(as_position_independent_swizzle_tensor(sSFQ));
|
||||
Tensor tSrSFQ_copy_view = smem_thr_copy_SFQ.retile_D(tSrSFQ);
|
||||
|
||||
auto smem_tiled_copy_SFK = make_tiled_copy_impl(SmemCopyAtomSF{},
|
||||
get_layoutSFB_TV(tiled_mma_qk),
|
||||
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
|
||||
);
|
||||
auto smem_thr_copy_SFK = smem_tiled_copy_SFK.get_thread_slice(thread_idx);
|
||||
Tensor tSsSFK = smem_thr_copy_SFK.partition_S(as_position_independent_swizzle_tensor(sSFK));
|
||||
Tensor tSrSFK_copy_view = smem_thr_copy_SFK.retile_D(tSrSFK);
|
||||
|
||||
auto smem_tiled_copy_SFV = make_tiled_copy_impl(SmemCopyAtomSF{},
|
||||
get_layoutSFB_TV(tiled_mma_pv),
|
||||
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
|
||||
);
|
||||
auto smem_thr_copy_SFV = smem_tiled_copy_SFV.get_thread_slice(thread_idx);
|
||||
Tensor tOsSFVt = smem_thr_copy_SFV.partition_S(as_position_independent_swizzle_tensor(sSFVt));
|
||||
Tensor tOrSFVt_copy_view = smem_thr_copy_SFV.retile_D(tOrSFVt);
|
||||
|
||||
auto consumer_wait = [](auto& pipeline, auto& smem_pipe_read) {
|
||||
auto barrier_token = pipeline.consumer_try_wait(smem_pipe_read);
|
||||
pipeline.consumer_wait(smem_pipe_read, barrier_token);
|
||||
};
|
||||
|
||||
int const seqlen_q = get<0>(mainloop_params.shape_Q);
|
||||
int const seqlen_k = get<0>(mainloop_params.shape_K);
|
||||
int const unpadded_seqlen_k = get<0>(mainloop_params.unpadded_shape_K);
|
||||
int n_block = n_block_count - 1;
|
||||
|
||||
auto copy_k_block = [&](auto block_id) {
|
||||
auto tSsK_stage = tSsK(_, _, _, smem_pipe_read_k.index());
|
||||
auto tSsSFK_stage = tSsSFK(_, _, _, smem_pipe_read_k.index());
|
||||
copy(smem_tiled_copy_K, tSsK_stage(_, _, block_id), tSrK_copy_view(_, _, block_id));
|
||||
copy(smem_tiled_copy_SFK, tSsSFK_stage(_, _, block_id), tSrSFK_copy_view(_, _, block_id));
|
||||
};
|
||||
|
||||
auto copy_v_block = [&](auto block_id) {
|
||||
auto tOsVt_stage = tOsVt(_, _, _, smem_pipe_read_v.index());
|
||||
auto tOsSFVt_stage = tOsSFVt(_, _, _, smem_pipe_read_v.index());
|
||||
copy(smem_tiled_copy_V, tOsVt_stage(_, _, block_id), tOrVt_copy_view(_, _, block_id));
|
||||
copy(smem_tiled_copy_SFV, tOsSFVt_stage(_, _, block_id), tOrSFVt_copy_view(_, _, block_id));
|
||||
};
|
||||
// auto gemm_qk = [&](auto block_id) {
|
||||
// cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, block_id), tSrSFQ(_, _, block_id)), make_zip_tensor(tSrK(_, _, block_id), tSrSFK(_, _, block_id)), tSrS);
|
||||
// };
|
||||
// auto gemm_pv = [&](auto block_id) {
|
||||
// cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, block_id), tOrSFP(_, _, block_id)), make_zip_tensor(tOrVt(_, _, block_id), tOrSFVt(_, _, block_id)), tOrO);
|
||||
// };
|
||||
auto add_delta_s = [&](auto& acc) {
|
||||
auto tSsDS_stage = recast<float4>(sDS(_, _, smem_pipe_read_k.index()));
|
||||
auto acc_float4 = recast<float4>(acc);
|
||||
int quad_id = (threadIdx.x % 4) * 2;
|
||||
for (int i = 0; i < 4; i++) {
|
||||
auto num = quad_id + i * 8;
|
||||
float4 delta_s_0 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num, _0{}));
|
||||
float4 delta_s_1 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num + 1, _0{}));
|
||||
acc_float4(make_coord(make_coord(_0{}, _0{}), _0{}), _0{}, i) = delta_s_0;
|
||||
acc_float4(make_coord(make_coord(_0{}, _0{}), _1{}), _0{}, i) = delta_s_0;
|
||||
acc_float4(make_coord(make_coord(_0{}, _1{}), _0{}), _0{}, i) = delta_s_1;
|
||||
acc_float4(make_coord(make_coord(_0{}, _1{}), _1{}), _0{}, i) = delta_s_1;
|
||||
}
|
||||
};
|
||||
consumer_wait(pipeline_q, smem_pipe_read_q);
|
||||
copy(smem_tiled_copy_Q, tSsQ, tSrQ_copy_view);
|
||||
copy(smem_tiled_copy_SFQ, tSsSFQ, tSrSFQ_copy_view);
|
||||
pipeline_q.consumer_release(smem_pipe_read_q);
|
||||
++smem_pipe_read_q;
|
||||
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
Tensor AbsMaxP = make_tensor_like<float>(
|
||||
make_layout(shape(group<1, 4>(flatten(tSrS_converion_view.layout()(make_coord(_0{}, _), _, _)))))
|
||||
);
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
copy_k_block(_0{});
|
||||
add_delta_s(tSrS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
|
||||
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
|
||||
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
|
||||
if (k_block < size<2>(tSrQ) - 1) {
|
||||
copy_k_block(k_block + 1);
|
||||
} else {
|
||||
pipeline_k.consumer_release(smem_pipe_read_k);
|
||||
++smem_pipe_read_k;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
auto col_limit_causal = [&](int row, int n_block) {
|
||||
return row + 1 + seqlen_k - n_block * kBlockN - seqlen_q + m_block * kBlockM;
|
||||
};
|
||||
{
|
||||
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tScS = thread_mma_qk.partition_C(cS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size(tSrS); ++i) {
|
||||
if constexpr (!Is_causal) { // Just masking based on col
|
||||
if (int(get<1>(tScS(i))) >= int(unpadded_seqlen_k - n_block * kBlockN)) { tSrS(i) = -INFINITY; }
|
||||
} else {
|
||||
if (int(get<1>(tScS(i))) >= std::min(seqlen_k - n_block * kBlockN,
|
||||
col_limit_causal(int(get<0>(tScS(i))), n_block))) {
|
||||
tSrS(i) = -INFINITY;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
auto quantize = [&](auto mma_k, auto acc_conversion_view) {
|
||||
Tensor AbsMaxP_stagek = AbsMaxP(_, make_coord(_, _, mma_k));
|
||||
Tensor acc_conversion_stagek = acc_conversion_view(_, _, mma_k);
|
||||
Tensor SFP = make_tensor_like<cutlass::float_ue4m3_t>(AbsMaxP_stagek.layout());
|
||||
Tensor SFP_uint32_view = recast<uint32_t>(SFP);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size(AbsMaxP_stagek); i += 4) {
|
||||
uint32_t& tmp = SFP_uint32_view(i / 4);
|
||||
flash::packed_float_to_ue4m3(
|
||||
AbsMaxP_stagek(i),
|
||||
AbsMaxP_stagek(i + 1),
|
||||
AbsMaxP_stagek(i + 2),
|
||||
AbsMaxP_stagek(i + 3),
|
||||
tmp
|
||||
);
|
||||
}
|
||||
int const quad_id = threadIdx.x & 3;
|
||||
uint32_t MASK = (0xFF00FF) << ((quad_id & 1) * 8);
|
||||
Tensor tOrSFP_uint32_view = recast<uint32_t>(tOrSFP(_, _, mma_k));
|
||||
Tensor tOrP_uint32_view = recast<uint32_t>(tOrP(_, _, mma_k));
|
||||
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mma_m = 0; mma_m < size<1>(tOrP); ++mma_m) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
flash::packed_float_to_e2m1(
|
||||
acc_conversion_stagek(make_coord(_0{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_1{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_2{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_3{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_4{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_5{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_6{}, i), mma_m),
|
||||
acc_conversion_stagek(make_coord(_7{}, i), mma_m),
|
||||
tOrP_uint32_view(i, mma_m)
|
||||
);
|
||||
}
|
||||
uint32_t local_sfp = SFP_uint32_view(_0{}, _0{}, mma_m);
|
||||
uint32_t peer_sfp = __shfl_xor_sync(int32_t(-1), local_sfp, 2);
|
||||
if ((quad_id & 1) == 0) {
|
||||
uint32_t sfp = (local_sfp & MASK) | ((peer_sfp & MASK) << 8);
|
||||
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
|
||||
} else {
|
||||
uint32_t sfp = (peer_sfp & MASK) | ((local_sfp & MASK) >> 8);
|
||||
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
softmax_fused.template online_softmax_with_quant</*Is_first=*/true>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
|
||||
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
copy_v_block(_0{});
|
||||
quantize(_0{}, tSrS_converion_view);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO_store);
|
||||
if (v_block < size<2>(tOrP) - 1) {
|
||||
copy_v_block(v_block + 1);
|
||||
quantize(v_block + 1, tSrS_converion_view);
|
||||
} else {
|
||||
pipeline_v.consumer_release(smem_pipe_read_v);
|
||||
++smem_pipe_read_v;
|
||||
}
|
||||
}
|
||||
|
||||
n_block--;
|
||||
constexpr int n_masking_steps = !Is_causal ? 1 : cute::ceil_div(kBlockM, kBlockN) + 1;
|
||||
// // Only go through these if Is_causal, since n_masking_steps = 1 when !Is_causal
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int masking_step = 0; masking_step < n_masking_steps - 1 && n_block >= 0; ++masking_step, --n_block) {
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
copy_k_block(_0{});
|
||||
add_delta_s(tSrS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
|
||||
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
|
||||
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
|
||||
if (k_block < size<2>(tSrQ) - 1) {
|
||||
copy_k_block(k_block + 1);
|
||||
}
|
||||
}
|
||||
pipeline_k.consumer_release(smem_pipe_read_k); // release K
|
||||
++smem_pipe_read_k;
|
||||
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tScS = thread_mma_qk.partition_C(cS);
|
||||
#pragma unroll
|
||||
for (int i = 0; i < size(tSrS); ++i) {
|
||||
if (int(get<1>(tScS(i))) >= col_limit_causal(int(get<0>(tScS(i))), n_block - 1)) {
|
||||
tSrS(i) = -INFINITY;
|
||||
}
|
||||
}
|
||||
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
|
||||
Tensor tOrO = make_fragment_like(tOrO_store);
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
copy_v_block(_0{});
|
||||
quantize(_0{}, tSrS_converion_view);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
|
||||
if (v_block < size<2>(tOrP) - 1) {
|
||||
copy_v_block(v_block + 1);
|
||||
quantize(v_block + 1, tSrS_converion_view);
|
||||
}
|
||||
}
|
||||
pipeline_v.consumer_release(smem_pipe_read_v);
|
||||
++smem_pipe_read_v;
|
||||
if (masking_step > 0) { softmax_fused.rescale_o(tOrO_store, tOrO); }
|
||||
}
|
||||
|
||||
#pragma unroll 1
|
||||
for (; n_block >= 0; --n_block) {
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
copy_k_block(_0{});
|
||||
add_delta_s(tSrS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
|
||||
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
|
||||
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
|
||||
if (k_block < size<2>(tSrQ) - 1) {
|
||||
copy_k_block(k_block + 1);
|
||||
} else {
|
||||
pipeline_k.consumer_release(smem_pipe_read_k);
|
||||
++smem_pipe_read_k;
|
||||
}
|
||||
}
|
||||
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
|
||||
Tensor tOrO = make_fragment_like(tOrO_store);
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
copy_v_block(_0{});
|
||||
quantize(_0{}, tSrS_converion_view);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
|
||||
if (v_block < size<2>(tOrP) - 1) {
|
||||
copy_v_block(v_block + 1);
|
||||
quantize(v_block + 1, tSrS_converion_view);
|
||||
} else {
|
||||
pipeline_v.consumer_release(smem_pipe_read_v);
|
||||
++smem_pipe_read_v;
|
||||
}
|
||||
}
|
||||
softmax_fused.rescale_o(tOrO_store, tOrO);
|
||||
}
|
||||
softmax_fused.finalize(tOrO_store);
|
||||
return;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
} // namespace flash
|
||||
|
||||
@@ -1,119 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cutlass/pipeline/sm90_pipeline.hpp"
|
||||
|
||||
namespace flash {
|
||||
|
||||
enum class FP4NamedBarriers {
|
||||
QueryEmpty = 1,
|
||||
WarpSpecializedConsumer = 2,
|
||||
WarpSpecializedPingPongConsumer1 = 3,
|
||||
WarpSpecializedPingPongConsumer2 = 4,
|
||||
ProducerEnd = 5,
|
||||
ConsumerEnd = 6,
|
||||
EpilogueBarrier = 7
|
||||
};
|
||||
|
||||
template<int SequenceDepth, int SequenceLength>
|
||||
struct OrderedSequenceBarrierVarGroupSizeSharedStorage {
|
||||
using Barrier = cutlass::arch::ClusterBarrier;
|
||||
Barrier barrier_[SequenceDepth][SequenceLength];
|
||||
};
|
||||
|
||||
template<int SequenceDepth_, int SequenceLength_>
|
||||
class OrderedSequenceBarrierVarGroupSize {
|
||||
public:
|
||||
static constexpr int SequenceDepth = SequenceDepth_;
|
||||
static constexpr int SequenceLength = SequenceLength_;
|
||||
using Barrier = cutlass::arch::ClusterBarrier;
|
||||
using SharedStorage = flash::OrderedSequenceBarrierVarGroupSizeSharedStorage<SequenceDepth, SequenceLength>;
|
||||
|
||||
|
||||
struct Params {
|
||||
uint32_t group_id;
|
||||
uint32_t* group_size_list;
|
||||
};
|
||||
|
||||
private :
|
||||
// In future this Params object can be replaced easily with a CG object
|
||||
Params params_;
|
||||
Barrier *barrier_ptr_;
|
||||
cutlass::PipelineState<SequenceDepth> stage_;
|
||||
|
||||
static constexpr int Depth = SequenceDepth;
|
||||
static constexpr int Length = SequenceLength;
|
||||
|
||||
public:
|
||||
OrderedSequenceBarrierVarGroupSize() = delete;
|
||||
OrderedSequenceBarrierVarGroupSize(const OrderedSequenceBarrierVarGroupSize&) = delete;
|
||||
OrderedSequenceBarrierVarGroupSize(OrderedSequenceBarrierVarGroupSize&&) = delete;
|
||||
OrderedSequenceBarrierVarGroupSize& operator=(const OrderedSequenceBarrierVarGroupSize&) = delete;
|
||||
OrderedSequenceBarrierVarGroupSize& operator=(OrderedSequenceBarrierVarGroupSize&&) = delete;
|
||||
~OrderedSequenceBarrierVarGroupSize() = default;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
OrderedSequenceBarrierVarGroupSize(SharedStorage& storage, Params const& params) :
|
||||
params_(params),
|
||||
barrier_ptr_(&storage.barrier_[0][0]),
|
||||
// Group 0 - starts with an opposite phase
|
||||
stage_({0, params.group_id == 0, 0}) {
|
||||
int warp_idx = cutlass::canonical_warp_idx_sync();
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
|
||||
// Barrier FULL, EMPTY init
|
||||
// Init is done only by the one elected thread of the block
|
||||
if (warp_idx == 0 && lane_predicate) {
|
||||
for (int d = 0; d < Depth; ++d) {
|
||||
for (int l = 0; l < Length; ++l) {
|
||||
barrier_ptr_[d * Length + l].init(*(params.group_size_list + l));
|
||||
}
|
||||
}
|
||||
}
|
||||
cutlass::arch::fence_barrier_init();
|
||||
}
|
||||
|
||||
// Wait on a stage to be unlocked
|
||||
CUTLASS_DEVICE
|
||||
void wait() {
|
||||
get_barrier_for_current_stage(params_.group_id).wait(stage_.phase());
|
||||
}
|
||||
|
||||
// Signal completion of Stage and move to the next stage
|
||||
// (group_id) signals to (group_id+1)
|
||||
CUTLASS_DEVICE
|
||||
void arrive() {
|
||||
int signalling_id = (params_.group_id + 1) % Length;
|
||||
get_barrier_for_current_stage(signalling_id).arrive();
|
||||
++stage_;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
void advance() {
|
||||
++stage_;
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
CUTLASS_DEVICE
|
||||
Barrier& get_barrier_for_current_stage(int group_id) {
|
||||
return barrier_ptr_[stage_.index() * Length + group_id];
|
||||
}
|
||||
};
|
||||
|
||||
} // flash
|
||||
@@ -1,180 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cuda.h>
|
||||
#include <vector>
|
||||
|
||||
#ifdef OLD_GENERATOR_PATH
|
||||
#include <ATen/CUDAGeneratorImpl.h>
|
||||
#else
|
||||
#include <ATen/cuda/CUDAGeneratorImpl.h>
|
||||
#endif
|
||||
|
||||
#include <ATen/cuda/CUDAGraphsUtils.cuh> // For at::cuda::philox::unpack
|
||||
|
||||
#include "cutlass/fast_math.h" // For cutlass::FastDivmod
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
struct Qkv_params {
|
||||
using index_t = int64_t;
|
||||
// The QKV matrices.
|
||||
void *__restrict__ q_ptr;
|
||||
void *__restrict__ k_ptr;
|
||||
void *__restrict__ v_ptr;
|
||||
void *__restrict__ delta_s_ptr;
|
||||
// The QKV scale factor matrices.
|
||||
void *__restrict__ sfq_ptr;
|
||||
void *__restrict__ sfk_ptr;
|
||||
void *__restrict__ sfv_ptr;
|
||||
// The stride between rows of the Q, K and V matrices.
|
||||
index_t q_batch_stride;
|
||||
index_t k_batch_stride;
|
||||
index_t v_batch_stride;
|
||||
index_t q_row_stride;
|
||||
index_t k_row_stride;
|
||||
index_t v_row_stride;
|
||||
index_t q_head_stride;
|
||||
index_t k_head_stride;
|
||||
index_t v_head_stride;
|
||||
index_t ds_batch_stride;
|
||||
index_t ds_row_stride;
|
||||
index_t ds_head_stride;
|
||||
// The stride of the Q, K and V scale factor matrices.
|
||||
index_t sfq_batch_stride;
|
||||
index_t sfk_batch_stride;
|
||||
index_t sfv_batch_stride;
|
||||
index_t sfq_row_stride;
|
||||
index_t sfk_row_stride;
|
||||
index_t sfv_row_stride;
|
||||
index_t sfq_head_stride;
|
||||
index_t sfk_head_stride;
|
||||
index_t sfv_head_stride;
|
||||
|
||||
// The number of heads.
|
||||
int h, h_k;
|
||||
// In the case of multi-query and grouped-query attention (MQA/GQA), nheads_k could be
|
||||
// different from nheads (query).
|
||||
int h_h_k_ratio; // precompute h / h_k,
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
struct Flash_fwd_params : public Qkv_params {
|
||||
|
||||
// The O matrix (output).
|
||||
void * __restrict__ o_ptr;
|
||||
void * __restrict__ oaccum_ptr;
|
||||
void * __restrict__ s_ptr;
|
||||
|
||||
// The stride between rows of O.
|
||||
index_t o_batch_stride;
|
||||
index_t o_row_stride;
|
||||
index_t o_head_stride;
|
||||
|
||||
// The pointer to the P matrix.
|
||||
void * __restrict__ p_ptr;
|
||||
|
||||
// The pointer to the softmax sum.
|
||||
void * __restrict__ softmax_lse_ptr;
|
||||
void * __restrict__ softmax_lseaccum_ptr;
|
||||
|
||||
// The dimensions.
|
||||
int b, seqlen_q, seqlen_k, seqlen_knew, d, seqlen_q_rounded, seqlen_k_rounded, d_rounded, rotary_dim, unpadded_seqlen_k;
|
||||
cutlass::FastDivmod head_divmod, m_block_divmod;
|
||||
int total_blocks;
|
||||
int seqlen_s;
|
||||
|
||||
// The scaling factors for the kernel.
|
||||
float scale_softmax;
|
||||
float scale_softmax_log2;
|
||||
uint32_t scale_softmax_log2_half2;
|
||||
|
||||
// array of length b+1 holding starting offset of each sequence.
|
||||
int * __restrict__ cu_seqlens_q;
|
||||
int * __restrict__ cu_seqlens_k;
|
||||
|
||||
// If provided, the actual length of each k sequence.
|
||||
int * __restrict__ seqused_k;
|
||||
|
||||
int *__restrict__ blockmask;
|
||||
|
||||
// The K_new and V_new matrices.
|
||||
void * __restrict__ knew_ptr;
|
||||
void * __restrict__ vnew_ptr;
|
||||
|
||||
// The stride between rows of the Q, K and V matrices.
|
||||
index_t knew_batch_stride;
|
||||
index_t vnew_batch_stride;
|
||||
index_t knew_row_stride;
|
||||
index_t vnew_row_stride;
|
||||
index_t knew_head_stride;
|
||||
index_t vnew_head_stride;
|
||||
|
||||
// The cos and sin matrices for rotary embedding.
|
||||
void * __restrict__ rotary_cos_ptr;
|
||||
void * __restrict__ rotary_sin_ptr;
|
||||
|
||||
// The indices to index into the KV cache.
|
||||
int * __restrict__ cache_batch_idx;
|
||||
|
||||
// Paged KV cache
|
||||
int * __restrict__ block_table;
|
||||
index_t block_table_batch_stride;
|
||||
int page_block_size;
|
||||
|
||||
// The dropout probability (probability of keeping an activation).
|
||||
float p_dropout;
|
||||
// uint32_t p_dropout_in_uint;
|
||||
// uint16_t p_dropout_in_uint16_t;
|
||||
uint8_t p_dropout_in_uint8_t;
|
||||
|
||||
// Scale factor of 1 / (1 - p_dropout).
|
||||
float rp_dropout;
|
||||
float scale_softmax_rp_dropout;
|
||||
|
||||
// Local window size
|
||||
int window_size_left, window_size_right;
|
||||
|
||||
// Random state.
|
||||
at::PhiloxCudaState philox_args;
|
||||
|
||||
// Pointer to the RNG seed (idx 0) and offset (idx 1).
|
||||
uint64_t * rng_state;
|
||||
|
||||
bool is_bf16;
|
||||
bool is_e4m3;
|
||||
bool is_causal;
|
||||
bool per_block_mean;
|
||||
bool single_level_p_quant; // If true, use single-level 1x16 block scale quantization for P (like V), instead of two-level quantization
|
||||
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
|
||||
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
|
||||
bool is_seqlens_k_cumulative;
|
||||
|
||||
bool is_rotary_interleaved;
|
||||
|
||||
int num_splits; // For split-KV version
|
||||
|
||||
void * __restrict__ alibi_slopes_ptr;
|
||||
index_t alibi_slopes_batch_stride;
|
||||
|
||||
int * __restrict__ tile_count_semaphore;
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -1,190 +0,0 @@
|
||||
// Modified from the original SageAttention3 code
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#pragma once
|
||||
|
||||
#include <cmath>
|
||||
#include "cute/tensor.hpp"
|
||||
#include "cutlass/numeric_types.h"
|
||||
#include "utils.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
template <int Rows>
|
||||
struct SoftmaxFused{
|
||||
|
||||
using TensorT = decltype(make_fragment_like<float>(Shape<Int<Rows>>{}));
|
||||
TensorT row_sum, row_max, scores_scale;
|
||||
static constexpr float fp8_scalexfp4_scale = 1.f / (448 * 6);
|
||||
static constexpr float fp8_scalexfp4_scale_log2 = -11.392317422778762f; //log2f(fp8_scalexfp4_scale)
|
||||
static constexpr float fp4_scale_log2 = -2.584962500721156f; // log2f(fp4_scale)
|
||||
static constexpr int RowReductionThr = 4;
|
||||
|
||||
// If true, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly (standard per-block FP4 quantization like V)
|
||||
// If false (default), use two-level quantization: s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1)
|
||||
bool single_level_p_quant;
|
||||
|
||||
CUTLASS_DEVICE SoftmaxFused(bool single_level = false) : single_level_p_quant(single_level) {};
|
||||
|
||||
template<bool FirstTile, bool InfCheck = false, typename TensorAcc, typename TensorMax>
|
||||
CUTLASS_DEVICE auto online_softmax_with_quant(
|
||||
TensorAcc& acc,
|
||||
TensorMax& AbsMaxP,
|
||||
const float softmax_scale_log2
|
||||
) {
|
||||
Tensor acc_reduction_view = make_tensor(acc.data(), flash::convert_to_reduction_layout(acc.layout()));
|
||||
Tensor acc_conversion_view = make_tensor(acc.data(), flash::convert_to_conversion_layout(acc.layout()));
|
||||
Tensor acc_conversion_flatten = group_modes<1, 5>(group_modes<0, 2>(flatten(acc_conversion_view)));
|
||||
|
||||
if constexpr (FirstTile) {
|
||||
fill(row_max, -INFINITY);
|
||||
clear(row_sum);
|
||||
fill(scores_scale, 1.f);
|
||||
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
|
||||
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), acc_reduction_view(mi, make_coord(ei, ni)));
|
||||
}
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), AbsMaxP(mi, ni), 1); // exchange max with neighbour thread of 8 elements
|
||||
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), max_recv);
|
||||
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
|
||||
}
|
||||
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
|
||||
row_max(mi) = fmaxf(row_max(mi), max_recv);
|
||||
|
||||
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
|
||||
// - Pre-scales P to [0, 448×6] range before φ, output scaled by s_P1
|
||||
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
|
||||
// - No s_P1, just standard per-block FP4 quantization φ
|
||||
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
|
||||
const float max_scaled = InfCheck
|
||||
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
|
||||
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
|
||||
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
|
||||
}
|
||||
// s_P2 = max(P_block)/6 — per-block scale factor from φ function (same formula for both modes)
|
||||
// The difference is in max_scaled: two-level includes 448×6 pre-scaling, single-level doesn't
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
|
||||
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
|
||||
}
|
||||
}
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
|
||||
row_sum(mi) += acc_reduction_view(mi, ni);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
Tensor scores_max_prev = make_fragment_like(row_max);
|
||||
cute::copy(row_max, scores_max_prev);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
|
||||
float local_max = -INFINITY;
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
|
||||
local_max = fmaxf(local_max, acc_reduction_view(mi, make_coord(ei, ni)));
|
||||
}
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), local_max, 1); // exchange max with neighbour thread of 8 elements
|
||||
AbsMaxP(mi, ni) = fmaxf(local_max, max_recv);
|
||||
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
|
||||
}
|
||||
|
||||
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
|
||||
row_max(mi) = fmaxf(row_max(mi), max_recv);
|
||||
|
||||
float scores_max_cur = !InfCheck
|
||||
? row_max(mi)
|
||||
: (row_max(mi) == -INFINITY ? 0.0f : row_max(mi));
|
||||
scores_scale(mi) = flash::ptx_exp2((scores_max_prev(mi) - scores_max_cur) * softmax_scale_log2);
|
||||
|
||||
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
|
||||
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
|
||||
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
|
||||
const float max_scaled = InfCheck
|
||||
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
|
||||
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
|
||||
row_sum(mi) = row_sum(mi) * scores_scale(mi);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
|
||||
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
|
||||
row_sum(mi) += acc_reduction_view(mi, ni);
|
||||
}
|
||||
// s_P2 = max(P_block)/6 — per-block scale factor from φ function
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
|
||||
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
|
||||
}
|
||||
// scores_scale(mi) = max_scaled;
|
||||
}
|
||||
}
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size(AbsMaxP); ++i) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int j = 0; j < size<0>(acc_conversion_flatten); ++j)
|
||||
acc_conversion_flatten(j, i) /= AbsMaxP(i);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename TensorAcc>
|
||||
CUTLASS_DEVICE void finalize(TensorAcc& o_store) {
|
||||
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size(row_max); ++mi) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 1; i < RowReductionThr; i <<= 1) {
|
||||
float sum_recv = __shfl_xor_sync(int32_t(-1), row_sum(mi), i);
|
||||
row_sum(mi) += sum_recv;
|
||||
}
|
||||
float sum = row_sum(mi);
|
||||
float inv_sum = (sum == 0.f || sum != sum) ? 0.f : 1 / sum;
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
|
||||
o_store_reduction_view(mi, ni) *= inv_sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename TensorAcc>
|
||||
CUTLASS_DEVICE void rescale_o(TensorAcc& o_store, TensorAcc const& o_tmp) {
|
||||
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
|
||||
Tensor o_tmp_reduction_view = make_tensor(o_tmp.data(), flash::convert_to_reduction_layout(o_tmp.layout()));
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int mi = 0; mi < size(row_max); ++mi) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
|
||||
o_store_reduction_view(mi, ni) = o_store_reduction_view(mi, ni) * scores_scale(mi) + o_tmp_reduction_view(mi, ni);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
};
|
||||
} // namespace flash
|
||||
@@ -1,83 +0,0 @@
|
||||
// Inspired by
|
||||
// https://github.com/NVIDIA/DALI/blob/main/include/dali/core/static_switch.h
|
||||
// and https://github.com/pytorch/pytorch/blob/master/aten/src/ATen/Dispatch.h
|
||||
|
||||
#pragma once
|
||||
|
||||
/// @param COND - a boolean expression to switch by
|
||||
/// @param CONST_NAME - a name given for the constexpr bool variable.
|
||||
/// @param ... - code to execute for true and false
|
||||
///
|
||||
/// Usage:
|
||||
/// ```
|
||||
/// BOOL_SWITCH(flag, BoolConst, [&] {
|
||||
/// some_function<BoolConst>(...);
|
||||
/// });
|
||||
/// ```
|
||||
//
|
||||
|
||||
#define BOOL_SWITCH(COND, CONST_NAME, ...) \
|
||||
[&] { \
|
||||
if (COND) { \
|
||||
constexpr static bool CONST_NAME = true; \
|
||||
return __VA_ARGS__(); \
|
||||
} else { \
|
||||
constexpr static bool CONST_NAME = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
|
||||
#define PREC_SWITCH(PRECTYPE, ...) \
|
||||
[&] { \
|
||||
if (PRECTYPE == 1) { \
|
||||
using kPrecType = cutlass::half_t; \
|
||||
constexpr static bool kSoftFp16 = false; \
|
||||
constexpr static bool kHybrid = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (PRECTYPE == 2) { \
|
||||
using kPrecType = cutlass::float_e4m3_t; \
|
||||
constexpr static bool kSoftFp16 = false; \
|
||||
constexpr static bool kHybrid = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (PRECTYPE == 3) { \
|
||||
using kPrecType = cutlass::float_e4m3_t; \
|
||||
constexpr static bool kSoftFp16 = false; \
|
||||
constexpr static bool kHybrid = true; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (PRECTYPE == 4) { \
|
||||
using kPrecType = cutlass::float_e4m3_t; \
|
||||
constexpr static bool kSoftFp16 = true; \
|
||||
constexpr static bool kHybrid = false; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
|
||||
#define HEADDIM_SWITCH(HEADDIM, ...) \
|
||||
[&] { \
|
||||
if (HEADDIM == 64) { \
|
||||
constexpr static int kHeadSize = 64; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (HEADDIM == 128) { \
|
||||
constexpr static int kHeadSize = 128; \
|
||||
return __VA_ARGS__(); \
|
||||
} else if (HEADDIM == 256) { \
|
||||
constexpr static int kHeadSize = 256; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
|
||||
#define SEQLEN_SWITCH(USE_VAR_SEQ_LEN, SEQ_LEN_OUT_OF_BOUND_CHECK, ...) \
|
||||
[&] { \
|
||||
if (!USE_VAR_SEQ_LEN) { \
|
||||
if (SEQ_LEN_OUT_OF_BOUND_CHECK) { \
|
||||
using kSeqLenTraitsType = FixedSeqLenTraits<true>; \
|
||||
return __VA_ARGS__(); \
|
||||
} else { \
|
||||
using kSeqLenTraitsType = FixedSeqLenTraits<false>; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
} else { \
|
||||
using kSeqLenTraitsType = VarSeqLenTraits; \
|
||||
return __VA_ARGS__(); \
|
||||
} \
|
||||
}()
|
||||
@@ -1,304 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
|
||||
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/fast_math.h"
|
||||
|
||||
namespace flash {
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
class StaticPersistentTileSchedulerOld {
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
private:
|
||||
int current_work_linear_idx_;
|
||||
cutlass::FastDivmod const &m_block_divmod, &head_divmod;
|
||||
int const total_blocks;
|
||||
|
||||
public:
|
||||
struct WorkTileInfo {
|
||||
int M_idx = 0;
|
||||
int H_idx = 0;
|
||||
int B_idx = 0;
|
||||
bool is_valid_tile = false;
|
||||
|
||||
CUTLASS_HOST_DEVICE
|
||||
bool
|
||||
is_valid() const {
|
||||
return is_valid_tile;
|
||||
}
|
||||
|
||||
CUTLASS_HOST_DEVICE
|
||||
static WorkTileInfo
|
||||
invalid_work_tile() {
|
||||
return {-1, -1, -1, false};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
public:
|
||||
|
||||
CUTLASS_DEVICE explicit StaticPersistentTileSchedulerOld(cutlass::FastDivmod const &m_block_divmod_,
|
||||
cutlass::FastDivmod const &head_divmod_,
|
||||
int const total_blocks_) :
|
||||
m_block_divmod(m_block_divmod_), head_divmod(head_divmod_), total_blocks(total_blocks_) {
|
||||
|
||||
// MSVC requires protecting use of CUDA-specific nonstandard syntax,
|
||||
// like blockIdx and gridDim, with __CUDA_ARCH__.
|
||||
#if defined(__CUDA_ARCH__)
|
||||
// current_work_linear_idx_ = blockIdx.x + blockIdx.y * gridDim.x + blockIdx.z * gridDim.x * gridDim.y;
|
||||
current_work_linear_idx_ = blockIdx.x;
|
||||
#else
|
||||
CUTLASS_ASSERT(false && "This line should never be reached");
|
||||
#endif
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_current_work() const {
|
||||
return get_current_work_for_linear_idx(current_work_linear_idx_);
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_current_work_for_linear_idx(int linear_idx) const {
|
||||
if (linear_idx >= total_blocks) {
|
||||
return WorkTileInfo::invalid_work_tile();
|
||||
}
|
||||
|
||||
// Map worker's linear index into the CTA tiled problem shape to the corresponding MHB indices
|
||||
int M_idx, H_idx, B_idx;
|
||||
int quotient = m_block_divmod.divmod(M_idx, linear_idx);
|
||||
B_idx = head_divmod.divmod(H_idx, quotient);
|
||||
return {M_idx, H_idx, B_idx, true};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
void
|
||||
// advance_to_next_work(int advance_count = 1) {
|
||||
advance_to_next_work() {
|
||||
// current_work_linear_idx_ += int(gridDim.x * gridDim.y * gridDim.z);
|
||||
current_work_linear_idx_ += int(gridDim.x);
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
fetch_next_work() {
|
||||
WorkTileInfo new_work_tile_info;
|
||||
advance_to_next_work();
|
||||
new_work_tile_info = get_current_work();
|
||||
return new_work_tile_info;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
class SingleTileScheduler {
|
||||
|
||||
public:
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
int const num_blocks_m, num_head, num_batch;
|
||||
int const* tile_count_semaphore = nullptr;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
return {};
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_grid_dim(Arguments const& args, int num_sm) {
|
||||
return {uint32_t(args.num_blocks_m), uint32_t(args.num_head), uint32_t(args.num_batch)};
|
||||
}
|
||||
|
||||
struct WorkTileInfo {
|
||||
int M_idx = 0;
|
||||
int H_idx = 0;
|
||||
int B_idx = 0;
|
||||
bool is_valid_tile = false;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
bool
|
||||
is_valid(Params const& params) const {
|
||||
return is_valid_tile;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
cute::tuple<int32_t, int32_t, int32_t>
|
||||
get_block_coord(Params const& params) const {
|
||||
return {M_idx, H_idx, B_idx};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params) const {
|
||||
return {-1, -1, -1, false};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_initial_work() const {
|
||||
return {int(blockIdx.x), int(blockIdx.y), int(blockIdx.z), true};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
|
||||
return {-1, -1, -1, false};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
class StaticPersistentTileScheduler {
|
||||
|
||||
public:
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
int const num_blocks_m, num_head, num_batch;
|
||||
int const* tile_count_semaphore = nullptr;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
int total_blocks;
|
||||
cutlass::FastDivmod m_block_divmod, head_divmod;
|
||||
};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
return {args.num_blocks_m * args.num_head * args.num_batch,
|
||||
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head)};
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_grid_dim(Arguments const& args, int num_sm) {
|
||||
return {uint32_t(num_sm)};
|
||||
}
|
||||
|
||||
struct WorkTileInfo {
|
||||
int tile_idx;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
bool
|
||||
is_valid(Params const& params) const {
|
||||
return tile_idx < params.total_blocks;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
cute::tuple<int32_t, int32_t, int32_t>
|
||||
get_block_coord(Params const& params) const {
|
||||
int m_block, bidh, bidb;
|
||||
bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
|
||||
return {m_block, bidh, bidb};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_initial_work() const {
|
||||
return {int(blockIdx.x)};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
|
||||
return {current_work.tile_idx + int(gridDim.x)};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
class DynamicPersistentTileScheduler {
|
||||
|
||||
public:
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
int const num_blocks_m, num_head, num_batch;
|
||||
int const* tile_count_semaphore;
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
struct Params {
|
||||
int const total_blocks;
|
||||
cutlass::FastDivmod const m_block_divmod, head_divmod;
|
||||
int const* tile_count_semaphore;
|
||||
};
|
||||
|
||||
static Params
|
||||
to_underlying_arguments(Arguments const& args) {
|
||||
return {args.num_blocks_m * args.num_head * args.num_batch,
|
||||
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head),
|
||||
args.tile_count_semaphore};
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_grid_dim(Arguments const& args, int num_sm) {
|
||||
return {uint32_t(num_sm)};
|
||||
}
|
||||
|
||||
using WorkTileInfo = StaticPersistentTileScheduler::WorkTileInfo;
|
||||
// struct WorkTileInfo {
|
||||
// int tile_idx;
|
||||
|
||||
// CUTLASS_DEVICE
|
||||
// bool
|
||||
// is_valid(Params const& params) const {
|
||||
// return tile_idx < params.total_blocks;
|
||||
// }
|
||||
|
||||
// CUTLASS_DEVICE
|
||||
// cute::tuple<int32_t, int32_t, int32_t>
|
||||
// get_block_coord(Params const& params) const {
|
||||
// int m_block, bidh, bidb;
|
||||
// bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
|
||||
// return {m_block, bidh, bidb};
|
||||
// }
|
||||
|
||||
// };
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_initial_work() const {
|
||||
return {int(blockIdx.x)};
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
WorkTileInfo
|
||||
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
|
||||
return {current_work.tile_idx + int(gridDim.x)};
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
} // flash
|
||||
@@ -1,408 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <assert.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
#include <cuda_fp16.h>
|
||||
|
||||
#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800
|
||||
#include <cuda_bf16.h>
|
||||
#endif
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
|
||||
#include <cutlass/array.h>
|
||||
#include <cutlass/cutlass.h>
|
||||
#include <cutlass/numeric_conversion.h>
|
||||
#include <cutlass/numeric_types.h>
|
||||
|
||||
namespace flash {
|
||||
|
||||
using namespace cute;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename T>
|
||||
struct MaxOp {
|
||||
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x > y ? x : y; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct MaxOp<float> {
|
||||
// This is slightly faster
|
||||
__device__ __forceinline__ float operator()(float const &x, float const &y) { return max(x, y); }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename T>
|
||||
struct SumOp {
|
||||
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x + y; }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<int THREADS>
|
||||
struct Allreduce {
|
||||
static_assert(THREADS == 32 || THREADS == 16 || THREADS == 8 || THREADS == 4);
|
||||
template<typename T, typename Operator>
|
||||
static __device__ __forceinline__ T run(T x, Operator &op) {
|
||||
constexpr int OFFSET = THREADS / 2;
|
||||
x = op(x, __shfl_xor_sync(uint32_t(-1), x, OFFSET));
|
||||
return Allreduce<OFFSET>::run(x, op);
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<>
|
||||
struct Allreduce<2> {
|
||||
template<typename T, typename Operator>
|
||||
static __device__ __forceinline__ T run(T x, Operator &op) {
|
||||
x = op(x, __shfl_xor_sync(uint32_t(-1), x, 1));
|
||||
return x;
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
|
||||
__device__ __forceinline__ void thread_reduce_(Tensor<Engine0, Layout0> const &tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
|
||||
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
|
||||
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
|
||||
CUTE_STATIC_ASSERT_V(size<0>(summary) == size<0>(tensor));
|
||||
#pragma unroll
|
||||
for (int mi = 0; mi < size<0>(tensor); mi++) {
|
||||
summary(mi) = zero_init ? tensor(mi, 0) : op(summary(mi), tensor(mi, 0));
|
||||
#pragma unroll
|
||||
for (int ni = 1; ni < size<1>(tensor); ni++) {
|
||||
summary(mi) = op(summary(mi), tensor(mi, ni));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
|
||||
__device__ __forceinline__ void quad_allreduce_(Tensor<Engine0, Layout0> &dst, Tensor<Engine1, Layout1> &src, Operator &op) {
|
||||
CUTE_STATIC_ASSERT_V(size(dst) == size(src));
|
||||
#pragma unroll
|
||||
for (int i = 0; i < size(dst); i++){
|
||||
dst(i) = Allreduce<4>::run(src(i), op);
|
||||
}
|
||||
}
|
||||
|
||||
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
|
||||
__device__ __forceinline__ void reduce_(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
|
||||
thread_reduce_<zero_init>(tensor, summary, op);
|
||||
quad_allreduce_(summary, summary, op);
|
||||
}
|
||||
|
||||
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__device__ __forceinline__ void reduce_max(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &max){
|
||||
MaxOp<float> max_op;
|
||||
reduce_<zero_init>(tensor, max, max_op);
|
||||
}
|
||||
|
||||
template<bool zero_init=true, bool warp_reduce=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__device__ __forceinline__ void reduce_sum(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &sum){
|
||||
SumOp<float> sum_op;
|
||||
thread_reduce_<zero_init>(tensor, sum, sum_op);
|
||||
if constexpr (warp_reduce) { quad_allreduce_(sum, sum, sum_op); }
|
||||
}
|
||||
|
||||
__forceinline__ __device__ __half2 half_exp(__half2 x) {
|
||||
uint32_t tmp_out, tmp_in;
|
||||
tmp_in = reinterpret_cast<uint32_t&>(x);
|
||||
asm ("ex2.approx.f16x2 %0, %1;\n"
|
||||
: "=r"(tmp_out)
|
||||
: "r"(tmp_in));
|
||||
__half2 out = reinterpret_cast<__half2&>(tmp_out);
|
||||
return out;
|
||||
}
|
||||
|
||||
// Apply the exp to all the elements.
|
||||
template <bool zero_init=false, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__forceinline__ __device__ void max_scale_exp2_sum(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> &max, Tensor<Engine1, Layout1> &sum, const float scale) {
|
||||
static_assert(Layout0::rank == 2, "Only support 2D Tensor"); static_assert(Layout1::rank == 1, "Only support 1D Tensor"); CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
|
||||
#pragma unroll
|
||||
for (int mi = 0; mi < size<0>(tensor); ++mi) {
|
||||
MaxOp<float> max_op;
|
||||
max(mi) = zero_init ? tensor(mi, 0) : max_op(max(mi), tensor(mi, 0));
|
||||
#pragma unroll
|
||||
for (int ni = 1; ni < size<1>(tensor); ni++) {
|
||||
max(mi) = max_op(max(mi), tensor(mi, ni));
|
||||
}
|
||||
max(mi) = Allreduce<4>::run(max(mi), max_op);
|
||||
// If max is -inf, then all elements must have been -inf (possibly due to masking).
|
||||
// We don't want (-inf - (-inf)) since that would give NaN.
|
||||
const float max_scaled = max(mi) == -INFINITY ? 0.f : max(mi) * scale;
|
||||
sum(mi) = 0;
|
||||
#pragma unroll
|
||||
for (int ni = 0; ni < size<1>(tensor); ++ni) {
|
||||
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
|
||||
// max * log_2(e)) This allows the compiler to use the ffma
|
||||
// instruction instead of fadd and fmul separately.
|
||||
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
|
||||
sum(mi) += tensor(mi, ni);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Apply the exp to all the elements.
|
||||
template <bool Scale_max=true, bool Check_inf=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
|
||||
__forceinline__ __device__ void scale_apply_exp2(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> const &max, const float scale) {
|
||||
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
|
||||
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
|
||||
CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
|
||||
#pragma unroll
|
||||
for (int mi = 0; mi < size<0>(tensor); ++mi) {
|
||||
// If max is -inf, then all elements must have been -inf (possibly due to masking).
|
||||
// We don't want (-inf - (-inf)) since that would give NaN.
|
||||
// If we don't have float around M_LOG2E the multiplication is done in fp64.
|
||||
const float max_scaled = Check_inf
|
||||
? (max(mi) == -INFINITY ? 0.f : (max(mi) * (Scale_max ? scale : float(M_LOG2E))))
|
||||
: (max(mi) * (Scale_max ? scale : float(M_LOG2E)));
|
||||
#pragma unroll
|
||||
for (int ni = 0; ni < size<1>(tensor); ++ni) {
|
||||
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
|
||||
// max * log_2(e)) This allows the compiler to use the ffma
|
||||
// instruction instead of fadd and fmul separately.
|
||||
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
__forceinline__ __device__ float ptx_exp2(float x) {
|
||||
float y;
|
||||
asm volatile("ex2.approx.ftz.f32 %0, %1;" : "=f"(y) : "f"(x));
|
||||
return y;
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
packed_float_to_ue4m3(
|
||||
float const &f0, float const &f1, float const &f2, float const &f3,
|
||||
uint32_t &out
|
||||
) {
|
||||
asm volatile( \
|
||||
"{\n" \
|
||||
".reg .b16 lo;\n" \
|
||||
".reg .b16 hi;\n" \
|
||||
"cvt.rn.satfinite.e4m3x2.f32 lo, %2, %1;\n" \
|
||||
"cvt.rn.satfinite.e4m3x2.f32 hi, %4, %3;\n" \
|
||||
"mov.b32 %0, {lo, hi};\n" \
|
||||
"}" \
|
||||
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
packed_float_to_e2m1(
|
||||
float const &f0, float const &f1, float const &f2, float const& f3,
|
||||
float const &f4, float const &f5, float const &f6, float const& f7,
|
||||
uint32_t &out
|
||||
) {
|
||||
|
||||
asm volatile( \
|
||||
"{\n" \
|
||||
".reg .b8 byte0;\n" \
|
||||
".reg .b8 byte1;\n" \
|
||||
".reg .b8 byte2;\n" \
|
||||
".reg .b8 byte3;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n" \
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n" \
|
||||
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n" \
|
||||
"}" \
|
||||
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3),
|
||||
"f"(f4), "f"(f5), "f"(f6), "f"(f7));
|
||||
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
add(float2 & c,
|
||||
float2 const& a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("add.f32x2 %0, %1, %2;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(c))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
add_inplace(float2 &a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("add.f32x2 %0, %0, %1;\n"
|
||||
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
|
||||
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
sub(float2 & c,
|
||||
float2 const& a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("sub.f32x2 %0, %1, %2;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(c))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
sub_inplace(float2 &a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("sub.f32x2 %0, %0, %1;\n"
|
||||
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
|
||||
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
mul(float2 & c,
|
||||
float2 const& a,
|
||||
float2 const& b)
|
||||
{
|
||||
asm volatile("mul.f32x2 %0, %1, %2;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(c))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
fma(float2 & d,
|
||||
float2 const& a,
|
||||
float2 const& b,
|
||||
float2 const& c)
|
||||
{
|
||||
asm volatile("fma.rn.f32x2 %0, %1, %2, %3;\n"
|
||||
: "=l"(reinterpret_cast<uint64_t &>(d))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(a)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(b)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(c)));
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
fma_inplace(float2 &a,
|
||||
float2 const& b,
|
||||
float2 const& c)
|
||||
{
|
||||
asm volatile("fma.rn.f32x2 %0, %0, %1, %2;\n"
|
||||
: "+l"(reinterpret_cast<uint64_t &>(a))
|
||||
: "l"(reinterpret_cast<uint64_t const&>(b)),
|
||||
"l"(reinterpret_cast<uint64_t const&>(c)));
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <
|
||||
class Layout
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_reduction_layout(Layout mma_layout) {
|
||||
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
|
||||
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
|
||||
|
||||
return make_layout(
|
||||
make_layout(get<0,1>(mma_layout), get<1>(mma_layout)),
|
||||
make_layout(get<0,0>(mma_layout), get<2>(mma_layout))
|
||||
);
|
||||
}
|
||||
|
||||
template <
|
||||
class Tensor
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_reduction_tensor(Tensor mma_tensor) {
|
||||
return make_tensor(mma_tensor.data(), convert_to_reduction_layout(mma_tensor.layout()));
|
||||
}
|
||||
|
||||
|
||||
template <
|
||||
class Layout
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_conversion_layout(Layout mma_layout) {
|
||||
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
|
||||
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
|
||||
|
||||
constexpr int MmaAtomN = size<0, 0>(mma_layout);
|
||||
constexpr int MmaAtomM = size<0, 1>(mma_layout);
|
||||
constexpr int MmaM = size<1>(mma_layout);
|
||||
constexpr int MmaN = size<2>(mma_layout);
|
||||
|
||||
static_assert(MmaAtomN == 8, "MmaAtomN should be 8.");
|
||||
static_assert(MmaAtomM == 2, "MmaAtomM should be 2.");
|
||||
static_assert(MmaN % 2 == 0, "MmaN should be multiple of 2.");
|
||||
|
||||
auto mma_n_division = zipped_divide(
|
||||
layout<2>(mma_layout), make_tile(_2{})
|
||||
);
|
||||
return make_layout(
|
||||
make_layout(layout<0,0>(mma_layout), make_layout(layout<0,1>(mma_layout), layout<0>(mma_n_division))),
|
||||
layout<1>(mma_layout), layout<1>(mma_n_division)
|
||||
);
|
||||
}
|
||||
|
||||
template <
|
||||
class Tensor
|
||||
>
|
||||
CUTLASS_DEVICE constexpr
|
||||
auto convert_to_conversion_tensor(Tensor mma_tensor) {
|
||||
return make_tensor(mma_tensor.data(), convert_to_conversion_layout(mma_tensor.layout()));
|
||||
}
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <bool Is_even_MN=true, bool Is_even_K=true, bool Clear_OOB_MN=false, bool Clear_OOB_K=true,
|
||||
typename TiledCopy, typename Engine0, typename Layout0, typename Engine1, typename Layout1,
|
||||
typename Engine2, typename Layout2, typename Engine3, typename Layout3>
|
||||
CUTLASS_DEVICE void copy(TiledCopy tiled_copy, Tensor<Engine0, Layout0> const &S,
|
||||
Tensor<Engine1, Layout1> &D, Tensor<Engine2, Layout2> const &identity_MN,
|
||||
Tensor<Engine3, Layout3> const &predicate_K, const int max_MN=0) {
|
||||
CUTE_STATIC_ASSERT_V(rank(S) == Int<3>{});
|
||||
CUTE_STATIC_ASSERT_V(rank(D) == Int<3>{});
|
||||
CUTE_STATIC_ASSERT_V(size<0>(S) == size<0>(D)); // MMA
|
||||
CUTE_STATIC_ASSERT_V(size<1>(S) == size<1>(D)); // MMA_M
|
||||
CUTE_STATIC_ASSERT_V(size<2>(S) == size<2>(D)); // MMA_K
|
||||
// There's no case where !Clear_OOB_K && Clear_OOB_MN
|
||||
static_assert(!(Clear_OOB_MN && !Clear_OOB_K));
|
||||
#pragma unroll
|
||||
for (int m = 0; m < size<1>(S); ++m) {
|
||||
if (Is_even_MN || get<0>(identity_MN(0, m, 0)) < max_MN) {
|
||||
#pragma unroll
|
||||
for (int k = 0; k < size<2>(S); ++k) {
|
||||
if (Is_even_K || predicate_K(k)) {
|
||||
cute::copy(tiled_copy, S(_, m, k), D(_, m, k));
|
||||
} else if (Clear_OOB_K) {
|
||||
cute::clear(D(_, m, k));
|
||||
}
|
||||
}
|
||||
} else if (Clear_OOB_MN) {
|
||||
cute::clear(D(_, m, _));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace flash
|
||||
@@ -1 +0,0 @@
|
||||
__version__ = "3.0.0.b1"
|
||||
@@ -1,90 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import fp4quant
|
||||
from triton.tools.mxfp import MXFP4Tensor
|
||||
|
||||
from bench_utils import bench_kineto
|
||||
b = 1
|
||||
h = 32
|
||||
n = 16384
|
||||
d = 128
|
||||
|
||||
def test():
|
||||
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
|
||||
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
|
||||
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
|
||||
fp4quant.scaled_fp4_quant_permute(q, o, o_s, 1)
|
||||
|
||||
test()
|
||||
|
||||
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
|
||||
|
||||
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
|
||||
throughput = IO / t * 1e-9
|
||||
|
||||
print(f"Throughput: {throughput:.2f} GB/s")
|
||||
|
||||
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
|
||||
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
|
||||
B, H, M, N = x.shape
|
||||
x = x.view(B, H, M, N // 16, 16)
|
||||
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
|
||||
if all_ones:
|
||||
scales = torch.ones_like(scales)
|
||||
x_scaled = x / scales
|
||||
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
|
||||
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
|
||||
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
|
||||
permuted_fp8_scale = None
|
||||
if permuted:
|
||||
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
|
||||
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
|
||||
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
|
||||
|
||||
b = 2
|
||||
h = 4
|
||||
n = 251
|
||||
n_padded = (n + 127) // 128 * 128
|
||||
d = 128
|
||||
|
||||
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
|
||||
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
|
||||
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
|
||||
|
||||
k_permute = [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
|
||||
o_permuted = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s_permuted = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
fp4quant.scaled_fp4_quant_permute(q, o_permuted, o_s_permuted, 1)
|
||||
|
||||
# padding
|
||||
if n % 128 != 0:
|
||||
o_permuted_gt = torch.cat([o, torch.zeros((b, h, n_padded - n, d // 2), dtype=torch.uint8, device='cuda')], dim=2)
|
||||
o_s_permuted_gt = torch.cat([o_s, torch.zeros((b, h, n_padded - n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')], dim=2)
|
||||
else:
|
||||
o_permuted_gt = o
|
||||
o_s_permuted_gt = o_s
|
||||
|
||||
# use scale_and_fp4_tensor + torch permutation to get the ground truth
|
||||
o_permuted_gt = o_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 2)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 2)
|
||||
o_s_permuted_gt = o_s_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 16)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 16)
|
||||
|
||||
assert((o_permuted - o_permuted_gt).abs().max() == 0)
|
||||
assert((o_s_permuted.float() - o_s_permuted_gt.float()).abs().max() == 0)
|
||||
|
||||
print("All tests passed!")
|
||||
@@ -1,86 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import fp4quant
|
||||
from triton.tools.mxfp import MXFP4Tensor
|
||||
|
||||
from bench_utils import bench_kineto
|
||||
b = 1
|
||||
h = 32
|
||||
n = 16384
|
||||
d = 128
|
||||
|
||||
def test():
|
||||
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
|
||||
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
|
||||
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
|
||||
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
|
||||
|
||||
test()
|
||||
|
||||
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
|
||||
|
||||
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
|
||||
throughput = IO / t * 1e-9
|
||||
|
||||
print(f"Throughput: {throughput:.2f} GB/s")
|
||||
|
||||
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
|
||||
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
|
||||
B, H, M, N = x.shape
|
||||
x = x.view(B, H, M, N // 16, 16)
|
||||
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
|
||||
if all_ones:
|
||||
scales = torch.ones_like(scales)
|
||||
x_scaled = x / scales
|
||||
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
|
||||
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
|
||||
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
|
||||
permuted_fp8_scale = None
|
||||
if permuted:
|
||||
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
|
||||
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
|
||||
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
|
||||
|
||||
b = 2
|
||||
h = 4
|
||||
n = 251
|
||||
d = 128
|
||||
|
||||
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
|
||||
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
|
||||
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
|
||||
|
||||
fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale = scale_and_fp4_tensor(q, packed_dim=3)
|
||||
|
||||
assert((fp8_scale.float() - o_s.float()).abs().max() == 0)
|
||||
|
||||
o_binary = [
|
||||
(int(bin_str[:4], 2), int(bin_str[4:], 2))
|
||||
for bin_str in [format(x.item(), '08b') for x in o.view(-1)]
|
||||
]
|
||||
o_binary_gt = [
|
||||
(int(bin_str[:4], 2), int(bin_str[4:], 2))
|
||||
for bin_str in [format(x.item(), '08b') for x in packed_fp4.view(-1)]
|
||||
]
|
||||
for i in range(len(o_binary)):
|
||||
# check contiguous 4 bits. Difference should be at most one
|
||||
assert(abs(o_binary[i][0] - o_binary_gt[i][0]) <= 1)
|
||||
assert(abs(o_binary[i][1] - o_binary_gt[i][1]) <= 1)
|
||||
|
||||
print("All tests passed!")
|
||||
@@ -1,86 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import torch
|
||||
import fp4quant
|
||||
from triton.tools.mxfp import MXFP4Tensor
|
||||
|
||||
from bench_utils import bench_kineto
|
||||
b = 1
|
||||
h = 32
|
||||
n = 16384
|
||||
d = 128
|
||||
|
||||
def test():
|
||||
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
|
||||
o = torch.empty((b, h, d, n // 2), device="cuda", dtype=torch.uint8)
|
||||
o_s = torch.empty((b, h, d, n // 16), device="cuda", dtype=torch.float8_e4m3fn)
|
||||
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
|
||||
|
||||
test()
|
||||
|
||||
t = bench_kineto(test, "scaled_fp4_quant_trans_kernel", suppress_kineto_output=True)
|
||||
|
||||
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
|
||||
throughput = IO / t * 1e-9
|
||||
|
||||
print(f"Throughput: {throughput:.2f} GB/s")
|
||||
|
||||
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
|
||||
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
|
||||
B, H, M, N = x.shape
|
||||
x = x.view(B, H, M, N // 16, 16)
|
||||
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
|
||||
if all_ones:
|
||||
scales = torch.ones_like(scales)
|
||||
x_scaled = x / scales
|
||||
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
|
||||
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
|
||||
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
|
||||
permuted_fp8_scale = None
|
||||
if permuted:
|
||||
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
|
||||
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
|
||||
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
|
||||
|
||||
b = 2
|
||||
h = 4
|
||||
n = 491
|
||||
n_padded = (n + 127) // 128 * 128
|
||||
d = 128
|
||||
|
||||
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
|
||||
o = torch.empty((b, h, d, n_padded // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s = torch.empty((b, h, d, n_padded // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
|
||||
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
|
||||
|
||||
if n % 128 != 0:
|
||||
q_padded = torch.cat([q, torch.zeros((b, h, n_padded - n, d), dtype=torch.float16, device='cuda')], dim=2)
|
||||
else:
|
||||
q_padded = q
|
||||
|
||||
# use torch transpose + scaled_fp4_quant to get the ground truth
|
||||
q_padded = q_padded.transpose(2, 3).reshape(b, h, n_padded, d).contiguous()
|
||||
o_gt = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
|
||||
o_s_gt = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
|
||||
fp4quant.scaled_fp4_quant(q_padded, o_gt, o_s_gt, 1)
|
||||
o_gt = o_gt.reshape(b, h, d, n_padded // 2).contiguous()
|
||||
o_s_gt = o_s_gt.reshape(b, h, d, n_padded // 16).contiguous()
|
||||
|
||||
assert((o_s_gt.float() - o_s.float()).abs().max() == 0)
|
||||
assert((o_gt - o).abs().max() == 0)
|
||||
|
||||
print("All tests passed!")
|
||||
@@ -1,169 +0,0 @@
|
||||
"""
|
||||
Copyright (c) 2025 by SageAttention team.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
import torch
|
||||
import torch.distributed as dist
|
||||
|
||||
|
||||
def bench(fn, num_warmups: int = 5, num_tests: int = 10,
|
||||
high_precision: bool = False):
|
||||
# Flush L2 cache with 256 MB data
|
||||
torch.cuda.synchronize()
|
||||
cache = torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda')
|
||||
cache.zero_()
|
||||
|
||||
# Warmup
|
||||
for _ in range(num_warmups):
|
||||
fn()
|
||||
|
||||
# Add a large kernel to eliminate the CPU launch overhead
|
||||
if high_precision:
|
||||
x = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
y = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
x @ y
|
||||
|
||||
# Testing
|
||||
start_event = torch.cuda.Event(enable_timing=True)
|
||||
end_event = torch.cuda.Event(enable_timing=True)
|
||||
start_event.record()
|
||||
for i in range(num_tests):
|
||||
fn()
|
||||
end_event.record()
|
||||
torch.cuda.synchronize()
|
||||
|
||||
return start_event.elapsed_time(end_event) / num_tests
|
||||
|
||||
|
||||
class empty_suppress:
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *_):
|
||||
pass
|
||||
|
||||
|
||||
class suppress_stdout_stderr:
|
||||
def __enter__(self):
|
||||
self.outnull_file = open(os.devnull, 'w')
|
||||
self.errnull_file = open(os.devnull, 'w')
|
||||
|
||||
self.old_stdout_fileno_undup = sys.stdout.fileno()
|
||||
self.old_stderr_fileno_undup = sys.stderr.fileno()
|
||||
|
||||
self.old_stdout_fileno = os.dup(sys.stdout.fileno())
|
||||
self.old_stderr_fileno = os.dup(sys.stderr.fileno())
|
||||
|
||||
self.old_stdout = sys.stdout
|
||||
self.old_stderr = sys.stderr
|
||||
|
||||
os.dup2(self.outnull_file.fileno(), self.old_stdout_fileno_undup)
|
||||
os.dup2(self.errnull_file.fileno(), self.old_stderr_fileno_undup)
|
||||
|
||||
sys.stdout = self.outnull_file
|
||||
sys.stderr = self.errnull_file
|
||||
return self
|
||||
|
||||
def __exit__(self, *_):
|
||||
sys.stdout = self.old_stdout
|
||||
sys.stderr = self.old_stderr
|
||||
|
||||
os.dup2(self.old_stdout_fileno, self.old_stdout_fileno_undup)
|
||||
os.dup2(self.old_stderr_fileno, self.old_stderr_fileno_undup)
|
||||
|
||||
os.close(self.old_stdout_fileno)
|
||||
os.close(self.old_stderr_fileno)
|
||||
|
||||
self.outnull_file.close()
|
||||
self.errnull_file.close()
|
||||
|
||||
|
||||
def bench_kineto(fn, kernel_names, num_tests: int = 30, suppress_kineto_output: bool = False,
|
||||
trace_path: str = None, barrier_comm_profiling: bool = False, flush_l2: bool = False):
|
||||
# Conflict with Nsight Systems
|
||||
using_nsys = os.environ.get('DG_NSYS_PROFILING', False)
|
||||
|
||||
# For some auto-tuning kernels with prints
|
||||
fn()
|
||||
|
||||
# Profile
|
||||
suppress = suppress_stdout_stderr if suppress_kineto_output and not using_nsys else empty_suppress
|
||||
with suppress():
|
||||
schedule = torch.profiler.schedule(wait=0, warmup=1, active=1, repeat=1) if not using_nsys else None
|
||||
profiler = torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CUDA], schedule=schedule) if not using_nsys else empty_suppress()
|
||||
with profiler:
|
||||
for i in range(2):
|
||||
# NOTES: use a large kernel and a barrier to eliminate the unbalanced CPU launch overhead
|
||||
if barrier_comm_profiling:
|
||||
lhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
rhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
|
||||
lhs @ rhs
|
||||
dist.all_reduce(torch.ones(1, dtype=torch.float, device='cuda'))
|
||||
for _ in range(num_tests):
|
||||
if flush_l2:
|
||||
torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda').zero_()
|
||||
fn()
|
||||
|
||||
if not using_nsys:
|
||||
profiler.step()
|
||||
|
||||
# Return 1 if using Nsight Systems
|
||||
if using_nsys:
|
||||
return 1
|
||||
|
||||
# Parse the profiling table
|
||||
assert isinstance(kernel_names, str) or isinstance(kernel_names, tuple)
|
||||
is_tupled = isinstance(kernel_names, tuple)
|
||||
prof_lines = profiler.key_averages().table(sort_by='cuda_time_total', max_name_column_width=100).split('\n')
|
||||
kernel_names = (kernel_names, ) if isinstance(kernel_names, str) else kernel_names
|
||||
assert all([isinstance(name, str) for name in kernel_names])
|
||||
for name in kernel_names:
|
||||
assert sum([name in line for line in prof_lines]) == 1, f'Errors of the kernel {name} in the profiling table'
|
||||
|
||||
# Save chrome traces
|
||||
if trace_path is not None:
|
||||
profiler.export_chrome_trace(trace_path)
|
||||
|
||||
# Return average kernel times
|
||||
units = {'ms': 1e3, 'us': 1e6}
|
||||
kernel_times = []
|
||||
for name in kernel_names:
|
||||
for line in prof_lines:
|
||||
if name in line:
|
||||
time_str = line.split()[-2]
|
||||
for unit, scale in units.items():
|
||||
if unit in time_str:
|
||||
kernel_times.append(float(time_str.replace(unit, '')) / scale)
|
||||
break
|
||||
break
|
||||
return tuple(kernel_times) if is_tupled else kernel_times[0]
|
||||
|
||||
|
||||
def calc_diff(x, y):
|
||||
x, y = x.double(), y.double()
|
||||
denominator = (x * x + y * y).sum()
|
||||
sim = 2 * (x * y).sum() / denominator
|
||||
return 1 - sim
|
||||
|
||||
|
||||
def count_bytes(tensors):
|
||||
total = 0
|
||||
for t in tensors:
|
||||
if isinstance(t, tuple):
|
||||
total += count_bytes(t)
|
||||
else:
|
||||
total += t.numel() * t.element_size()
|
||||
return total
|
||||
@@ -1,52 +0,0 @@
|
||||
#pragma once
|
||||
|
||||
#include <stdio.h>
|
||||
|
||||
#if defined(__HIPCC__)
|
||||
#define HOST_DEVICE_INLINE __host__ __device__
|
||||
#define DEVICE_INLINE __device__
|
||||
#define HOST_INLINE __host__
|
||||
#elif defined(__CUDACC__) || defined(_NVHPC_CUDA)
|
||||
#define HOST_DEVICE_INLINE __host__ __device__ __forceinline__
|
||||
#define DEVICE_INLINE __device__ __forceinline__
|
||||
#define HOST_INLINE __host__ __forceinline__
|
||||
#else
|
||||
#define HOST_DEVICE_INLINE inline
|
||||
#define DEVICE_INLINE inline
|
||||
#define HOST_INLINE inline
|
||||
#endif
|
||||
|
||||
#define CUDA_CHECK(cmd) \
|
||||
do { \
|
||||
cudaError_t e = cmd; \
|
||||
if (e != cudaSuccess) { \
|
||||
printf("Failed: Cuda error %s:%d '%s'\n", __FILE__, __LINE__, \
|
||||
cudaGetErrorString(e)); \
|
||||
exit(EXIT_FAILURE); \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
int64_t get_device_attribute(int64_t attribute, int64_t device_id) {
|
||||
static int value = [=]() {
|
||||
int device = static_cast<int>(device_id);
|
||||
if (device < 0) {
|
||||
CUDA_CHECK(cudaGetDevice(&device));
|
||||
}
|
||||
int value;
|
||||
CUDA_CHECK(cudaDeviceGetAttribute(
|
||||
&value, static_cast<cudaDeviceAttr>(attribute), device));
|
||||
return static_cast<int>(value);
|
||||
}();
|
||||
|
||||
return value;
|
||||
}
|
||||
|
||||
namespace cuda_utils {
|
||||
|
||||
template <typename T>
|
||||
HOST_DEVICE_INLINE constexpr std::enable_if_t<std::is_integral_v<T>, T>
|
||||
ceil_div(T a, T b) {
|
||||
return (a + b - 1) / b;
|
||||
}
|
||||
|
||||
}; // namespace cuda_utils
|
||||
@@ -1,629 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 by SageAttention team.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
#include <torch/all.h>
|
||||
#include <torch/python.h>
|
||||
#include <torch/nn/functional.h>
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <c10/cuda/CUDAGuard.h>
|
||||
#include <cuda_runtime_api.h>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <c10/cuda/CUDAGuard.h>
|
||||
|
||||
#include <cuda_fp8.h>
|
||||
|
||||
#include "cuda_utils.h"
|
||||
#include "../blackwell/block_config.h"
|
||||
|
||||
#define DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(pytorch_dtype, c_type, ...) \
|
||||
if (pytorch_dtype == at::ScalarType::Half) { \
|
||||
using c_type = half; \
|
||||
__VA_ARGS__ \
|
||||
} else if (pytorch_dtype == at::ScalarType::BFloat16) { \
|
||||
using c_type = nv_bfloat16; \
|
||||
__VA_ARGS__ \
|
||||
} else { \
|
||||
std::ostringstream oss; \
|
||||
oss << __PRETTY_FUNCTION__ << " failed to dispatch data type " << pytorch_dtype; \
|
||||
TORCH_CHECK(false, oss.str()); \
|
||||
}
|
||||
|
||||
#define DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, ...) \
|
||||
if (head_dim == 64) { \
|
||||
constexpr int HEAD_DIM = 64; \
|
||||
__VA_ARGS__ \
|
||||
} else if (head_dim == 128) { \
|
||||
constexpr int HEAD_DIM = 128; \
|
||||
__VA_ARGS__ \
|
||||
} else { \
|
||||
std::ostringstream err_msg; \
|
||||
err_msg << "Unsupported head dim: " << int(head_dim); \
|
||||
throw std::invalid_argument(err_msg.str()); \
|
||||
}
|
||||
|
||||
#define CHECK_CUDA(x) \
|
||||
TORCH_CHECK(x.is_cuda(), "Tensor " #x " must be on CUDA")
|
||||
#define CHECK_DTYPE(x, true_dtype) \
|
||||
TORCH_CHECK(x.dtype() == true_dtype, \
|
||||
"Tensor " #x " must have dtype (" #true_dtype ")")
|
||||
#define CHECK_DIMS(x, true_dim) \
|
||||
TORCH_CHECK(x.dim() == true_dim, \
|
||||
"Tensor " #x " must have dimension number (" #true_dim ")")
|
||||
#define CHECK_SHAPE(x, ...) \
|
||||
TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), \
|
||||
"Tensor " #x " must have shape (" #__VA_ARGS__ ")")
|
||||
#define CHECK_CONTIGUOUS(x) \
|
||||
TORCH_CHECK(x.is_contiguous(), "Tensor " #x " must be contiguous")
|
||||
#define CHECK_LASTDIM_CONTIGUOUS(x) \
|
||||
TORCH_CHECK(x.stride(-1) == 1, \
|
||||
"Tensor " #x " must be contiguous at the last dimension")
|
||||
|
||||
constexpr int CVT_FP4_ELTS_PER_THREAD = 16;
|
||||
|
||||
// Convert 4 float2 values into 8 e2m1 values (represented as one uint32_t).
|
||||
inline __device__ uint32_t fp32_vec_to_e2m1(float2 *array) {
|
||||
#if defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 1000)
|
||||
uint32_t val;
|
||||
asm volatile(
|
||||
"{\n"
|
||||
".reg .b8 byte0;\n"
|
||||
".reg .b8 byte1;\n"
|
||||
".reg .b8 byte2;\n"
|
||||
".reg .b8 byte3;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n"
|
||||
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n"
|
||||
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n"
|
||||
"}"
|
||||
: "=r"(val)
|
||||
: "f"(array[0].x), "f"(array[0].y), "f"(array[1].x), "f"(array[1].y),
|
||||
"f"(array[2].x), "f"(array[2].y), "f"(array[3].x), "f"(array[3].y));
|
||||
return val;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
// Get type2 from type or vice versa (applied to half and bfloat16)
|
||||
template <typename T>
|
||||
struct TypeConverter {
|
||||
using Type = half2;
|
||||
}; // keep for generality
|
||||
|
||||
template <>
|
||||
struct TypeConverter<half2> {
|
||||
using Type = half;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct TypeConverter<half> {
|
||||
using Type = half2;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct TypeConverter<__nv_bfloat162> {
|
||||
using Type = __nv_bfloat16;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct TypeConverter<__nv_bfloat16> {
|
||||
using Type = __nv_bfloat162;
|
||||
};
|
||||
|
||||
// Define a 32 bytes packed data type.
|
||||
template <class Type>
|
||||
struct PackedVec {
|
||||
typename TypeConverter<Type>::Type elts[8];
|
||||
};
|
||||
|
||||
template <uint32_t head_dim, uint32_t BLOCK_SIZE, bool permute, typename T>
|
||||
__global__ void scaled_fp4_quant_kernel(
|
||||
const T* input, uint8_t* output, uint8_t* output_sf,
|
||||
int batch_size, int num_heads, int num_tokens,
|
||||
int stride_bz_input, int stride_h_input, int stride_seq_input,
|
||||
int stride_bz_output, int stride_h_output, int stride_seq_output,
|
||||
int stride_bz_output_sf, int stride_h_output_sf, int stride_seq_output_sf) {
|
||||
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
|
||||
using PackedVec = PackedVec<T>;
|
||||
|
||||
const int batch_id = blockIdx.y;
|
||||
const int head_id = blockIdx.z;
|
||||
const int token_block_id = blockIdx.x;
|
||||
|
||||
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
|
||||
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
|
||||
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
|
||||
"Vec size is not matched.");
|
||||
|
||||
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
|
||||
|
||||
// load input
|
||||
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
|
||||
|
||||
int load_token_id;
|
||||
if constexpr (!permute) {
|
||||
load_token_id = token_id;
|
||||
} else {
|
||||
int local_token_id = threadIdx.x / NUM_THREADS_PER_TOKEN;
|
||||
int local_token_id_residue = local_token_id % 32;
|
||||
// [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
|
||||
load_token_id = token_block_id * BLOCK_SIZE + (local_token_id / 32) * 32 +
|
||||
(local_token_id_residue / 8) * 2 +
|
||||
((local_token_id_residue % 8) / 2) * 8 +
|
||||
(local_token_id_residue % 8) % 2;
|
||||
}
|
||||
|
||||
PackedVec in_vec;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
|
||||
}
|
||||
|
||||
if (load_token_id < num_tokens) {
|
||||
in_vec = reinterpret_cast<PackedVec const*>(input +
|
||||
batch_id * stride_bz_input + // batch dim
|
||||
head_id * stride_h_input + // head dim
|
||||
load_token_id * stride_seq_input + // seq dim
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
|
||||
}
|
||||
|
||||
// calculate max of every consecutive 16 elements
|
||||
auto localMax = __habs2(in_vec.elts[0]);
|
||||
#pragma unroll
|
||||
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
|
||||
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
|
||||
}
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
|
||||
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
|
||||
}
|
||||
|
||||
float vecMax = float(__hmax(localMax.x, localMax.y));
|
||||
|
||||
// scaling factor
|
||||
float SFValue = vecMax / 6.0f;
|
||||
uint8_t SFValueFP8;
|
||||
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
|
||||
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
|
||||
|
||||
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
|
||||
|
||||
// convert input to float2 and apply scale
|
||||
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
if constexpr (std::is_same<T, half>::value) {
|
||||
fp2Vals[i] = __half22float2(in_vec.elts[i]);
|
||||
} else {
|
||||
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
|
||||
}
|
||||
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
|
||||
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
|
||||
}
|
||||
|
||||
// convert to e2m1
|
||||
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
|
||||
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
|
||||
}
|
||||
|
||||
// save, do not check range
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
|
||||
reinterpret_cast<uint32_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
token_id * stride_seq_output +
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = e2m1Vals[0];
|
||||
} else {
|
||||
reinterpret_cast<uint64_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
token_id * stride_seq_output +
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
|
||||
}
|
||||
|
||||
uint8_t* output_sf_save_base = output_sf + batch_id * stride_bz_output_sf + head_id * stride_h_output_sf + (token_id / 64) * 64 * stride_seq_output_sf;
|
||||
uint32_t token_id_local = token_id % 64;
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
|
||||
uint32_t col_id_local = threadIdx.x % NUM_THREADS_PER_TOKEN;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
} else {
|
||||
if (threadIdx.x % 2 == 0) {
|
||||
uint32_t col_id_local = (threadIdx.x % NUM_THREADS_PER_TOKEN) / 2;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <uint32_t head_dim, uint32_t BLOCK_SIZE, typename T>
|
||||
__global__ void scaled_fp4_quant_trans_kernel(
|
||||
const T* input, uint8_t* output, uint8_t* output_sf,
|
||||
int batch_size, int num_heads, int num_tokens,
|
||||
int stride_bz_input, int stride_h_input, int stride_seq_input,
|
||||
int stride_bz_output, int stride_h_output, int stride_d_output,
|
||||
int stride_bz_output_sf, int stride_h_output_sf, int stride_d_output_sf) {
|
||||
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
|
||||
using PackedVec = PackedVec<T>;
|
||||
|
||||
const int batch_id = blockIdx.y;
|
||||
const int head_id = blockIdx.z;
|
||||
const int token_block_id = blockIdx.x;
|
||||
|
||||
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
|
||||
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
|
||||
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
|
||||
"Vec size is not matched.");
|
||||
|
||||
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
|
||||
constexpr uint32_t NUM_THREADS_PER_SEQ = BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD;
|
||||
|
||||
// load input
|
||||
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
|
||||
|
||||
PackedVec in_vec;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
|
||||
}
|
||||
|
||||
if (token_id < num_tokens) {
|
||||
in_vec = reinterpret_cast<PackedVec const*>(input +
|
||||
batch_id * stride_bz_input + // batch dim
|
||||
head_id * stride_h_input + // head dim
|
||||
token_id * stride_seq_input + // seq dim
|
||||
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
|
||||
}
|
||||
|
||||
// transpose
|
||||
__shared__ T shared_input[BLOCK_SIZE * head_dim];
|
||||
reinterpret_cast<PackedVec*>(shared_input)[threadIdx.x] = in_vec;
|
||||
__syncthreads();
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
in_vec.elts[i].x = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i) * head_dim];
|
||||
in_vec.elts[i].y = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i + 1) * head_dim];
|
||||
}
|
||||
|
||||
// calculate max of every consecutive 16 elements
|
||||
auto localMax = __habs2(in_vec.elts[0]);
|
||||
#pragma unroll
|
||||
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
|
||||
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
|
||||
}
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
|
||||
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
|
||||
}
|
||||
|
||||
float vecMax = float(__hmax(localMax.x, localMax.y));
|
||||
|
||||
// scaling factor
|
||||
float SFValue = vecMax / 6.0f;
|
||||
uint8_t SFValueFP8;
|
||||
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
|
||||
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
|
||||
|
||||
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
|
||||
|
||||
// convert input to float2 and apply scale
|
||||
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
|
||||
if constexpr (std::is_same<T, half>::value) {
|
||||
fp2Vals[i] = __half22float2(in_vec.elts[i]);
|
||||
} else {
|
||||
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
|
||||
}
|
||||
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
|
||||
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
|
||||
}
|
||||
|
||||
// convert to e2m1
|
||||
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
|
||||
#pragma unroll
|
||||
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
|
||||
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
|
||||
}
|
||||
|
||||
// save
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
|
||||
reinterpret_cast<uint32_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
|
||||
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = e2m1Vals[0];
|
||||
} else {
|
||||
reinterpret_cast<uint64_t*>(output +
|
||||
batch_id * stride_bz_output +
|
||||
head_id * stride_h_output +
|
||||
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
|
||||
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
|
||||
}
|
||||
|
||||
uint8_t *output_sf_save_base = output_sf +
|
||||
batch_id * stride_bz_output_sf +
|
||||
head_id * stride_h_output_sf +
|
||||
(threadIdx.x / NUM_THREADS_PER_SEQ / 64) * 64 * stride_d_output_sf;
|
||||
uint32_t row_id_local = (threadIdx.x / NUM_THREADS_PER_SEQ) % 64;
|
||||
|
||||
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
|
||||
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + threadIdx.x % NUM_THREADS_PER_SEQ;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
} else {
|
||||
if (threadIdx.x % 2 == 0) {
|
||||
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + (threadIdx.x % NUM_THREADS_PER_SEQ) / 2;
|
||||
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
|
||||
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
|
||||
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void scaled_fp4_quant(torch::Tensor const& input,
|
||||
torch::Tensor const& output,
|
||||
torch::Tensor const& output_sf,
|
||||
int tensor_layout) {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
|
||||
CHECK_CUDA(input);
|
||||
CHECK_CUDA(output);
|
||||
CHECK_CUDA(output_sf);
|
||||
|
||||
CHECK_LASTDIM_CONTIGUOUS(input);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output_sf);
|
||||
|
||||
CHECK_DTYPE(output, at::ScalarType::Byte);
|
||||
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
|
||||
|
||||
CHECK_DIMS(input, 4);
|
||||
CHECK_DIMS(output, 4);
|
||||
CHECK_DIMS(output_sf, 4);
|
||||
|
||||
const int batch_size = input.size(0);
|
||||
const int head_dim = input.size(3);
|
||||
|
||||
const int stride_bz_input = input.stride(0);
|
||||
const int stride_bz_output = output.stride(0);
|
||||
const int stride_bz_output_sf = output_sf.stride(0);
|
||||
|
||||
int num_tokens, num_heads;
|
||||
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
|
||||
int stride_h_input, stride_h_output, stride_h_output_sf;
|
||||
if (tensor_layout == 0) {
|
||||
num_tokens = input.size(1);
|
||||
num_heads = input.size(2);
|
||||
stride_seq_input = input.stride(1);
|
||||
stride_seq_output = output.stride(1);
|
||||
stride_seq_output_sf = output_sf.stride(1);
|
||||
stride_h_input = input.stride(2);
|
||||
stride_h_output = output.stride(2);
|
||||
stride_h_output_sf = output_sf.stride(2);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_tokens, num_heads, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_tokens, num_heads, head_dim / 16);
|
||||
} else {
|
||||
num_tokens = input.size(2);
|
||||
num_heads = input.size(1);
|
||||
stride_seq_input = input.stride(2);
|
||||
stride_seq_output = output.stride(2);
|
||||
stride_seq_output_sf = output_sf.stride(2);
|
||||
stride_h_input = input.stride(1);
|
||||
stride_h_output = output.stride(1);
|
||||
stride_h_output_sf = output_sf.stride(1);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_heads, num_tokens, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_heads, num_tokens, head_dim / 16);
|
||||
}
|
||||
|
||||
auto input_dtype = input.scalar_type();
|
||||
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
|
||||
|
||||
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
|
||||
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
|
||||
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
|
||||
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
|
||||
|
||||
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, false, c_type>
|
||||
<<<grid, block, 0, stream>>>(
|
||||
reinterpret_cast<c_type*>(input.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
|
||||
batch_size, num_heads, num_tokens,
|
||||
stride_bz_input, stride_h_input, stride_seq_input,
|
||||
stride_bz_output, stride_h_output, stride_seq_output,
|
||||
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
void scaled_fp4_quant_permute(torch::Tensor const& input,
|
||||
torch::Tensor const& output,
|
||||
torch::Tensor const& output_sf,
|
||||
int tensor_layout) {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
|
||||
CHECK_CUDA(input);
|
||||
CHECK_CUDA(output);
|
||||
CHECK_CUDA(output_sf);
|
||||
|
||||
CHECK_LASTDIM_CONTIGUOUS(input);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output_sf);
|
||||
|
||||
CHECK_DTYPE(output, at::ScalarType::Byte);
|
||||
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
|
||||
|
||||
CHECK_DIMS(input, 4);
|
||||
CHECK_DIMS(output, 4);
|
||||
CHECK_DIMS(output_sf, 4);
|
||||
|
||||
const int batch_size = input.size(0);
|
||||
const int head_dim = input.size(3);
|
||||
|
||||
const int stride_bz_input = input.stride(0);
|
||||
const int stride_bz_output = output.stride(0);
|
||||
const int stride_bz_output_sf = output_sf.stride(0);
|
||||
|
||||
int num_tokens, num_heads;
|
||||
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
|
||||
int stride_h_input, stride_h_output, stride_h_output_sf;
|
||||
if (tensor_layout == 0) {
|
||||
num_tokens = input.size(1);
|
||||
num_heads = input.size(2);
|
||||
stride_seq_input = input.stride(1);
|
||||
stride_seq_output = output.stride(1);
|
||||
stride_seq_output_sf = output_sf.stride(1);
|
||||
stride_h_input = input.stride(2);
|
||||
stride_h_output = output.stride(2);
|
||||
stride_h_output_sf = output_sf.stride(2);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 16);
|
||||
} else {
|
||||
num_tokens = input.size(2);
|
||||
num_heads = input.size(1);
|
||||
stride_seq_input = input.stride(2);
|
||||
stride_seq_output = output.stride(2);
|
||||
stride_seq_output_sf = output_sf.stride(2);
|
||||
stride_h_input = input.stride(1);
|
||||
stride_h_output = output.stride(1);
|
||||
stride_h_output_sf = output_sf.stride(1);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 16);
|
||||
}
|
||||
|
||||
auto input_dtype = input.scalar_type();
|
||||
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
|
||||
|
||||
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
|
||||
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
|
||||
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
|
||||
|
||||
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, true, c_type>
|
||||
<<<grid, block, 0, stream>>>(
|
||||
reinterpret_cast<c_type*>(input.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
|
||||
batch_size, num_heads, num_tokens,
|
||||
stride_bz_input, stride_h_input, stride_seq_input,
|
||||
stride_bz_output, stride_h_output, stride_seq_output,
|
||||
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
void scaled_fp4_quant_trans(torch::Tensor const& input,
|
||||
torch::Tensor const& output,
|
||||
torch::Tensor const& output_sf,
|
||||
int tensor_layout) {
|
||||
constexpr int BLOCK_SIZE = flash::BLOCK_M;
|
||||
|
||||
CHECK_CUDA(input);
|
||||
CHECK_CUDA(output);
|
||||
CHECK_CUDA(output_sf);
|
||||
|
||||
CHECK_LASTDIM_CONTIGUOUS(input);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output);
|
||||
CHECK_LASTDIM_CONTIGUOUS(output_sf);
|
||||
|
||||
CHECK_DTYPE(output, at::ScalarType::Byte);
|
||||
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
|
||||
|
||||
CHECK_DIMS(input, 4);
|
||||
CHECK_DIMS(output, 4);
|
||||
CHECK_DIMS(output_sf, 4);
|
||||
|
||||
const int batch_size = input.size(0);
|
||||
const int head_dim = input.size(3);
|
||||
|
||||
const int stride_bz_input = input.stride(0);
|
||||
const int stride_bz_output = output.stride(0);
|
||||
const int stride_bz_output_sf = output_sf.stride(0);
|
||||
|
||||
int num_tokens, num_heads;
|
||||
int stride_seq_input;
|
||||
int stride_d_output, stride_d_output_sf;
|
||||
int stride_h_input, stride_h_output, stride_h_output_sf;
|
||||
if (tensor_layout == 0) {
|
||||
num_tokens = input.size(1);
|
||||
num_heads = input.size(2);
|
||||
stride_seq_input = input.stride(1);
|
||||
stride_d_output = output.stride(1);
|
||||
stride_d_output_sf = output_sf.stride(1);
|
||||
stride_h_input = input.stride(2);
|
||||
stride_h_output = output.stride(2);
|
||||
stride_h_output_sf = output_sf.stride(2);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
|
||||
} else {
|
||||
num_tokens = input.size(2);
|
||||
num_heads = input.size(1);
|
||||
stride_seq_input = input.stride(2);
|
||||
stride_d_output = output.stride(2);
|
||||
stride_d_output_sf = output_sf.stride(2);
|
||||
stride_h_input = input.stride(1);
|
||||
stride_h_output = output.stride(1);
|
||||
stride_h_output_sf = output_sf.stride(1);
|
||||
|
||||
CHECK_SHAPE(output, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
|
||||
CHECK_SHAPE(output_sf, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
|
||||
}
|
||||
|
||||
auto input_dtype = input.scalar_type();
|
||||
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
|
||||
|
||||
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
|
||||
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
|
||||
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
|
||||
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
|
||||
|
||||
scaled_fp4_quant_trans_kernel<HEAD_DIM, BLOCK_SIZE, c_type>
|
||||
<<<grid, block, 0, stream>>>(
|
||||
reinterpret_cast<c_type*>(input.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output.data_ptr()),
|
||||
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
|
||||
batch_size, num_heads, num_tokens,
|
||||
stride_bz_input, stride_h_input, stride_seq_input,
|
||||
stride_bz_output, stride_h_output, stride_d_output,
|
||||
stride_bz_output_sf, stride_h_output_sf, stride_d_output_sf);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
m.def("scaled_fp4_quant", &scaled_fp4_quant);
|
||||
m.def("scaled_fp4_quant_permute", &scaled_fp4_quant_permute);
|
||||
m.def("scaled_fp4_quant_trans", &scaled_fp4_quant_trans);
|
||||
}
|
||||
@@ -1,14 +0,0 @@
|
||||
"""Make local benchmark scripts runnable from common repo entrypoints."""
|
||||
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
|
||||
BENCHMARKS_DIR = Path(__file__).resolve().parent
|
||||
KERNEL_ROOT = BENCHMARKS_DIR.parent
|
||||
REPO_ROOT = KERNEL_ROOT.parent
|
||||
|
||||
for path in (KERNEL_ROOT, REPO_ROOT):
|
||||
path_str = str(path)
|
||||
if path_str not in sys.path:
|
||||
sys.path.insert(0, path_str)
|
||||
@@ -1,287 +0,0 @@
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import torch
|
||||
|
||||
from attn_qat_infer.api import (
|
||||
blockscaled_fp4_attn,
|
||||
preprocess_qkv,
|
||||
scale_and_quant_fp4,
|
||||
scale_and_quant_fp4_permute,
|
||||
scale_and_quant_fp4_transpose,
|
||||
)
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def benchmark_blockscaled_fp4_attn(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
per_block_mean=True, single_level_p_quant=False,
|
||||
num_warmups=100, num_tests=1000):
|
||||
"""
|
||||
Benchmark blockscaled_fp4_attn function (excluding quantization overhead).
|
||||
|
||||
This benchmarks ONLY the core FP4 attention kernel, with pre-quantized inputs.
|
||||
The quantization step is performed once before benchmarking.
|
||||
|
||||
Args:
|
||||
batch_size: Batch size
|
||||
num_heads: Number of attention heads
|
||||
seq_len: Sequence length (same for Q, K, V)
|
||||
head_dim: Head dimension
|
||||
is_causal: Whether to use causal masking
|
||||
dtype: Data type (torch.bfloat16 or torch.float16)
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization for P matrix
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
|
||||
Returns:
|
||||
dict with performance metrics
|
||||
"""
|
||||
device = 'cuda'
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
# Create input tensors
|
||||
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
|
||||
# Pre-process and quantize inputs (done once, not included in benchmark)
|
||||
is_bf16 = dtype == torch.bfloat16
|
||||
KL = k.size(2)
|
||||
q_processed, k_processed, v_processed, delta_s = preprocess_qkv(q, k, v, per_block_mean)
|
||||
qlist = scale_and_quant_fp4(q_processed)
|
||||
klist = scale_and_quant_fp4_permute(k_processed)
|
||||
vlist = scale_and_quant_fp4_transpose(v_processed)
|
||||
|
||||
# Synchronize to ensure quantization is complete
|
||||
torch.cuda.synchronize()
|
||||
|
||||
# Create closure for benchmarking (only the attention kernel)
|
||||
def run_attention():
|
||||
return blockscaled_fp4_attn(
|
||||
qlist, klist, vlist,
|
||||
delta_s,
|
||||
KL,
|
||||
is_causal=is_causal,
|
||||
per_block_mean=per_block_mean,
|
||||
is_bf16=is_bf16,
|
||||
single_level_p_quant=single_level_p_quant
|
||||
)
|
||||
|
||||
# Benchmark using the bench utility (handles warmup and timing)
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
avg_time_s = avg_time_ms / 1000.0
|
||||
|
||||
# Calculate FLOPs (use processed sequence length after padding)
|
||||
processed_seq_len = q_processed.size(2)
|
||||
total_flops = calculate_attention_flops(
|
||||
batch_size, num_heads, processed_seq_len, processed_seq_len, head_dim, is_causal
|
||||
)
|
||||
|
||||
# Calculate TFLOPs
|
||||
tflops = total_flops / (avg_time_s * 1e12)
|
||||
|
||||
# Calculate throughput (tokens/sec) - use original seq_len for meaningful metric
|
||||
tokens_per_second = (batch_size * seq_len) / avg_time_s
|
||||
|
||||
return {
|
||||
'batch_size': batch_size,
|
||||
'num_heads': num_heads,
|
||||
'seq_len': seq_len,
|
||||
'processed_seq_len': processed_seq_len,
|
||||
'head_dim': head_dim,
|
||||
'is_causal': is_causal,
|
||||
'dtype': str(dtype),
|
||||
'per_block_mean': per_block_mean,
|
||||
'single_level_p_quant': single_level_p_quant,
|
||||
'avg_time_ms': avg_time_ms,
|
||||
'avg_time_s': avg_time_s,
|
||||
'total_flops': total_flops,
|
||||
'tflops': tflops,
|
||||
'tokens_per_second': tokens_per_second,
|
||||
}
|
||||
|
||||
|
||||
def print_results(results):
|
||||
"""Print benchmark results in a formatted table."""
|
||||
print("\n" + "="*100)
|
||||
print("blockscaled_fp4_attn Benchmark Results (Kernel Only, No Quantization)")
|
||||
print("="*100)
|
||||
print(f"Configuration:")
|
||||
print(f" Batch Size: {results['batch_size']}")
|
||||
print(f" Num Heads: {results['num_heads']}")
|
||||
print(f" Sequence Length: {results['seq_len']}")
|
||||
print(f" Processed Seq Length: {results['processed_seq_len']} (after padding)")
|
||||
print(f" Head Dimension: {results['head_dim']}")
|
||||
print(f" Causal: {results['is_causal']}")
|
||||
print(f" Data Type: {results['dtype']}")
|
||||
print(f" Per Block Mean: {results['per_block_mean']}")
|
||||
print(f" Single Level P Quant: {results['single_level_p_quant']}")
|
||||
print(f"\nPerformance:")
|
||||
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
|
||||
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
|
||||
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
|
||||
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
|
||||
print("="*100 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def run_benchmark_suite():
|
||||
"""Run a comprehensive benchmark suite with various configurations."""
|
||||
print("Starting blockscaled_fp4_attn Benchmark Suite (Kernel Only)...")
|
||||
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}\n")
|
||||
print("Note: This benchmark measures only the FP4 attention kernel,")
|
||||
print(" excluding quantization overhead.\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
# Default configurations to test
|
||||
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
|
||||
configs = [
|
||||
(1, 16, 512, 64, False, torch.bfloat16),
|
||||
(1, 16, 1024, 64, False, torch.bfloat16),
|
||||
(1, 16, 2048, 64, False, torch.bfloat16),
|
||||
(1, 16, 4096, 64, False, torch.bfloat16),
|
||||
(1, 16, 8192, 64, False, torch.bfloat16),
|
||||
(1, 16, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 16, 512, 128, False, torch.bfloat16),
|
||||
(1, 16, 1024, 128, False, torch.bfloat16),
|
||||
(1, 16, 2048, 128, False, torch.bfloat16),
|
||||
(1, 16, 4096, 128, False, torch.bfloat16),
|
||||
(1, 16, 8192, 128, False, torch.bfloat16),
|
||||
(1, 16, 16384, 128, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 512, 64, False, torch.bfloat16),
|
||||
(1, 32, 1024, 64, False, torch.bfloat16),
|
||||
(1, 32, 2048, 64, False, torch.bfloat16),
|
||||
(1, 32, 4096, 64, False, torch.bfloat16),
|
||||
(1, 32, 8192, 64, False, torch.bfloat16),
|
||||
(1, 32, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 512, 128, False, torch.bfloat16),
|
||||
(1, 32, 1024, 128, False, torch.bfloat16),
|
||||
(1, 32, 2048, 128, False, torch.bfloat16),
|
||||
(1, 32, 4096, 128, False, torch.bfloat16),
|
||||
(1, 32, 8192, 128, False, torch.bfloat16),
|
||||
(1, 32, 16384, 128, False, torch.bfloat16),
|
||||
]
|
||||
|
||||
all_results = []
|
||||
|
||||
for config in configs:
|
||||
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
|
||||
|
||||
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
|
||||
f"Causal={is_causal}, dtype={dtype}...")
|
||||
sys.stdout.flush()
|
||||
|
||||
try:
|
||||
results = benchmark_blockscaled_fp4_attn(
|
||||
batch_size=batch_size,
|
||||
num_heads=num_heads,
|
||||
seq_len=seq_len,
|
||||
head_dim=head_dim,
|
||||
is_causal=is_causal,
|
||||
dtype=dtype,
|
||||
num_warmups=10,
|
||||
num_tests=50
|
||||
)
|
||||
|
||||
print_results(results)
|
||||
all_results.append(results)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error benchmarking configuration {config}:")
|
||||
print(f" Exception: {e}")
|
||||
traceback.print_exc()
|
||||
sys.stdout.flush()
|
||||
continue
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*120)
|
||||
print("Summary Table (blockscaled_fp4_attn Kernel Only)")
|
||||
print("="*120)
|
||||
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
|
||||
print("-"*120)
|
||||
|
||||
for r in all_results:
|
||||
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
|
||||
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
|
||||
f"{r['tokens_per_second']:<15,.0f}")
|
||||
|
||||
print("="*120 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description='Benchmark blockscaled_fp4_attn kernel (excluding quantization)')
|
||||
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
|
||||
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
|
||||
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
|
||||
help='Data type')
|
||||
parser.add_argument('--per-block-mean', action='store_true', default=True,
|
||||
help='Use per-block mean for Q smoothing (default: True)')
|
||||
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
|
||||
help='Disable per-block mean for Q smoothing')
|
||||
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
|
||||
help='Use single-level P quantization (default: False)')
|
||||
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
|
||||
help='Use two-level P quantization')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
dtype_map = {
|
||||
'bfloat16': torch.bfloat16,
|
||||
'float16': torch.float16
|
||||
}
|
||||
|
||||
if args.suite:
|
||||
run_benchmark_suite()
|
||||
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
|
||||
results = benchmark_blockscaled_fp4_attn(
|
||||
batch_size=args.batch_size,
|
||||
num_heads=args.num_heads,
|
||||
seq_len=args.seq_len,
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
dtype=dtype_map[args.dtype],
|
||||
per_block_mean=args.per_block_mean,
|
||||
single_level_p_quant=args.single_level_p_quant,
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests
|
||||
)
|
||||
print_results(results)
|
||||
else:
|
||||
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
|
||||
print(
|
||||
"Example: python benchmarks/benchmark_blockscaled_fp4_attn.py "
|
||||
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
|
||||
)
|
||||
print("\nNote: This benchmark measures only the FP4 attention kernel,")
|
||||
print(" excluding quantization overhead.")
|
||||
sys.stdout.flush()
|
||||
run_benchmark_suite()
|
||||
@@ -1,380 +0,0 @@
|
||||
import argparse
|
||||
import sys
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import matplotlib
|
||||
matplotlib.use('Agg') # Use non-interactive backend for server environments
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from flash_attn import flash_attn_func
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
# Import SageAttn components for direct control
|
||||
from attn_qat_infer.api import (
|
||||
preprocess_qkv,
|
||||
scale_and_quant_fp4,
|
||||
scale_and_quant_fp4_permute,
|
||||
scale_and_quant_fp4_transpose,
|
||||
blockscaled_fp4_attn
|
||||
)
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def sageattn_blackwell_configurable(q, k, v, is_causal=False, per_block_mean=True,
|
||||
single_level_p_quant=True,
|
||||
enable_smoothing_q=False, enable_smoothing_k=False):
|
||||
"""
|
||||
Configurable SageAttention3 Blackwell kernel with explicit smoothing control.
|
||||
|
||||
Args:
|
||||
q: Query tensor [B, H, L, D]
|
||||
k: Key tensor [B, H, L, D]
|
||||
v: Value tensor [B, H, L, D]
|
||||
is_causal: Whether to use causal masking
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization for P matrix
|
||||
enable_smoothing_q: Enable Q smoothing
|
||||
enable_smoothing_k: Enable K smoothing
|
||||
|
||||
Returns:
|
||||
Output tensor [B, H, L, D]
|
||||
"""
|
||||
QL = q.size(2)
|
||||
KL = k.size(2)
|
||||
is_bf16 = q.dtype == torch.bfloat16
|
||||
|
||||
# Preprocess with explicit smoothing control
|
||||
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean, enable_smoothing_q, enable_smoothing_k)
|
||||
|
||||
qlist_from_cuda = scale_and_quant_fp4(q)
|
||||
klist_from_cuda = scale_and_quant_fp4_permute(k)
|
||||
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
|
||||
|
||||
o_fp4 = blockscaled_fp4_attn(
|
||||
qlist_from_cuda,
|
||||
klist_from_cuda,
|
||||
vlist_from_cuda,
|
||||
delta_s,
|
||||
KL,
|
||||
is_causal,
|
||||
per_block_mean,
|
||||
is_bf16,
|
||||
single_level_p_quant
|
||||
)[0][:, :, :QL, :].contiguous()
|
||||
|
||||
return o_fp4
|
||||
|
||||
|
||||
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
num_warmups=10, num_tests=50):
|
||||
"""Benchmark FlashAttention2."""
|
||||
device = 'cuda'
|
||||
|
||||
# FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
|
||||
q = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
|
||||
|
||||
def run_attention():
|
||||
return flash_attn_func(q, k, v, causal=is_causal)
|
||||
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
return avg_time_ms
|
||||
|
||||
|
||||
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
per_block_mean=True, single_level_p_quant=False,
|
||||
enable_smoothing_q=True, enable_smoothing_k=True,
|
||||
num_warmups=10, num_tests=50):
|
||||
"""Benchmark SageAttention3 with configurable smoothing."""
|
||||
device = 'cuda'
|
||||
|
||||
# SageAttn expects (batch, num_heads, seq_len, head_dim)
|
||||
q = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
|
||||
|
||||
def run_attention():
|
||||
return sageattn_blackwell_configurable(
|
||||
q, k, v,
|
||||
is_causal=is_causal,
|
||||
per_block_mean=per_block_mean,
|
||||
single_level_p_quant=single_level_p_quant,
|
||||
enable_smoothing_q=enable_smoothing_q,
|
||||
enable_smoothing_k=enable_smoothing_k
|
||||
)
|
||||
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
return avg_time_ms
|
||||
|
||||
|
||||
def time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal=False):
|
||||
"""Convert time to TFLOPs (Tera FLOPs per Second)."""
|
||||
total_flops = calculate_attention_flops(batch_size, num_heads, seq_len, seq_len, head_dim, is_causal)
|
||||
time_s = time_ms / 1000.0
|
||||
tflops = total_flops / (time_s * 1e12)
|
||||
return tflops
|
||||
|
||||
|
||||
def run_benchmark_suite(head_dim=64, is_causal=False, num_heads=12, batch_size=1,
|
||||
num_warmups=10, num_tests=50,
|
||||
seq_lens=None, output_file="benchmark_attention.png"):
|
||||
"""
|
||||
Run comprehensive benchmark suite and generate plot.
|
||||
|
||||
Args:
|
||||
head_dim: Head dimension (64 or 128)
|
||||
is_causal: Whether to use causal attention
|
||||
num_heads: Number of attention heads
|
||||
batch_size: Batch size
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
seq_lens: List of sequence lengths to test
|
||||
output_file: Output plot filename
|
||||
"""
|
||||
if seq_lens is None:
|
||||
seq_lens = [1024, 2048, 4096, 8192, 16384, 32768]
|
||||
|
||||
device_name = torch.cuda.get_device_name(0)
|
||||
# Extract short name (e.g., "RTX5090" from full name)
|
||||
short_name = device_name.split()[-1] if 'RTX' in device_name or 'A100' in device_name else device_name[:20]
|
||||
|
||||
print(f"Starting Combined Attention Benchmark Suite...")
|
||||
print(f"CUDA Device: {device_name}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}")
|
||||
print(f"Head Dim: {head_dim}, Causal: {is_causal}, Num Heads: {num_heads}, Batch Size: {batch_size}")
|
||||
print("="*80)
|
||||
sys.stdout.flush()
|
||||
|
||||
# Results storage: {method_name: {seq_len: tflops}}
|
||||
results: Dict[str, Dict[int, Optional[float]]] = {
|
||||
'FlashAttn': {},
|
||||
'SageAttn3': {},
|
||||
'FP4': {},
|
||||
}
|
||||
|
||||
dtype = torch.bfloat16
|
||||
|
||||
for seq_len in seq_lens:
|
||||
print(f"\n--- Sequence Length: {seq_len} ---")
|
||||
sys.stdout.flush()
|
||||
|
||||
# FlashAttention2
|
||||
print(f" Benchmarking FlashAttn2...", end=" ")
|
||||
sys.stdout.flush()
|
||||
try:
|
||||
time_ms = benchmark_flashattn2(
|
||||
batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=is_causal, dtype=dtype,
|
||||
num_warmups=num_warmups, num_tests=num_tests
|
||||
)
|
||||
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
|
||||
results['FlashAttn'][seq_len] = tflops
|
||||
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
|
||||
except Exception as e:
|
||||
print(f"OOM or Error: {e}")
|
||||
results['FlashAttn'][seq_len] = None
|
||||
sys.stdout.flush()
|
||||
|
||||
# SageAttn3 (with smoothing: single_level_p_quant=False, enable_smoothing_q=True, enable_smoothing_k=True)
|
||||
print(f" Benchmarking SageAttn3 (smoothing ON)...", end=" ")
|
||||
sys.stdout.flush()
|
||||
try:
|
||||
time_ms = benchmark_sageattn3(
|
||||
batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=is_causal, dtype=dtype,
|
||||
per_block_mean=True,
|
||||
single_level_p_quant=False,
|
||||
enable_smoothing_q=True,
|
||||
enable_smoothing_k=True,
|
||||
num_warmups=num_warmups, num_tests=num_tests
|
||||
)
|
||||
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
|
||||
results['SageAttn3'][seq_len] = tflops
|
||||
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
|
||||
except Exception as e:
|
||||
print(f"OOM or Error: {e}")
|
||||
results['SageAttn3'][seq_len] = None
|
||||
sys.stdout.flush()
|
||||
|
||||
# FP4 (no smoothing: single_level_p_quant=True, enable_smoothing_q=False, enable_smoothing_k=False)
|
||||
print(f" Benchmarking FP4 (smoothing OFF)...", end=" ")
|
||||
sys.stdout.flush()
|
||||
try:
|
||||
time_ms = benchmark_sageattn3(
|
||||
batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=is_causal, dtype=dtype,
|
||||
per_block_mean=True,
|
||||
single_level_p_quant=True,
|
||||
enable_smoothing_q=False,
|
||||
enable_smoothing_k=False,
|
||||
num_warmups=num_warmups, num_tests=num_tests
|
||||
)
|
||||
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
|
||||
results['FP4'][seq_len] = tflops
|
||||
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
|
||||
except Exception as e:
|
||||
print(f"OOM or Error: {e}")
|
||||
results['FP4'][seq_len] = None
|
||||
sys.stdout.flush()
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*100)
|
||||
print("Summary Table (TFLOPs)")
|
||||
print("="*100)
|
||||
header = f"{'SeqLen':<10}"
|
||||
for method in results.keys():
|
||||
header += f"{method:<15}"
|
||||
print(header)
|
||||
print("-"*100)
|
||||
|
||||
for seq_len in seq_lens:
|
||||
row = f"{seq_len:<10}"
|
||||
for method in results.keys():
|
||||
val = results[method].get(seq_len)
|
||||
if val is not None:
|
||||
row += f"{val:<15.0f}"
|
||||
else:
|
||||
row += f"{'OOM':<15}"
|
||||
print(row)
|
||||
print("="*100)
|
||||
sys.stdout.flush()
|
||||
|
||||
# Generate plot
|
||||
generate_plot(results, seq_lens, head_dim, is_causal, short_name, output_file)
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def generate_plot(results: Dict[str, Dict[int, Optional[float]]],
|
||||
seq_lens: List[int],
|
||||
head_dim: int,
|
||||
is_causal: bool,
|
||||
device_name: str,
|
||||
output_file: str):
|
||||
"""Generate bar plot comparing attention implementations."""
|
||||
|
||||
# Prepare data
|
||||
methods = list(results.keys())
|
||||
x_labels = [f"{sl//1024}K" for sl in seq_lens]
|
||||
|
||||
# Colors for each method (red, blue, green scheme)
|
||||
colors = {
|
||||
'FlashAttn': '#1E90FF', # Blue (Dodger Blue)
|
||||
'SageAttn3': '#228B22', # Green (Forest Green)
|
||||
'FP4': '#DC143C', # Red (Crimson)
|
||||
}
|
||||
|
||||
# Number of methods and positions
|
||||
n_methods = len(methods)
|
||||
n_positions = len(seq_lens)
|
||||
|
||||
# Bar width and positions
|
||||
bar_width = 0.25
|
||||
x = np.arange(n_positions)
|
||||
|
||||
# Create figure
|
||||
fig, ax = plt.subplots(figsize=(12, 6))
|
||||
|
||||
# Plot bars for each method
|
||||
for i, method in enumerate(methods):
|
||||
values = []
|
||||
for seq_len in seq_lens:
|
||||
val = results[method].get(seq_len)
|
||||
values.append(val if val is not None else 0)
|
||||
|
||||
offset = (i - n_methods/2 + 0.5) * bar_width
|
||||
bars = ax.bar(x + offset, values, bar_width,
|
||||
label=method, color=colors.get(method, f'C{i}'),
|
||||
edgecolor='black', linewidth=0.5)
|
||||
|
||||
# Add value labels on top of bars
|
||||
for bar, val, seq_len in zip(bars, values, seq_lens):
|
||||
if results[method].get(seq_len) is None:
|
||||
label = 'OOM'
|
||||
else:
|
||||
label = f'{int(val)}'
|
||||
|
||||
height = bar.get_height()
|
||||
ax.annotate(label,
|
||||
xy=(bar.get_x() + bar.get_width() / 2, height),
|
||||
xytext=(0, 3), # 3 points vertical offset
|
||||
textcoords="offset points",
|
||||
ha='center', va='bottom',
|
||||
fontsize=8, rotation=0)
|
||||
|
||||
# Customize plot
|
||||
ax.set_xlabel('Sequence Length', fontsize=12, fontweight='bold')
|
||||
ax.set_ylabel('Speed (TFLOPs)', fontsize=12, fontweight='bold')
|
||||
ax.set_title(f'{device_name}, (Head dim = {head_dim}, causal = {is_causal})', fontsize=14, fontweight='bold')
|
||||
ax.set_xticks(x)
|
||||
ax.set_xticklabels(x_labels)
|
||||
legend = ax.legend(loc='upper left', ncol=len(methods), fontsize=10)
|
||||
# Make legend text bold
|
||||
for text in legend.get_texts():
|
||||
text.set_fontweight('bold')
|
||||
|
||||
# Set y-axis to start from 0
|
||||
ax.set_ylim(bottom=0)
|
||||
|
||||
# Add grid for readability
|
||||
ax.yaxis.grid(True, linestyle='--', alpha=0.7)
|
||||
ax.set_axisbelow(True)
|
||||
|
||||
# Tight layout
|
||||
plt.tight_layout()
|
||||
|
||||
# Save plot
|
||||
plt.savefig(output_file, dpi=150, bbox_inches='tight')
|
||||
print(f"\nPlot saved to: {output_file}")
|
||||
|
||||
# Also save as PDF for high quality
|
||||
pdf_file = output_file.rsplit('.', 1)[0] + '.pdf'
|
||||
plt.savefig(pdf_file, bbox_inches='tight')
|
||||
print(f"PDF saved to: {pdf_file}")
|
||||
|
||||
plt.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description='Combined Attention Benchmark (FlashAttn2 vs SageAttn3 vs FP4)')
|
||||
parser.add_argument('--batch-size', type=int, default=1, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=16, help='Number of attention heads')
|
||||
parser.add_argument('--head-dim', type=int, default=64, choices=[64, 128], help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--seq-lens', type=int, nargs='+',
|
||||
default=[1024, 2048, 4096, 8192, 16384, 32768],
|
||||
help='Sequence lengths to benchmark')
|
||||
parser.add_argument('--output', type=str, default='benchmark_attention.png',
|
||||
help='Output plot filename')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
run_benchmark_suite(
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
num_heads=args.num_heads,
|
||||
batch_size=args.batch_size,
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests,
|
||||
seq_lens=args.seq_lens,
|
||||
output_file=args.output
|
||||
)
|
||||
@@ -1,234 +0,0 @@
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import torch
|
||||
|
||||
from flash_attn import flash_attn_func
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
num_warmups=100, num_tests=1000):
|
||||
"""
|
||||
Benchmark FlashAttention2 and return performance metrics.
|
||||
|
||||
Args:
|
||||
batch_size: Batch size
|
||||
num_heads: Number of attention heads
|
||||
seq_len: Sequence length (same for Q, K, V)
|
||||
head_dim: Head dimension
|
||||
is_causal: Whether to use causal masking
|
||||
dtype: Data type (torch.bfloat16 or torch.float16)
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
|
||||
Returns:
|
||||
dict with performance metrics
|
||||
"""
|
||||
device = 'cuda'
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
# Create input tensors - FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
|
||||
q = torch.randn(batch_size, seq_len, num_heads, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, seq_len, num_heads, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, seq_len, num_heads, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
|
||||
# Create closure for benchmarking
|
||||
def run_attention():
|
||||
return flash_attn_func(
|
||||
q, k, v,
|
||||
causal=is_causal,
|
||||
)
|
||||
|
||||
# Benchmark using the bench utility (handles warmup and timing)
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
avg_time_s = avg_time_ms / 1000.0
|
||||
|
||||
# Calculate FLOPs
|
||||
total_flops = calculate_attention_flops(
|
||||
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
|
||||
)
|
||||
|
||||
# Calculate TFLOPs
|
||||
tflops = total_flops / (avg_time_s * 1e12)
|
||||
|
||||
# Calculate throughput (tokens/sec)
|
||||
tokens_per_second = (batch_size * seq_len) / avg_time_s
|
||||
|
||||
return {
|
||||
'batch_size': batch_size,
|
||||
'num_heads': num_heads,
|
||||
'seq_len': seq_len,
|
||||
'head_dim': head_dim,
|
||||
'is_causal': is_causal,
|
||||
'dtype': str(dtype),
|
||||
'avg_time_ms': avg_time_ms,
|
||||
'avg_time_s': avg_time_s,
|
||||
'total_flops': total_flops,
|
||||
'tflops': tflops,
|
||||
'tokens_per_second': tokens_per_second,
|
||||
}
|
||||
|
||||
|
||||
def print_results(results):
|
||||
"""Print benchmark results in a formatted table."""
|
||||
print("\n" + "="*100)
|
||||
print("FlashAttention2 Benchmark Results")
|
||||
print("="*100)
|
||||
print(f"Configuration:")
|
||||
print(f" Batch Size: {results['batch_size']}")
|
||||
print(f" Num Heads: {results['num_heads']}")
|
||||
print(f" Sequence Length: {results['seq_len']}")
|
||||
print(f" Head Dimension: {results['head_dim']}")
|
||||
print(f" Causal: {results['is_causal']}")
|
||||
print(f" Data Type: {results['dtype']}")
|
||||
print(f"\nPerformance:")
|
||||
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
|
||||
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
|
||||
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
|
||||
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
|
||||
print("="*100 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def run_benchmark_suite():
|
||||
"""Run a comprehensive benchmark suite with various configurations."""
|
||||
print("Starting FlashAttention2 Benchmark Suite...")
|
||||
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
# Default configurations to test
|
||||
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
|
||||
configs = [
|
||||
(1, 16, 1024, 64, False, torch.bfloat16),
|
||||
(1, 16, 2048, 64, False, torch.bfloat16),
|
||||
(1, 16, 4096, 64, False, torch.bfloat16),
|
||||
(1, 16, 8192, 64, False, torch.bfloat16),
|
||||
(1, 16, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 16, 1024, 128, False, torch.bfloat16),
|
||||
(1, 16, 2048, 128, False, torch.bfloat16),
|
||||
(1, 16, 4096, 128, False, torch.bfloat16),
|
||||
(1, 16, 8192, 128, False, torch.bfloat16),
|
||||
(1, 16, 16384, 128, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 64, False, torch.bfloat16),
|
||||
(1, 32, 2048, 64, False, torch.bfloat16),
|
||||
(1, 32, 4096, 64, False, torch.bfloat16),
|
||||
(1, 32, 8192, 64, False, torch.bfloat16),
|
||||
(1, 32, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 128, False, torch.bfloat16),
|
||||
(1, 32, 2048, 128, False, torch.bfloat16),
|
||||
(1, 32, 4096, 128, False, torch.bfloat16),
|
||||
(1, 32, 8192, 128, False, torch.bfloat16),
|
||||
(1, 32, 16384, 128, False, torch.bfloat16),
|
||||
]
|
||||
|
||||
all_results = []
|
||||
|
||||
for config in configs:
|
||||
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
|
||||
|
||||
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
|
||||
f"Causal={is_causal}, dtype={dtype}...")
|
||||
sys.stdout.flush()
|
||||
|
||||
try:
|
||||
results = benchmark_flashattn2(
|
||||
batch_size=batch_size,
|
||||
num_heads=num_heads,
|
||||
seq_len=seq_len,
|
||||
head_dim=head_dim,
|
||||
is_causal=is_causal,
|
||||
dtype=dtype,
|
||||
num_warmups=10,
|
||||
num_tests=50
|
||||
)
|
||||
|
||||
print_results(results)
|
||||
all_results.append(results)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error benchmarking configuration {config}:")
|
||||
print(f" Exception: {e}")
|
||||
traceback.print_exc()
|
||||
sys.stdout.flush()
|
||||
continue
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*120)
|
||||
print("Summary Table")
|
||||
print("="*120)
|
||||
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
|
||||
print("-"*120)
|
||||
|
||||
for r in all_results:
|
||||
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
|
||||
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
|
||||
f"{r['tokens_per_second']:<15,.0f}")
|
||||
|
||||
print("="*120 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description='Benchmark FlashAttention2 in TFLOPs')
|
||||
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
|
||||
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
|
||||
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
|
||||
help='Data type')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
dtype_map = {
|
||||
'bfloat16': torch.bfloat16,
|
||||
'float16': torch.float16
|
||||
}
|
||||
|
||||
if args.suite:
|
||||
run_benchmark_suite()
|
||||
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
|
||||
results = benchmark_flashattn2(
|
||||
batch_size=args.batch_size,
|
||||
num_heads=args.num_heads,
|
||||
seq_len=args.seq_len,
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
dtype=dtype_map[args.dtype],
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests
|
||||
)
|
||||
print_results(results)
|
||||
else:
|
||||
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
|
||||
print(
|
||||
"Example: python benchmarks/benchmark_flashattn2.py "
|
||||
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
|
||||
)
|
||||
sys.stdout.flush()
|
||||
run_benchmark_suite()
|
||||
@@ -1,253 +0,0 @@
|
||||
import sys
|
||||
import traceback
|
||||
|
||||
import _bootstrap # noqa: F401
|
||||
import torch
|
||||
|
||||
from attn_qat_infer.api import sageattn_blackwell
|
||||
from attn_qat_infer.quantization.bench.bench_utils import bench
|
||||
|
||||
|
||||
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
|
||||
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
|
||||
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
|
||||
if is_causal:
|
||||
f = f // 2
|
||||
return f
|
||||
|
||||
|
||||
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
|
||||
is_causal=False, dtype=torch.bfloat16,
|
||||
per_block_mean=True, single_level_p_quant=False,
|
||||
num_warmups=100, num_tests=1000):
|
||||
"""
|
||||
Benchmark SageAttention3 and return performance metrics.
|
||||
|
||||
Args:
|
||||
batch_size: Batch size
|
||||
num_heads: Number of attention heads
|
||||
seq_len: Sequence length (same for Q, K, V)
|
||||
head_dim: Head dimension
|
||||
is_causal: Whether to use causal masking
|
||||
dtype: Data type (torch.bfloat16 or torch.float16)
|
||||
per_block_mean: Whether to use per-block mean for Q smoothing
|
||||
single_level_p_quant: If True, use single-level quantization for P matrix
|
||||
num_warmups: Number of warmup iterations
|
||||
num_tests: Number of test iterations
|
||||
|
||||
Returns:
|
||||
dict with performance metrics
|
||||
"""
|
||||
device = 'cuda'
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
|
||||
|
||||
# Create input tensors
|
||||
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
|
||||
device=device, dtype=dtype)
|
||||
|
||||
# Create closure for benchmarking (no extra stream needed - bench handles synchronization)
|
||||
def run_attention():
|
||||
return sageattn_blackwell(
|
||||
q, k, v,
|
||||
is_causal=is_causal,
|
||||
per_block_mean=per_block_mean,
|
||||
single_level_p_quant=single_level_p_quant
|
||||
)
|
||||
|
||||
# Benchmark using the bench utility (handles warmup and timing)
|
||||
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
|
||||
avg_time_s = avg_time_ms / 1000.0
|
||||
|
||||
# Calculate FLOPs
|
||||
total_flops = calculate_attention_flops(
|
||||
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
|
||||
)
|
||||
|
||||
# Calculate TFLOPs
|
||||
tflops = total_flops / (avg_time_s * 1e12)
|
||||
|
||||
# Calculate throughput (tokens/sec)
|
||||
tokens_per_second = (batch_size * seq_len) / avg_time_s
|
||||
|
||||
return {
|
||||
'batch_size': batch_size,
|
||||
'num_heads': num_heads,
|
||||
'seq_len': seq_len,
|
||||
'head_dim': head_dim,
|
||||
'is_causal': is_causal,
|
||||
'dtype': str(dtype),
|
||||
'per_block_mean': per_block_mean,
|
||||
'single_level_p_quant': single_level_p_quant,
|
||||
'avg_time_ms': avg_time_ms,
|
||||
'avg_time_s': avg_time_s,
|
||||
'total_flops': total_flops,
|
||||
'tflops': tflops,
|
||||
'tokens_per_second': tokens_per_second,
|
||||
}
|
||||
|
||||
|
||||
def print_results(results):
|
||||
"""Print benchmark results in a formatted table."""
|
||||
print("\n" + "="*100)
|
||||
print("SageAttention3 Benchmark Results")
|
||||
print("="*100)
|
||||
print(f"Configuration:")
|
||||
print(f" Batch Size: {results['batch_size']}")
|
||||
print(f" Num Heads: {results['num_heads']}")
|
||||
print(f" Sequence Length: {results['seq_len']}")
|
||||
print(f" Head Dimension: {results['head_dim']}")
|
||||
print(f" Causal: {results['is_causal']}")
|
||||
print(f" Data Type: {results['dtype']}")
|
||||
print(f" Per Block Mean: {results['per_block_mean']}")
|
||||
print(f" Single Level P Quant: {results['single_level_p_quant']}")
|
||||
print(f"\nPerformance:")
|
||||
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
|
||||
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
|
||||
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
|
||||
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
|
||||
print("="*100 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def run_benchmark_suite():
|
||||
"""Run a comprehensive benchmark suite with various configurations."""
|
||||
print("Starting SageAttention3 Benchmark Suite...")
|
||||
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
|
||||
print(f"CUDA Version: {torch.version.cuda}")
|
||||
print(f"PyTorch Version: {torch.__version__}\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
# Default configurations to test
|
||||
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
|
||||
configs = [
|
||||
(1, 16, 1024, 64, False, torch.bfloat16),
|
||||
(1, 16, 2048, 64, False, torch.bfloat16),
|
||||
(1, 16, 4096, 64, False, torch.bfloat16),
|
||||
(1, 16, 8192, 64, False, torch.bfloat16),
|
||||
(1, 16, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 16, 1024, 128, False, torch.bfloat16),
|
||||
(1, 16, 2048, 128, False, torch.bfloat16),
|
||||
(1, 16, 4096, 128, False, torch.bfloat16),
|
||||
(1, 16, 8192, 128, False, torch.bfloat16),
|
||||
(1, 16, 16384, 128, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 64, False, torch.bfloat16),
|
||||
(1, 32, 2048, 64, False, torch.bfloat16),
|
||||
(1, 32, 4096, 64, False, torch.bfloat16),
|
||||
(1, 32, 8192, 64, False, torch.bfloat16),
|
||||
(1, 32, 16384, 64, False, torch.bfloat16),
|
||||
|
||||
(1, 32, 1024, 128, False, torch.bfloat16),
|
||||
(1, 32, 2048, 128, False, torch.bfloat16),
|
||||
(1, 32, 4096, 128, False, torch.bfloat16),
|
||||
(1, 32, 8192, 128, False, torch.bfloat16),
|
||||
(1, 32, 16384, 128, False, torch.bfloat16),
|
||||
]
|
||||
|
||||
all_results = []
|
||||
|
||||
for config in configs:
|
||||
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
|
||||
|
||||
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
|
||||
f"Causal={is_causal}, dtype={dtype}...")
|
||||
sys.stdout.flush()
|
||||
|
||||
try:
|
||||
results = benchmark_sageattn3(
|
||||
batch_size=batch_size,
|
||||
num_heads=num_heads,
|
||||
seq_len=seq_len,
|
||||
head_dim=head_dim,
|
||||
is_causal=is_causal,
|
||||
dtype=dtype,
|
||||
num_warmups=10,
|
||||
num_tests=50
|
||||
)
|
||||
|
||||
print_results(results)
|
||||
all_results.append(results)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error benchmarking configuration {config}:")
|
||||
print(f" Exception: {e}")
|
||||
traceback.print_exc()
|
||||
sys.stdout.flush()
|
||||
continue
|
||||
|
||||
# Print summary table
|
||||
print("\n" + "="*120)
|
||||
print("Summary Table")
|
||||
print("="*120)
|
||||
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
|
||||
print("-"*120)
|
||||
|
||||
for r in all_results:
|
||||
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
|
||||
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
|
||||
f"{r['tokens_per_second']:<15,.0f}")
|
||||
|
||||
print("="*120 + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description='Benchmark SageAttention3 in TFLOPs')
|
||||
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
|
||||
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
|
||||
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
|
||||
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
|
||||
parser.add_argument('--causal', action='store_true', help='Use causal attention')
|
||||
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
|
||||
help='Data type')
|
||||
parser.add_argument('--per-block-mean', action='store_true', default=True,
|
||||
help='Use per-block mean for Q smoothing (default: True)')
|
||||
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
|
||||
help='Disable per-block mean for Q smoothing')
|
||||
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
|
||||
help='Use single-level P quantization (default: True)')
|
||||
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
|
||||
help='Use two-level P quantization')
|
||||
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
|
||||
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
|
||||
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
dtype_map = {
|
||||
'bfloat16': torch.bfloat16,
|
||||
'float16': torch.float16
|
||||
}
|
||||
|
||||
if args.suite:
|
||||
run_benchmark_suite()
|
||||
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
|
||||
results = benchmark_sageattn3(
|
||||
batch_size=args.batch_size,
|
||||
num_heads=args.num_heads,
|
||||
seq_len=args.seq_len,
|
||||
head_dim=args.head_dim,
|
||||
is_causal=args.causal,
|
||||
dtype=dtype_map[args.dtype],
|
||||
per_block_mean=args.per_block_mean,
|
||||
single_level_p_quant=args.single_level_p_quant,
|
||||
num_warmups=args.num_warmups,
|
||||
num_tests=args.num_tests
|
||||
)
|
||||
print_results(results)
|
||||
else:
|
||||
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
|
||||
print(
|
||||
"Example: python benchmarks/benchmark_sageattn3.py "
|
||||
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
|
||||
)
|
||||
sys.stdout.flush()
|
||||
run_benchmark_suite()
|
||||
@@ -23,7 +23,7 @@ classifiers = [
|
||||
]
|
||||
dependencies = [
|
||||
"torch>=2.5.0",
|
||||
"triton>=2.0.0; sys_platform == 'linux'",
|
||||
"triton>=2.0.0",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
@@ -32,4 +32,4 @@ dependencies = [
|
||||
[tool.scikit-build]
|
||||
cmake.build-type = "Release"
|
||||
minimum-version = "build-system.requires"
|
||||
wheel.packages = ["python/fastvideo_kernel", "attn_qat_infer"]
|
||||
wheel.packages = ["python/fastvideo_kernel"]
|
||||
|
||||
@@ -1,5 +0,0 @@
|
||||
"""Triton kernel entrypoints exposed by ``fastvideo_kernel``."""
|
||||
|
||||
from .fused_attention import attention as fused_attention
|
||||
|
||||
__all__ = ["fused_attention"]
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,55 +0,0 @@
|
||||
"""Compatibility shim for the legacy non-QAT Triton attention import path.
|
||||
|
||||
Historically callers imported
|
||||
``fastvideo_kernel.triton_kernels.fused_attention`` directly. The shared
|
||||
implementation now lives in ``attn_qat_train.py`` and is parameterized by the
|
||||
``IS_QAT`` flag. This module preserves the original public API for tests and
|
||||
downstream users while always dispatching to the non-QAT configuration.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import torch
|
||||
|
||||
from .attn_qat_train import attention as _attention
|
||||
|
||||
|
||||
def attention(
|
||||
q: torch.Tensor,
|
||||
k: torch.Tensor,
|
||||
v: torch.Tensor,
|
||||
causal: bool,
|
||||
sm_scale: float,
|
||||
warp_specialize: bool = True,
|
||||
) -> torch.Tensor:
|
||||
"""Run the shared Triton attention kernel in non-QAT mode."""
|
||||
use_qat_qkv_backward = True
|
||||
smooth_k = False
|
||||
is_qat = False
|
||||
two_level_quant_p = False
|
||||
fake_quant_p = False
|
||||
use_high_prec_o = False
|
||||
smooth_q = False
|
||||
use_global_sf_p = False
|
||||
use_global_sf_qkv = False
|
||||
|
||||
return _attention(
|
||||
q,
|
||||
k,
|
||||
v,
|
||||
causal,
|
||||
sm_scale,
|
||||
use_qat_qkv_backward,
|
||||
smooth_k,
|
||||
warp_specialize,
|
||||
is_qat,
|
||||
two_level_quant_p,
|
||||
fake_quant_p,
|
||||
use_high_prec_o,
|
||||
smooth_q,
|
||||
use_global_sf_p,
|
||||
use_global_sf_qkv,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["attention"]
|
||||
@@ -1,237 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
# Adapted from https://github.com/triton-lang/triton/blob/main/python/triton_kernels/triton_kernels/numerics_details/mxfp_details/_upcast_from_mxfp.py
|
||||
# and https://github.com/triton-lang/triton/blob/main/python/triton_kernels/triton_kernels/numerics_details/mxfp_details/_downcast_to_mxfp.py
|
||||
|
||||
import triton
|
||||
import triton.language as tl
|
||||
from triton.language.target_info import cuda_capability_geq
|
||||
|
||||
MXFP_BLOCK_SIZE = tl.constexpr(16)
|
||||
|
||||
@triton.jit
|
||||
def _compute_quant_and_scale(
|
||||
src_tensor,
|
||||
valid_src_mask,
|
||||
mx_tensor_dtype: tl.constexpr = tl.uint8,
|
||||
use_global_sf=True,
|
||||
two_level_quant_P=False,
|
||||
):
|
||||
BLOCK_SIZE_OUT_DIM: tl.constexpr = src_tensor.shape[0]
|
||||
BLOCK_SIZE_QUANT_DIM: tl.constexpr = src_tensor.shape[1]
|
||||
BLOCK_SIZE_QUANT_MX_SCALE: tl.constexpr = src_tensor.shape[1] // MXFP_BLOCK_SIZE
|
||||
is_fp4: tl.constexpr = mx_tensor_dtype == tl.uint8
|
||||
|
||||
tl.static_assert(
|
||||
is_fp4
|
||||
or mx_tensor_dtype == tl.float8e4nv
|
||||
or mx_tensor_dtype == tl.float8e5,
|
||||
"mx_tensor_dtype must be uint8, float8e4nv, or float8e5",
|
||||
)
|
||||
|
||||
# Explicit cast to fp32 since most ops are not supported on bfloat16. We avoid needless conversions to and from bf16
|
||||
f32_tensor = src_tensor.to(tl.float32)
|
||||
abs_tensor = tl.abs(f32_tensor)
|
||||
abs_tensor = tl.where(valid_src_mask, abs_tensor, -1.0) # Don't consider padding tensors in scale computation
|
||||
|
||||
if two_level_quant_P:
|
||||
# row max from SageAttn3 paper
|
||||
global_max_val = tl.max(f32_tensor, axis=1, keep_dims=True) # (BLOCK_SIZE_OUT_DIM, 1)
|
||||
global_max_val = tl.maximum(global_max_val, 1e-8)
|
||||
s_enc = ((6 * 448) / global_max_val).reshape([BLOCK_SIZE_OUT_DIM, 1, 1])
|
||||
s_dec = (1 / s_enc)
|
||||
|
||||
abs_tensor = tl.reshape(abs_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
|
||||
|
||||
if use_global_sf and not two_level_quant_P:
|
||||
global_max_val = tl.max(abs_tensor)
|
||||
# Avoid division by zero: if all values are padding (max is 0), use a default scale
|
||||
global_max_val = tl.maximum(global_max_val, 1e-8)
|
||||
s_enc = (6 * 448) / global_max_val
|
||||
s_dec = (1 / s_enc)
|
||||
elif not two_level_quant_P and not use_global_sf:
|
||||
s_dec = 1.0
|
||||
s_enc = 1.0
|
||||
|
||||
max_val = tl.max(abs_tensor, axis=2, keep_dims=True) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1) # per block maxima
|
||||
s_dec_b = max_val / 6 # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
|
||||
s_dec_b_e4m3 = (s_dec_b * s_enc).to(tl.float8e4nv) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
|
||||
s_enc_b = 1 / (s_dec_b_e4m3.to(tl.float32) * s_dec) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
|
||||
|
||||
f32_tensor = tl.reshape(f32_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
|
||||
quant_tensor = f32_tensor * s_enc_b
|
||||
|
||||
# Reshape the tensors after scaling
|
||||
quant_tensor = quant_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM])
|
||||
# Set the invalid portions of the tensor to 0. This will ensure that any padding tensors are 0 in the mx format.
|
||||
quant_tensor = tl.where(valid_src_mask, quant_tensor, 0.0)
|
||||
dequant_scale = s_dec_b_e4m3.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE])
|
||||
|
||||
if is_fp4 and cuda_capability_geq(10, 0):
|
||||
# Convert scaled values to two f32 lanes and use PTX cvt to e2m1x2 with two f32 operands.
|
||||
pairs = tl.reshape(quant_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM // 2, 2])
|
||||
lo_f, hi_f = tl.split(pairs)
|
||||
lo_f32 = lo_f.to(tl.float32)
|
||||
hi_f32 = hi_f.to(tl.float32)
|
||||
|
||||
# Inline PTX: cvt.rn.satfinite.e2m1x2.f32 takes two f32 sources and produces one .b8 packed e2m1x2.
|
||||
out_tensor = tl.inline_asm_elementwise(
|
||||
"""
|
||||
{
|
||||
.reg .b8 r;
|
||||
cvt.rn.satfinite.e2m1x2.f32 r, $1, $2;
|
||||
mov.b32 $0, {r, r, r, r};
|
||||
}
|
||||
""",
|
||||
constraints="=r,f,f",
|
||||
args=[hi_f32, lo_f32],
|
||||
dtype=tl.uint8,
|
||||
is_pure=True,
|
||||
pack=1,
|
||||
)
|
||||
elif is_fp4:
|
||||
quant_tensor = quant_tensor.to(tl.uint32, bitcast=True)
|
||||
signs = quant_tensor & 0x80000000
|
||||
exponents = (quant_tensor >> 23) & 0xFF
|
||||
mantissas_orig = (quant_tensor & 0x7FFFFF)
|
||||
|
||||
# For RTNE: 0.25 < x < 0.75 maps to 0.5 (denormal); exactly 0.25 maps to 0.0
|
||||
E8_BIAS = 127
|
||||
E2_BIAS = 1
|
||||
# Move implicit bit 1 at the beginning to mantissa for denormals
|
||||
is_subnormal = exponents < E8_BIAS
|
||||
adjusted_exponents = tl.core.sub(E8_BIAS, exponents + 1, sanitize_overflow=False)
|
||||
mantissas_pre = (0x400000 | (mantissas_orig >> 1))
|
||||
mantissas = tl.where(is_subnormal, mantissas_pre >> adjusted_exponents, mantissas_orig)
|
||||
|
||||
# For normal numbers, we change the bias from 127 to 1, and for subnormals, we keep exponent as 0.
|
||||
exponents = tl.maximum(exponents, E8_BIAS - E2_BIAS) - (E8_BIAS - E2_BIAS)
|
||||
|
||||
# Combine sign, exponent, and mantissa, while saturating
|
||||
# Round to nearest, ties to even (RTNE): use guard/sticky and LSB to decide increment
|
||||
m2bits = mantissas >> 21
|
||||
lsb_keep = (m2bits >> 1) & 0x1
|
||||
guard = m2bits & 0x1
|
||||
IS_SRC_FP32: tl.constexpr = src_tensor.dtype == tl.float32
|
||||
if IS_SRC_FP32:
|
||||
bit0_dropped = (mantissas_orig & 0x1) != 0
|
||||
mask = (1 << tl.minimum(adjusted_exponents, 31)) - 1
|
||||
dropped_post = (mantissas_pre & mask) != 0
|
||||
sticky = is_subnormal & (bit0_dropped | dropped_post)
|
||||
sticky |= ((mantissas & 0x1FFFFF) != 0).to(tl.uint32)
|
||||
else:
|
||||
sticky = ((mantissas & 0x1FFFFF) != 0).to(tl.uint32)
|
||||
round_inc = guard & (sticky | lsb_keep)
|
||||
e2m1_tmp = tl.minimum((((exponents << 2) | m2bits) + round_inc) >> 1, 0x7)
|
||||
e2m1_value = ((signs >> 28) | e2m1_tmp).to(tl.uint8)
|
||||
|
||||
e2m1_value = tl.reshape(e2m1_value, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM // 2, 2])
|
||||
evens, odds = tl.split(e2m1_value)
|
||||
out_tensor = evens | (odds << 4)
|
||||
else:
|
||||
out_tensor = quant_tensor.to(mx_tensor_dtype)
|
||||
|
||||
return out_tensor, dequant_scale, s_dec
|
||||
|
||||
@triton.jit
|
||||
def _compute_dequant(
|
||||
mx_tensor,
|
||||
scale,
|
||||
s_dec,
|
||||
BLOCK_SIZE_OUT_DIM: tl.constexpr,
|
||||
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
|
||||
dst_dtype: tl.constexpr,
|
||||
):
|
||||
tl.static_assert(BLOCK_SIZE_QUANT_DIM % MXFP_BLOCK_SIZE == 0, f"Block size along quantization block must be a multiple of {MXFP_BLOCK_SIZE=}")
|
||||
# uint8 signifies two fp4 e2m1 values packed into a single byte
|
||||
mx_tensor_dtype: tl.constexpr = mx_tensor.dtype
|
||||
tl.static_assert(dst_dtype == tl.float16 or dst_dtype == tl.bfloat16 or dst_dtype == tl.float32)
|
||||
tl.static_assert(
|
||||
mx_tensor_dtype == tl.uint8
|
||||
or ((mx_tensor_dtype == tl.float8e4nv or mx_tensor_dtype == tl.float8e5) or mx_tensor_dtype == dst_dtype),
|
||||
"mx_tensor_ptr must be uint8 or float8 or dst_dtype")
|
||||
tl.static_assert(scale.dtype == tl.float8e4nv, "scale must be float8e4nv")
|
||||
|
||||
# Determine if we are dealing with fp8 types.
|
||||
is_fp4: tl.constexpr = mx_tensor_dtype == tl.uint8
|
||||
BLOCK_SIZE_QUANT_MX_SCALE: tl.constexpr = BLOCK_SIZE_QUANT_DIM // MXFP_BLOCK_SIZE
|
||||
|
||||
# Upcast the scale to the destination type.
|
||||
if dst_dtype == tl.bfloat16:
|
||||
dst_scale = scale.to(tl.bfloat16)
|
||||
else:
|
||||
dst_scale = scale.to(tl.float32)
|
||||
if dst_dtype == tl.float16:
|
||||
dst_scale = dst_scale.to(tl.float16)
|
||||
|
||||
# Now upcast the tensor.
|
||||
intermediate_dtype: tl.constexpr = tl.bfloat16 if dst_dtype == tl.float32 else dst_dtype
|
||||
if cuda_capability_geq(10, 0):
|
||||
assert is_fp4
|
||||
packed_u32 = tl.inline_asm_elementwise(
|
||||
asm="""
|
||||
{
|
||||
.reg .b8 in_8;
|
||||
.reg .f16x2 out;
|
||||
cvt.u8.u32 in_8, $1;
|
||||
cvt.rn.f16x2.e2m1x2 out, in_8;
|
||||
mov.b32 $0, out;
|
||||
}
|
||||
""",
|
||||
constraints="=r,r",
|
||||
args=[mx_tensor], # tl.uint8 passed in as a 32-bit reg with value in low 8 bits
|
||||
dtype=tl.uint32,
|
||||
is_pure=True,
|
||||
pack=1,
|
||||
)
|
||||
lo_u16 = (packed_u32 & 0xFFFF).to(tl.uint16)
|
||||
hi_u16 = (packed_u32 >> 16).to(tl.uint16)
|
||||
lo_f16 = lo_u16.to(tl.float16, bitcast=True)
|
||||
hi_f16 = hi_u16.to(tl.float16, bitcast=True)
|
||||
|
||||
if intermediate_dtype == tl.float16:
|
||||
x0, x1 = lo_f16, hi_f16
|
||||
else:
|
||||
x0 = lo_f16.to(intermediate_dtype)
|
||||
x1 = hi_f16.to(intermediate_dtype)
|
||||
|
||||
dst_tensor = tl.interleave(x0, x1)
|
||||
|
||||
else:
|
||||
assert is_fp4
|
||||
dst_bias: tl.constexpr = 127 if intermediate_dtype == tl.bfloat16 else 15 # exponent bias
|
||||
dst_0p5: tl.constexpr = 16128 if intermediate_dtype == tl.bfloat16 else 0x3800
|
||||
dst_m_bits: tl.constexpr = 7 if intermediate_dtype == tl.bfloat16 else 10 # mantissa bits
|
||||
# e2m1
|
||||
em0 = mx_tensor & 0x07
|
||||
em1 = mx_tensor & 0x70
|
||||
x0 = (em0.to(tl.uint16) << (dst_m_bits - 1)) | ((mx_tensor & 0x08).to(tl.uint16) << 12)
|
||||
x1 = (em1.to(tl.uint16) << (dst_m_bits - 5)) | ((mx_tensor & 0x80).to(tl.uint16) << 8)
|
||||
# Three cases:
|
||||
# 1) x is normal and non-zero: Correct bias
|
||||
x0 = tl.where((em0 & 0x06) != 0, x0 + ((dst_bias - 1) << dst_m_bits), x0)
|
||||
x1 = tl.where((em1 & 0x60) != 0, x1 + ((dst_bias - 1) << dst_m_bits), x1)
|
||||
# 2) x is subnormal (x == 0bs001 where s is the sign): Map to +-0.5 in the dst type
|
||||
x0 = tl.where(em0 == 0x01, dst_0p5 | (x0 & 0x8000), x0)
|
||||
x1 = tl.where(em1 == 0x10, dst_0p5 | (x1 & 0x8000), x1)
|
||||
# 3) x is zero, do nothing
|
||||
dst_tensor = tl.interleave(x0, x1).to(intermediate_dtype, bitcast=True)
|
||||
|
||||
dst_tensor = dst_tensor.to(dst_dtype)
|
||||
|
||||
# Reshape for proper broadcasting: the scale was stored with a 16‐sized “inner” grouping.
|
||||
dst_tensor = dst_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
|
||||
dst_scale = dst_scale.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1])
|
||||
scale = scale.reshape(dst_scale.shape)
|
||||
|
||||
out_tensor = dst_tensor * dst_scale * s_dec # NVFP4 has the additional global scale factor
|
||||
if dst_dtype == tl.float32:
|
||||
max_fin = 3.4028234663852886e+38
|
||||
elif dst_dtype == tl.bfloat16:
|
||||
max_fin = 3.3895313892515355e+38
|
||||
else:
|
||||
tl.static_assert(dst_dtype == tl.float16)
|
||||
max_fin = 65504
|
||||
out_tensor = tl.clamp(out_tensor, min=-max_fin, max=max_fin)
|
||||
out_tensor = out_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM])
|
||||
out_tensor = out_tensor.to(dst_dtype)
|
||||
return out_tensor
|
||||
@@ -1,80 +0,0 @@
|
||||
import triton
|
||||
import triton.language as tl
|
||||
|
||||
from .nvfp4_utils import _compute_quant_and_scale, _compute_dequant
|
||||
|
||||
@triton.jit
|
||||
def fake_quantize(src_tensor, valid_src_mask, BLOCK_SIZE_OUT_DIM: tl.constexpr,
|
||||
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
|
||||
dst_dtype: tl.constexpr,
|
||||
mx_tensor_dtype: tl.constexpr = tl.uint8,
|
||||
use_global_sf: tl.constexpr = True,
|
||||
two_level_quant_P: tl.constexpr = False):
|
||||
high_prec_src_tensor = src_tensor
|
||||
src_tensor, src_scale, src_s_dec = _compute_quant_and_scale(src_tensor=src_tensor,
|
||||
valid_src_mask=valid_src_mask,
|
||||
mx_tensor_dtype=mx_tensor_dtype,
|
||||
use_global_sf=use_global_sf,
|
||||
two_level_quant_P=two_level_quant_P)
|
||||
src_tensor = _compute_dequant(mx_tensor=src_tensor,
|
||||
scale=src_scale,
|
||||
s_dec=src_s_dec,
|
||||
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
|
||||
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
|
||||
dst_dtype=dst_dtype)
|
||||
return src_tensor, high_prec_src_tensor.to(src_tensor.dtype)
|
||||
|
||||
@triton.jit
|
||||
def fake_quantize_q(Q, fake_Q, stride_z_q, stride_h_q,
|
||||
stride_tok_q, stride_d_q,
|
||||
fake_stride_z_q, fake_stride_h_q,
|
||||
fake_stride_tok_q, fake_stride_d_q,
|
||||
H, N_CTX_Q,
|
||||
BLOCK_M: tl.constexpr,
|
||||
HEAD_DIM: tl.constexpr,
|
||||
use_global_sf: tl.constexpr = True):
|
||||
bhid = tl.program_id(1)
|
||||
adj_q = (stride_h_q * (bhid % H) + stride_z_q * (bhid // H))
|
||||
fake_adj_q = (fake_stride_h_q * (bhid % H) + fake_stride_z_q * (bhid // H))
|
||||
Q += adj_q
|
||||
fake_Q += fake_adj_q
|
||||
|
||||
pid = tl.program_id(0)
|
||||
start_m = pid * BLOCK_M
|
||||
offs_m = start_m + tl.arange(0, BLOCK_M)
|
||||
offs_k = tl.arange(0, HEAD_DIM)
|
||||
|
||||
q_valid = offs_m < N_CTX_Q
|
||||
q = tl.load(Q + offs_m[:, None] * stride_tok_q + offs_k[None, :] * stride_d_q, mask=q_valid[:, None], other=0.0)
|
||||
q, _ = fake_quantize(src_tensor=q, valid_src_mask=q_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_M, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=q.dtype, use_global_sf=use_global_sf)
|
||||
tl.store(fake_Q + offs_m[:, None] * fake_stride_tok_q + offs_k[None, :] * fake_stride_d_q, q, mask=q_valid[:, None])
|
||||
|
||||
@triton.jit
|
||||
def fake_quantize_kv(K, V, fake_K, fake_V, stride_z_kv, stride_h_kv,
|
||||
stride_tok_kv, stride_d_kv,
|
||||
fake_stride_z_kv, fake_stride_h_kv,
|
||||
fake_stride_tok_kv, fake_stride_d_kv,
|
||||
H, N_CTX_KV,
|
||||
BLOCK_N: tl.constexpr,
|
||||
HEAD_DIM: tl.constexpr,
|
||||
use_global_sf: tl.constexpr = True):
|
||||
bhid = tl.program_id(1)
|
||||
adj_kv = (stride_h_kv * (bhid % H) + stride_z_kv * (bhid // H))
|
||||
fake_adj_kv = (fake_stride_h_kv * (bhid % H) + fake_stride_z_kv * (bhid // H))
|
||||
K += adj_kv
|
||||
V += adj_kv
|
||||
fake_K += fake_adj_kv
|
||||
fake_V += fake_adj_kv
|
||||
|
||||
pid = tl.program_id(0)
|
||||
start_n = pid * BLOCK_N
|
||||
offs_n = start_n + tl.arange(0, BLOCK_N)
|
||||
offs_k = tl.arange(0, HEAD_DIM)
|
||||
|
||||
kv_valid = offs_n < N_CTX_KV
|
||||
k_block = tl.load(K + offs_n[:, None] * stride_tok_kv + offs_k[None, :] * stride_d_kv, mask=kv_valid[:, None], other=0.0)
|
||||
v_block = tl.load(V + offs_n[:, None] * stride_tok_kv + offs_k[None, :] * stride_d_kv, mask=kv_valid[:, None], other=0.0)
|
||||
k, _ = fake_quantize(src_tensor=k_block, valid_src_mask=kv_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_N, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=k_block.dtype, use_global_sf=use_global_sf)
|
||||
v, _ = fake_quantize(src_tensor=v_block, valid_src_mask=kv_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_N, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=v_block.dtype, use_global_sf=use_global_sf)
|
||||
tl.store(fake_K + offs_n[:, None] * fake_stride_tok_kv + offs_k[None, :] * fake_stride_d_kv, k, mask=kv_valid[:, None])
|
||||
tl.store(fake_V + offs_n[:, None] * fake_stride_tok_kv + offs_k[None, :] * fake_stride_d_kv, v, mask=kv_valid[:, None])
|
||||
@@ -1,4 +0,0 @@
|
||||
from ._bootstrap import ensure_local_kernel_sources_first
|
||||
|
||||
|
||||
ensure_local_kernel_sources_first()
|
||||
|
||||
@@ -1,56 +0,0 @@
|
||||
"""Test import helpers for preferring the in-tree fastvideo-kernel sources."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _prepend_import_path(path: Path) -> None:
|
||||
path_str = str(path)
|
||||
if path_str not in sys.path:
|
||||
sys.path.insert(0, path_str)
|
||||
|
||||
|
||||
def _purge_package(name: str) -> None:
|
||||
prefix = f"{name}."
|
||||
for module_name in tuple(sys.modules):
|
||||
if module_name == name or module_name.startswith(prefix):
|
||||
sys.modules.pop(module_name, None)
|
||||
|
||||
|
||||
def _module_is_from_checkout(module_file: str | None, checkout_root: Path) -> bool:
|
||||
if module_file is None:
|
||||
return False
|
||||
|
||||
try:
|
||||
module_path = Path(module_file).resolve()
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
return module_path.is_relative_to(checkout_root.resolve())
|
||||
|
||||
|
||||
def ensure_local_kernel_sources_first() -> None:
|
||||
tests_root = Path(__file__).resolve().parent
|
||||
kernel_root = tests_root.parent
|
||||
repo_root = kernel_root.parent
|
||||
kernel_python_root = kernel_root / "python"
|
||||
|
||||
# Keep the in-tree kernel sources ahead of any preinstalled wheel so tests
|
||||
# exercise the checkout under review.
|
||||
for path in (repo_root, kernel_root, kernel_python_root):
|
||||
_prepend_import_path(path)
|
||||
|
||||
importlib.invalidate_caches()
|
||||
|
||||
loaded_kernel = sys.modules.get("fastvideo_kernel")
|
||||
if loaded_kernel is None:
|
||||
return
|
||||
|
||||
if not _module_is_from_checkout(
|
||||
getattr(loaded_kernel, "__file__", None),
|
||||
kernel_python_root,
|
||||
):
|
||||
_purge_package("fastvideo_kernel")
|
||||
@@ -1,4 +0,0 @@
|
||||
from ._bootstrap import ensure_local_kernel_sources_first
|
||||
|
||||
|
||||
ensure_local_kernel_sources_first()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,30 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import types
|
||||
from pathlib import Path
|
||||
|
||||
from tests._bootstrap import ensure_local_kernel_sources_first
|
||||
|
||||
|
||||
def test_bootstrap_prefers_local_kernel_checkout(monkeypatch) -> None:
|
||||
tests_root = Path(__file__).resolve().parent
|
||||
kernel_root = tests_root.parent
|
||||
repo_root = kernel_root.parent
|
||||
kernel_python_root = kernel_root / "python"
|
||||
|
||||
stale_module = types.ModuleType("fastvideo_kernel")
|
||||
stale_module.__file__ = (
|
||||
"/tmp/site-packages/fastvideo_kernel/__init__.py"
|
||||
)
|
||||
monkeypatch.setitem(sys.modules, "fastvideo_kernel", stale_module)
|
||||
monkeypatch.setattr(sys, "path", ["/tmp/site-packages"])
|
||||
|
||||
ensure_local_kernel_sources_first()
|
||||
|
||||
assert "fastvideo_kernel" not in sys.modules
|
||||
assert sys.path[:3] == [
|
||||
str(kernel_python_root),
|
||||
str(kernel_root),
|
||||
str(repo_root),
|
||||
]
|
||||
@@ -1,503 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Precision test to compare numerical differences for fake_quantize triton function:
|
||||
1. fake_quantize (triton implementation) vs reference implementations
|
||||
2. Tests various shapes, dtypes, and value ranges
|
||||
3. Evaluates cosine similarity, max diff, and mean diff between input and output
|
||||
"""
|
||||
|
||||
import torch
|
||||
import triton
|
||||
import triton.language as tl
|
||||
from flashinfer import SfLayout, nvfp4_quantize, e2m1_and_ufp8sf_scale_to_float
|
||||
from fastvideo_kernel.triton_kernels.nvfp4_utils import (
|
||||
_compute_quant_and_scale,
|
||||
_compute_dequant,
|
||||
)
|
||||
from typing import Optional
|
||||
|
||||
# MXFP_BLOCK_SIZE is 16 - use Python int for runtime checks
|
||||
MXFP_BLOCK_SIZE = 16
|
||||
|
||||
DEVICE = torch.device("cuda")
|
||||
|
||||
|
||||
def cosine_similarity(tensor1, tensor2):
|
||||
"""
|
||||
Compute cosine similarity between two tensors.
|
||||
|
||||
Args:
|
||||
tensor1: First tensor
|
||||
tensor2: Second tensor (same shape as tensor1)
|
||||
|
||||
Returns:
|
||||
Cosine similarity value (scalar)
|
||||
- Returns 1.0 if both tensors are zero (identical zero vectors)
|
||||
- Returns 0.0 if only one tensor is zero (orthogonal to non-zero vector)
|
||||
"""
|
||||
# Flatten tensors for computation
|
||||
t1_flat = tensor1.flatten().float()
|
||||
t2_flat = tensor2.flatten().float()
|
||||
|
||||
# Compute cosine similarity: (A · B) / (||A|| * ||B||)
|
||||
dot_product = torch.dot(t1_flat, t2_flat)
|
||||
norm1 = torch.norm(t1_flat)
|
||||
norm2 = torch.norm(t2_flat)
|
||||
|
||||
# Handle zero vectors
|
||||
if norm1 == 0 and norm2 == 0:
|
||||
# Both are zero vectors - they are identical, so similarity is 1.0
|
||||
return 1.0
|
||||
elif norm1 == 0 or norm2 == 0:
|
||||
# One is zero, one is not - they are orthogonal, so similarity is 0.0
|
||||
return 0.0
|
||||
|
||||
cos_sim = dot_product / (norm1 * norm2)
|
||||
return cos_sim.item()
|
||||
|
||||
|
||||
@triton.jit
|
||||
def fake_quantize(src_tensor, valid_src_mask, BLOCK_SIZE_OUT_DIM: tl.constexpr,
|
||||
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
|
||||
dst_dtype: tl.constexpr,
|
||||
mx_tensor_dtype: tl.constexpr = tl.uint8):
|
||||
"""
|
||||
Fake quantize function - matches API from attn_qat_train.py.
|
||||
"""
|
||||
high_prec_src_tensor = src_tensor
|
||||
src_tensor, src_scale, src_s_dec = _compute_quant_and_scale(
|
||||
src_tensor=src_tensor,
|
||||
valid_src_mask=valid_src_mask,
|
||||
mx_tensor_dtype=mx_tensor_dtype
|
||||
)
|
||||
src_tensor = _compute_dequant(
|
||||
mx_tensor=src_tensor,
|
||||
scale=src_scale,
|
||||
s_dec=src_s_dec,
|
||||
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
|
||||
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
|
||||
dst_dtype=dst_dtype
|
||||
)
|
||||
return src_tensor, high_prec_src_tensor
|
||||
|
||||
|
||||
def get_fake_quant_reference(x: torch.Tensor):
|
||||
"""
|
||||
Reference implementation using FlashInfer for comparison.
|
||||
"""
|
||||
orig_shape = x.shape
|
||||
orig_dtype = x.dtype
|
||||
device = x.device
|
||||
x = x.view(-1, x.shape[-1])
|
||||
x_global_sf = (448 * 6) / x.float().abs().nan_to_num().max()
|
||||
x_fp4, x_scale = nvfp4_quantize(x, x_global_sf, sfLayout=SfLayout.layout_128x4, do_shuffle=False)
|
||||
x_dequant = e2m1_and_ufp8sf_scale_to_float(x_fp4, x_scale, 1 / x_global_sf)
|
||||
return x_dequant.view(orig_shape).to(orig_dtype).to(device)
|
||||
|
||||
|
||||
@triton.jit
|
||||
def fake_quantize_kernel(
|
||||
src_ptr,
|
||||
dst_ptr,
|
||||
BLOCK_SIZE_OUT_DIM: tl.constexpr,
|
||||
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
|
||||
dst_dtype: tl.constexpr,
|
||||
mx_tensor_dtype: tl.constexpr,
|
||||
stride_src_outer,
|
||||
stride_src_quant,
|
||||
stride_dst_outer,
|
||||
stride_dst_quant,
|
||||
outer_dim,
|
||||
quant_dim,
|
||||
):
|
||||
"""
|
||||
Kernel wrapper to call fake_quantize on a block of data.
|
||||
"""
|
||||
outer_idx = tl.program_id(0)
|
||||
quant_idx = tl.program_id(1)
|
||||
|
||||
# Compute offsets
|
||||
start_outer = outer_idx * BLOCK_SIZE_OUT_DIM
|
||||
start_quant = quant_idx * BLOCK_SIZE_QUANT_DIM
|
||||
|
||||
# Create offset arrays
|
||||
offs_outer = tl.arange(0, BLOCK_SIZE_OUT_DIM)[:, None]
|
||||
offs_quant = tl.arange(0, BLOCK_SIZE_QUANT_DIM)[None, :]
|
||||
|
||||
# Create masks for valid elements
|
||||
mask_outer = (start_outer + offs_outer) < outer_dim
|
||||
mask_quant = (start_quant + offs_quant) < quant_dim
|
||||
full_mask = mask_outer & mask_quant
|
||||
|
||||
# Load source tensor
|
||||
src_offsets = (start_outer + offs_outer) * stride_src_outer + (start_quant + offs_quant) * stride_src_quant
|
||||
src_tensor = tl.load(src_ptr + src_offsets, mask=full_mask, other=0.0)
|
||||
|
||||
# Call fake_quantize with valid_src_mask parameter
|
||||
quantized_tensor, high_prec_tensor = fake_quantize(
|
||||
src_tensor=src_tensor,
|
||||
valid_src_mask=full_mask,
|
||||
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
|
||||
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
|
||||
dst_dtype=dst_dtype,
|
||||
mx_tensor_dtype=mx_tensor_dtype
|
||||
)
|
||||
|
||||
# Store result
|
||||
dst_offsets = (start_outer + offs_outer) * stride_dst_outer + (start_quant + offs_quant) * stride_dst_quant
|
||||
tl.store(dst_ptr + dst_offsets, quantized_tensor, mask=full_mask)
|
||||
|
||||
|
||||
def triton_fake_quantize(
|
||||
x: torch.Tensor,
|
||||
BLOCK_SIZE_OUT_DIM: int = 128,
|
||||
BLOCK_SIZE_QUANT_DIM: int = 128,
|
||||
use_fp4: bool = True, # True for fp4 (uint8), False for fp8 (float8e4nv)
|
||||
dst_dtype: Optional[torch.dtype] = None
|
||||
) -> torch.Tensor:
|
||||
"""
|
||||
Call fake_quantize triton function on a tensor.
|
||||
|
||||
Args:
|
||||
x: Input tensor (2D or can be reshaped to 2D)
|
||||
BLOCK_SIZE_OUT_DIM: Block size for outer dimension
|
||||
BLOCK_SIZE_QUANT_DIM: Block size for quantization dimension (must be multiple of 16)
|
||||
use_fp4: If True, use fp4 (uint8), else use fp8 (float8e4nv)
|
||||
dst_dtype: Output dtype (defaults to input dtype)
|
||||
|
||||
Returns:
|
||||
Fake quantized tensor
|
||||
"""
|
||||
assert x.is_cuda, "Input must be on CUDA"
|
||||
assert BLOCK_SIZE_QUANT_DIM % 16 == 0, f"BLOCK_SIZE_QUANT_DIM must be multiple of 16"
|
||||
|
||||
orig_shape = x.shape
|
||||
orig_dtype = x.dtype
|
||||
|
||||
# Reshape to 2D
|
||||
x_2d = x.view(-1, x.shape[-1])
|
||||
outer_dim, quant_dim = x_2d.shape
|
||||
|
||||
if dst_dtype is None:
|
||||
dst_dtype = orig_dtype
|
||||
|
||||
# Map torch dtype to triton dtype
|
||||
dtype_map = {
|
||||
torch.float32: tl.float32,
|
||||
torch.float16: tl.float16,
|
||||
torch.bfloat16: tl.bfloat16,
|
||||
}
|
||||
triton_dst_dtype = dtype_map.get(dst_dtype, tl.float16)
|
||||
|
||||
# Allocate output
|
||||
output = torch.empty_like(x_2d, dtype=dst_dtype)
|
||||
|
||||
# Launch kernel with appropriate quantization dtype
|
||||
grid = (
|
||||
triton.cdiv(outer_dim, BLOCK_SIZE_OUT_DIM),
|
||||
triton.cdiv(quant_dim, BLOCK_SIZE_QUANT_DIM),
|
||||
)
|
||||
|
||||
if use_fp4:
|
||||
# Use fp4 (uint8)
|
||||
fake_quantize_kernel[grid](
|
||||
src_ptr=x_2d,
|
||||
dst_ptr=output,
|
||||
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
|
||||
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
|
||||
dst_dtype=triton_dst_dtype,
|
||||
mx_tensor_dtype=tl.uint8, # fp4 uses uint8
|
||||
stride_src_outer=x_2d.stride(0),
|
||||
stride_src_quant=x_2d.stride(1),
|
||||
stride_dst_outer=output.stride(0),
|
||||
stride_dst_quant=output.stride(1),
|
||||
outer_dim=outer_dim,
|
||||
quant_dim=quant_dim,
|
||||
)
|
||||
else:
|
||||
# Use fp8 (float8e4nv)
|
||||
fake_quantize_kernel[grid](
|
||||
src_ptr=x_2d,
|
||||
dst_ptr=output,
|
||||
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
|
||||
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
|
||||
dst_dtype=triton_dst_dtype,
|
||||
mx_tensor_dtype=tl.float8e4nv, # fp8 uses float8e4nv
|
||||
stride_src_outer=x_2d.stride(0),
|
||||
stride_src_quant=x_2d.stride(1),
|
||||
stride_dst_outer=output.stride(0),
|
||||
stride_dst_quant=output.stride(1),
|
||||
outer_dim=outer_dim,
|
||||
quant_dim=quant_dim,
|
||||
)
|
||||
|
||||
return output.view(orig_shape).to(dst_dtype)
|
||||
|
||||
|
||||
def test_fake_quantize_basic():
|
||||
"""Test basic functionality of fake_quantize."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
# Test parameters
|
||||
shape = (128, 128)
|
||||
dtype = torch.bfloat16
|
||||
|
||||
# Create input tensor
|
||||
x = torch.randn(shape, dtype=dtype, device=DEVICE)
|
||||
|
||||
# Test triton fake_quantize
|
||||
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
# Check that outputs have correct shape
|
||||
assert x_fq.shape == x.shape
|
||||
assert x_fq.dtype == x.dtype
|
||||
|
||||
# Check that outputs are finite
|
||||
assert torch.isfinite(x_fq).all()
|
||||
|
||||
# Compare input and output
|
||||
max_diff = (x_fq - x).abs().max()
|
||||
mean_diff = (x_fq - x).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq, x)
|
||||
print(f" Input vs Output - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
print("✓ Basic test passed.")
|
||||
|
||||
|
||||
def test_fake_quantize_different_shapes():
|
||||
"""Test fake_quantize with different input shapes."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
test_configs = [
|
||||
(128, 64), # Small
|
||||
(256, 128), # Medium
|
||||
(512, 256), # Large
|
||||
(128, 256), # Rectangular
|
||||
(256, 128), # Rectangular (reversed)
|
||||
]
|
||||
|
||||
dtype = torch.bfloat16
|
||||
|
||||
for outer_dim, quant_dim in test_configs:
|
||||
# Create input tensor
|
||||
x = torch.randn((outer_dim, quant_dim), dtype=dtype, device=DEVICE)
|
||||
|
||||
# Test triton fake_quantize
|
||||
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
assert x_fq.shape == x.shape
|
||||
assert torch.isfinite(x_fq).all()
|
||||
|
||||
# Compare input and output
|
||||
max_diff = (x_fq - x).abs().max()
|
||||
mean_diff = (x_fq - x).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq, x)
|
||||
print(f" Shape ({outer_dim}, {quant_dim}) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
print(f"✓ Shape test passed for ({outer_dim}, {quant_dim})")
|
||||
|
||||
|
||||
def test_fake_quantize_different_dtypes():
|
||||
"""Test fake_quantize with different input dtypes."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
shape = (128, 128)
|
||||
dtypes = [torch.float32, torch.float16, torch.bfloat16]
|
||||
|
||||
for dtype in dtypes:
|
||||
# Create input tensor
|
||||
x = torch.randn(shape, dtype=dtype, device=DEVICE)
|
||||
|
||||
# Test triton fake_quantize
|
||||
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
assert x_fq.shape == x.shape
|
||||
assert x_fq.dtype == x.dtype
|
||||
assert torch.isfinite(x_fq).all()
|
||||
|
||||
# Compare input and output
|
||||
max_diff = (x_fq - x).abs().max()
|
||||
mean_diff = (x_fq - x).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq, x)
|
||||
print(f" Dtype {dtype} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
print(f"✓ Dtype test passed for {dtype}")
|
||||
|
||||
|
||||
def test_fake_quantize_3d_4d_tensors():
|
||||
"""Test fake_quantize with 3D and 4D tensors (reshaped to 2D internally)."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
dtype = torch.bfloat16
|
||||
|
||||
# Test 3D tensor (B, H, D)
|
||||
print("\nTesting 3D tensor (B, H, D)")
|
||||
x_3d = torch.randn((2, 8, 128), dtype=dtype, device=DEVICE)
|
||||
x_fq_3d = triton_fake_quantize(x_3d, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
assert x_fq_3d.shape == x_3d.shape
|
||||
assert torch.isfinite(x_fq_3d).all()
|
||||
|
||||
max_diff = (x_fq_3d - x_3d).abs().max()
|
||||
mean_diff = (x_fq_3d - x_3d).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq_3d, x_3d)
|
||||
print(f" 3D (2, 8, 128) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
# Test 4D tensor (B, H, L, D)
|
||||
print("\nTesting 4D tensor (B, H, L, D)")
|
||||
x_4d = torch.randn((1, 8, 256, 128), dtype=dtype, device=DEVICE)
|
||||
x_fq_4d = triton_fake_quantize(x_4d, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
assert x_fq_4d.shape == x_4d.shape
|
||||
assert torch.isfinite(x_fq_4d).all()
|
||||
|
||||
max_diff = (x_fq_4d - x_4d).abs().max()
|
||||
mean_diff = (x_fq_4d - x_4d).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq_4d, x_4d)
|
||||
print(f" 4D (1, 8, 256, 128) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
print("✓ 3D/4D tensor test passed.")
|
||||
|
||||
|
||||
def test_fake_quantize_edge_cases():
|
||||
"""Test edge cases like zeros, ones, extreme values."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
dtype = torch.bfloat16
|
||||
shape = (128, 128)
|
||||
|
||||
test_cases = [
|
||||
("zeros", torch.zeros(shape, dtype=dtype, device=DEVICE)),
|
||||
("ones", torch.ones(shape, dtype=dtype, device=DEVICE)),
|
||||
("negative_ones", -torch.ones(shape, dtype=dtype, device=DEVICE)),
|
||||
("very_small", torch.randn(shape, dtype=dtype, device=DEVICE) * 1e-6),
|
||||
("very_large", torch.randn(shape, dtype=dtype, device=DEVICE) * 1e6),
|
||||
]
|
||||
|
||||
for name, x in test_cases:
|
||||
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
assert x_fq.shape == x.shape
|
||||
assert torch.isfinite(x_fq).all()
|
||||
|
||||
max_diff = (x_fq - x).abs().max()
|
||||
mean_diff = (x_fq - x).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq, x)
|
||||
print(f" {name} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
print("✓ Edge cases test passed.")
|
||||
|
||||
|
||||
def test_fake_quantize_vs_reference():
|
||||
"""Test triton fake_quantize vs reference implementation."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
dtype = torch.bfloat16
|
||||
shape = (128, 128)
|
||||
|
||||
# Create input tensor
|
||||
x = torch.randn(shape, dtype=dtype, device=DEVICE)
|
||||
|
||||
# Test triton fake_quantize
|
||||
x_fq_triton = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
# Test reference implementation
|
||||
x_fq_ref = get_fake_quant_reference(x)
|
||||
|
||||
# Compare triton vs reference
|
||||
max_diff = (x_fq_triton - x_fq_ref).abs().max()
|
||||
mean_diff = (x_fq_triton - x_fq_ref).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq_triton, x_fq_ref)
|
||||
print(f" Triton vs Reference - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
# Also compare each vs input
|
||||
max_diff_triton = (x_fq_triton - x).abs().max()
|
||||
mean_diff_triton = (x_fq_triton - x).abs().mean()
|
||||
cos_sim_triton = cosine_similarity(x_fq_triton, x)
|
||||
print(f" Triton vs Input - Max diff: {max_diff_triton.item():.6f}, Mean diff: {mean_diff_triton.item():.6f}, Cosine sim: {cos_sim_triton:.6f}")
|
||||
|
||||
max_diff_ref = (x_fq_ref - x).abs().max()
|
||||
mean_diff_ref = (x_fq_ref - x).abs().mean()
|
||||
cos_sim_ref = cosine_similarity(x_fq_ref, x)
|
||||
print(f" Reference vs Input - Max diff: {max_diff_ref.item():.6f}, Mean diff: {mean_diff_ref.item():.6f}, Cosine sim: {cos_sim_ref:.6f}")
|
||||
|
||||
print("✓ Reference comparison test passed.")
|
||||
|
||||
|
||||
def test_fake_quantize_attention_shapes():
|
||||
"""Test fake_quantize with attention-like shapes (similar to test_attn_qat_train.py)."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
test_configs = [
|
||||
(2, 4, 128, 64), # Z, H, N_CTX, HEAD_DIM - Medium
|
||||
(1, 8, 256, 128), # Large head dim
|
||||
(1, 40, 9360, 128), # WAN shape
|
||||
]
|
||||
|
||||
dtype = torch.bfloat16
|
||||
|
||||
for Z, H, N_CTX, HEAD_DIM in test_configs:
|
||||
# Create input tensor in BLHD format (B, L, H, D) = (Z, N_CTX, H, HEAD_DIM)
|
||||
x = torch.randn((Z, N_CTX, H, HEAD_DIM), dtype=dtype, device=DEVICE)
|
||||
|
||||
# Test triton fake_quantize
|
||||
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
assert x_fq.shape == x.shape
|
||||
assert torch.isfinite(x_fq).all()
|
||||
|
||||
# Compare input and output
|
||||
max_diff = (x_fq - x).abs().max()
|
||||
mean_diff = (x_fq - x).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq, x)
|
||||
print(f" (Z={Z}, H={H}, N_CTX={N_CTX}, HEAD_DIM={HEAD_DIM}) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
print(f"✓ Attention shape test passed for (Z={Z}, H={H}, N_CTX={N_CTX}, HEAD_DIM={HEAD_DIM})")
|
||||
|
||||
|
||||
def test_fake_quantize_non_divisible_blocks():
|
||||
"""Test fake_quantize with shapes that are not divisible by block sizes."""
|
||||
torch.manual_seed(42)
|
||||
|
||||
dtype = torch.bfloat16
|
||||
|
||||
# Test with non-divisible dimensions
|
||||
test_shapes = [
|
||||
(100, 100), # Not divisible by 128
|
||||
(150, 200), # Not divisible by 128
|
||||
(256, 100), # One dimension divisible, one not
|
||||
]
|
||||
|
||||
for shape in test_shapes:
|
||||
x = torch.randn(shape, dtype=dtype, device=DEVICE)
|
||||
|
||||
# Test triton fake_quantize
|
||||
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
|
||||
|
||||
assert x_fq.shape == x.shape
|
||||
assert torch.isfinite(x_fq).all()
|
||||
|
||||
max_diff = (x_fq - x).abs().max()
|
||||
mean_diff = (x_fq - x).abs().mean()
|
||||
cos_sim = cosine_similarity(x_fq, x)
|
||||
print(f" Shape {shape} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
|
||||
|
||||
print(f"✓ Non-divisible block test passed for {shape}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("Running fake_quantize tests...")
|
||||
print(f"Device: {DEVICE}")
|
||||
print()
|
||||
|
||||
test_fake_quantize_basic()
|
||||
test_fake_quantize_different_shapes()
|
||||
test_fake_quantize_different_dtypes()
|
||||
test_fake_quantize_3d_4d_tensors()
|
||||
test_fake_quantize_edge_cases()
|
||||
test_fake_quantize_vs_reference()
|
||||
test_fake_quantize_attention_shapes()
|
||||
test_fake_quantize_non_divisible_blocks()
|
||||
|
||||
print()
|
||||
print("All tests passed! ✓")
|
||||
@@ -1,65 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from fastvideo.api.schema import (
|
||||
CompileConfig,
|
||||
ComponentConfig,
|
||||
ContinuationState,
|
||||
EngineConfig,
|
||||
GenerationPlan,
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
InputConfig,
|
||||
OffloadConfig,
|
||||
OutputConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
PlannedStage,
|
||||
QuantizationConfig,
|
||||
RequestRuntimeConfig,
|
||||
RunConfig,
|
||||
SamplingConfig,
|
||||
ServeConfig,
|
||||
ServerConfig,
|
||||
)
|
||||
from fastvideo.api.errors import ConfigValidationError
|
||||
from fastvideo.api.overrides import apply_overrides, parse_cli_overrides
|
||||
from fastvideo.api.parser import (
|
||||
config_to_dict,
|
||||
load_config,
|
||||
load_raw_config,
|
||||
load_run_config,
|
||||
load_serve_config,
|
||||
parse_config,
|
||||
)
|
||||
from fastvideo.api.results import GenerationResult
|
||||
|
||||
__all__ = [
|
||||
"CompileConfig",
|
||||
"ComponentConfig",
|
||||
"ContinuationState",
|
||||
"ConfigValidationError",
|
||||
"EngineConfig",
|
||||
"GenerationResult",
|
||||
"GenerationPlan",
|
||||
"GenerationRequest",
|
||||
"GeneratorConfig",
|
||||
"InputConfig",
|
||||
"OffloadConfig",
|
||||
"OutputConfig",
|
||||
"ParallelismConfig",
|
||||
"PipelineSelection",
|
||||
"PlannedStage",
|
||||
"QuantizationConfig",
|
||||
"RequestRuntimeConfig",
|
||||
"RunConfig",
|
||||
"SamplingConfig",
|
||||
"ServeConfig",
|
||||
"ServerConfig",
|
||||
"apply_overrides",
|
||||
"config_to_dict",
|
||||
"load_config",
|
||||
"load_raw_config",
|
||||
"load_run_config",
|
||||
"load_serve_config",
|
||||
"parse_cli_overrides",
|
||||
"parse_config",
|
||||
]
|
||||
@@ -1,503 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping
|
||||
from copy import deepcopy
|
||||
from dataclasses import fields, is_dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from fastvideo.api.overrides import apply_overrides, parse_cli_overrides
|
||||
from fastvideo.api.parser import config_to_dict, load_raw_config, parse_config
|
||||
from fastvideo.api.schema import (
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
InputConfig,
|
||||
OutputConfig,
|
||||
RequestRuntimeConfig,
|
||||
SamplingConfig,
|
||||
)
|
||||
from fastvideo.configs.sample import SamplingParam
|
||||
from fastvideo.fastvideo_args import FastVideoArgs
|
||||
from fastvideo.utils import shallow_asdict
|
||||
|
||||
_EXPLICIT_REQUEST_ATTR = "_fastvideo_explicit_request"
|
||||
_INPUT_FIELD_NAMES = {field.name for field in fields(InputConfig)}
|
||||
_SAMPLING_FIELD_NAMES = {field.name for field in fields(SamplingConfig)}
|
||||
_RUNTIME_FIELD_NAMES = {field.name for field in fields(RequestRuntimeConfig)}
|
||||
_OUTPUT_FIELD_NAMES = {field.name for field in fields(OutputConfig)}
|
||||
_MISSING = object()
|
||||
_LEGACY_REQUEST_ALIASES = {
|
||||
"neg_prompt": "negative_prompt",
|
||||
}
|
||||
_REQUEST_PIPELINE_OVERRIDE_FIELDS = frozenset({
|
||||
"embedded_cfg_scale",
|
||||
})
|
||||
|
||||
|
||||
def normalize_generator_config(config: GeneratorConfig | Mapping[str, Any], ) -> GeneratorConfig:
|
||||
if isinstance(config, GeneratorConfig):
|
||||
return config
|
||||
return parse_config(GeneratorConfig, config)
|
||||
|
||||
|
||||
def load_generator_config_from_file(
|
||||
path: str | Path,
|
||||
overrides: list[str] | Mapping[str, Any] | None = None,
|
||||
) -> GeneratorConfig:
|
||||
raw = load_raw_config(path)
|
||||
normalized_overrides = _normalize_overrides(overrides)
|
||||
|
||||
if _looks_like_run_or_serve_config(raw):
|
||||
if normalized_overrides:
|
||||
raw = apply_overrides(raw, normalized_overrides)
|
||||
return parse_config(GeneratorConfig, raw["generator"])
|
||||
|
||||
if normalized_overrides:
|
||||
adjusted = normalized_overrides
|
||||
if all(key.startswith("generator.") for key in adjusted):
|
||||
adjusted = {key[len("generator."):]: value for key, value in adjusted.items()}
|
||||
raw = apply_overrides(raw, adjusted)
|
||||
|
||||
return parse_config(GeneratorConfig, raw)
|
||||
|
||||
|
||||
def legacy_from_pretrained_to_config(
|
||||
model_path: str,
|
||||
kwargs: Mapping[str, Any],
|
||||
) -> GeneratorConfig:
|
||||
raw: dict[str, Any] = {"model_path": model_path}
|
||||
engine: dict[str, Any] = {}
|
||||
parallelism: dict[str, Any] = {}
|
||||
offload: dict[str, Any] = {}
|
||||
compile_config: dict[str, Any] = {}
|
||||
pipeline: dict[str, Any] = {}
|
||||
components: dict[str, Any] = {}
|
||||
quantization: dict[str, Any] = {}
|
||||
experimental: dict[str, Any] = {}
|
||||
|
||||
for key, value in kwargs.items():
|
||||
if key == "revision":
|
||||
raw["revision"] = value
|
||||
elif key == "trust_remote_code":
|
||||
raw["trust_remote_code"] = value
|
||||
elif key == "num_gpus":
|
||||
engine["num_gpus"] = value
|
||||
elif key == "distributed_executor_backend":
|
||||
engine["execution_backend"] = value
|
||||
elif key in {"tp_size", "sp_size", "hsdp_replicate_dim", "hsdp_shard_dim", "dist_timeout"}:
|
||||
parallelism[key] = value
|
||||
elif key == "dit_cpu_offload":
|
||||
offload["dit"] = value
|
||||
elif key == "dit_layerwise_offload":
|
||||
offload["dit_layerwise"] = value
|
||||
elif key == "text_encoder_cpu_offload":
|
||||
offload["text_encoder"] = value
|
||||
elif key == "image_encoder_cpu_offload":
|
||||
offload["image_encoder"] = value
|
||||
elif key == "vae_cpu_offload":
|
||||
offload["vae"] = value
|
||||
elif key == "pin_cpu_memory":
|
||||
offload["pin_cpu_memory"] = value
|
||||
elif key == "enable_torch_compile":
|
||||
compile_config["enabled"] = value
|
||||
elif key == "torch_compile_kwargs":
|
||||
compile_config["kwargs"] = deepcopy(value)
|
||||
elif key in {"enable_stage_verification", "use_fsdp_inference", "disable_autocast"}:
|
||||
engine[key] = value
|
||||
elif key == "override_text_encoder_quant":
|
||||
quantization["text_encoder_quant"] = value
|
||||
elif key == "transformer_quant":
|
||||
quantization["transformer_quant"] = value
|
||||
elif key == "workload_type":
|
||||
pipeline["workload_type"] = value
|
||||
elif key == "lora_path":
|
||||
components["lora_path"] = value
|
||||
elif key == "override_pipeline_cls_name":
|
||||
components["override_pipeline_cls_name"] = value
|
||||
elif key == "override_transformer_cls_name":
|
||||
components["override_transformer_cls_name"] = value
|
||||
elif key == "pipeline_config":
|
||||
if isinstance(value, str):
|
||||
components["pipeline_config_path"] = value
|
||||
else:
|
||||
experimental[key] = deepcopy(value)
|
||||
elif key == "override_text_encoder_safetensors":
|
||||
components["text_encoder_weights"] = value
|
||||
elif key == "init_weights_from_safetensors":
|
||||
components["transformer_weights"] = value
|
||||
elif key == "init_weights_from_safetensors_2":
|
||||
components["transformer_2_weights"] = value
|
||||
else:
|
||||
experimental[key] = deepcopy(value)
|
||||
|
||||
if parallelism:
|
||||
engine["parallelism"] = parallelism
|
||||
if offload:
|
||||
engine["offload"] = offload
|
||||
if compile_config:
|
||||
engine["compile"] = compile_config
|
||||
if quantization:
|
||||
engine["quantization"] = quantization
|
||||
if engine:
|
||||
raw["engine"] = engine
|
||||
|
||||
if components:
|
||||
pipeline["components"] = components
|
||||
if experimental:
|
||||
pipeline["experimental"] = experimental
|
||||
if pipeline:
|
||||
raw["pipeline"] = pipeline
|
||||
|
||||
return parse_config(GeneratorConfig, raw)
|
||||
|
||||
|
||||
def generator_config_to_fastvideo_args(config: GeneratorConfig | Mapping[str, Any], ) -> FastVideoArgs:
|
||||
normalized = normalize_generator_config(config)
|
||||
unsupported = []
|
||||
if normalized.pipeline.profile is not None:
|
||||
unsupported.append("pipeline.profile")
|
||||
if normalized.pipeline.profile_version is not None:
|
||||
unsupported.append("pipeline.profile_version")
|
||||
if normalized.pipeline.components.config_root is not None:
|
||||
unsupported.append("pipeline.components.config_root")
|
||||
if normalized.pipeline.components.vae_weights is not None:
|
||||
unsupported.append("pipeline.components.vae_weights")
|
||||
if normalized.pipeline.components.upsampler_weights is not None:
|
||||
unsupported.append("pipeline.components.upsampler_weights")
|
||||
if unsupported:
|
||||
joined = ", ".join(unsupported)
|
||||
raise NotImplementedError(f"VideoGenerator compatibility adapter does not support {joined} yet")
|
||||
|
||||
engine = normalized.engine
|
||||
kwargs: dict[str, Any] = {
|
||||
"model_path": normalized.model_path,
|
||||
"revision": normalized.revision,
|
||||
"trust_remote_code": normalized.trust_remote_code,
|
||||
"num_gpus": engine.num_gpus,
|
||||
"distributed_executor_backend": engine.execution_backend,
|
||||
"tp_size": engine.parallelism.tp_size,
|
||||
"sp_size": engine.parallelism.sp_size,
|
||||
"hsdp_replicate_dim": engine.parallelism.hsdp_replicate_dim,
|
||||
"hsdp_shard_dim": engine.parallelism.hsdp_shard_dim,
|
||||
"dist_timeout": engine.parallelism.dist_timeout,
|
||||
"dit_cpu_offload": engine.offload.dit,
|
||||
"dit_layerwise_offload": engine.offload.dit_layerwise,
|
||||
"text_encoder_cpu_offload": engine.offload.text_encoder,
|
||||
"image_encoder_cpu_offload": engine.offload.image_encoder,
|
||||
"vae_cpu_offload": engine.offload.vae,
|
||||
"pin_cpu_memory": engine.offload.pin_cpu_memory,
|
||||
"enable_torch_compile": engine.compile.enabled,
|
||||
"torch_compile_kwargs": deepcopy(engine.compile.kwargs),
|
||||
"enable_stage_verification": engine.enable_stage_verification,
|
||||
"use_fsdp_inference": engine.use_fsdp_inference,
|
||||
"disable_autocast": engine.disable_autocast,
|
||||
}
|
||||
if normalized.pipeline.workload_type is not None:
|
||||
kwargs["workload_type"] = normalized.pipeline.workload_type
|
||||
|
||||
quantization = engine.quantization
|
||||
if quantization is not None and quantization.text_encoder_quant is not None:
|
||||
kwargs["override_text_encoder_quant"] = quantization.text_encoder_quant
|
||||
if quantization is not None and quantization.transformer_quant is not None:
|
||||
kwargs["transformer_quant"] = quantization.transformer_quant
|
||||
|
||||
components = normalized.pipeline.components
|
||||
if components.pipeline_config_path is not None:
|
||||
kwargs["pipeline_config"] = components.pipeline_config_path
|
||||
if components.lora_path is not None:
|
||||
kwargs["lora_path"] = components.lora_path
|
||||
if components.override_pipeline_cls_name is not None:
|
||||
kwargs["override_pipeline_cls_name"] = components.override_pipeline_cls_name
|
||||
if components.override_transformer_cls_name is not None:
|
||||
kwargs["override_transformer_cls_name"] = components.override_transformer_cls_name
|
||||
if components.text_encoder_weights is not None:
|
||||
kwargs["override_text_encoder_safetensors"] = components.text_encoder_weights
|
||||
if components.transformer_weights is not None:
|
||||
kwargs["init_weights_from_safetensors"] = components.transformer_weights
|
||||
if components.transformer_2_weights is not None:
|
||||
kwargs["init_weights_from_safetensors_2"] = components.transformer_2_weights
|
||||
|
||||
kwargs.update(deepcopy(normalized.pipeline.profile_overrides))
|
||||
kwargs.update(deepcopy(normalized.pipeline.experimental))
|
||||
return FastVideoArgs.from_kwargs(**kwargs)
|
||||
|
||||
|
||||
def normalize_generation_request(request: GenerationRequest | Mapping[str, Any], ) -> GenerationRequest:
|
||||
normalized = (request if isinstance(request, GenerationRequest) else parse_config(GenerationRequest, request))
|
||||
|
||||
if not hasattr(normalized, _EXPLICIT_REQUEST_ATTR):
|
||||
setattr(normalized, _EXPLICIT_REQUEST_ATTR, _serialize_generation_request(normalized))
|
||||
return normalized
|
||||
|
||||
|
||||
def legacy_generate_call_to_request(
|
||||
prompt: str | None,
|
||||
sampling_param: SamplingParam | None,
|
||||
*,
|
||||
mouse_cond: Any | None = None,
|
||||
keyboard_cond: Any | None = None,
|
||||
grid_sizes: Any | None = None,
|
||||
legacy_kwargs: Mapping[str, Any] | None = None,
|
||||
) -> GenerationRequest:
|
||||
raw = _sampling_param_to_request_raw(sampling_param)
|
||||
if prompt is not None:
|
||||
raw["prompt"] = prompt
|
||||
|
||||
for key, value in (legacy_kwargs or {}).items():
|
||||
_apply_request_field(raw, key, value)
|
||||
|
||||
if mouse_cond is not None:
|
||||
raw.setdefault("inputs", {})["mouse_cond"] = mouse_cond
|
||||
if keyboard_cond is not None:
|
||||
raw.setdefault("inputs", {})["keyboard_cond"] = keyboard_cond
|
||||
if grid_sizes is not None:
|
||||
raw.setdefault("inputs", {})["grid_sizes"] = grid_sizes
|
||||
|
||||
normalized = parse_config(GenerationRequest, raw)
|
||||
setattr(normalized, _EXPLICIT_REQUEST_ATTR, deepcopy(raw))
|
||||
return normalized
|
||||
|
||||
|
||||
def request_to_sampling_param(
|
||||
request: GenerationRequest,
|
||||
*,
|
||||
model_path: str,
|
||||
) -> SamplingParam:
|
||||
if request.plan is not None:
|
||||
raise NotImplementedError("GenerationRequest.plan is not wired into VideoGenerator yet")
|
||||
if request.state is not None:
|
||||
raise NotImplementedError("GenerationRequest.state is not wired into VideoGenerator yet")
|
||||
|
||||
sampling_param = SamplingParam.from_pretrained(model_path)
|
||||
updates = _explicit_request_updates(request)
|
||||
|
||||
for key, value in updates.items():
|
||||
if hasattr(sampling_param, key):
|
||||
setattr(sampling_param, key, deepcopy(value))
|
||||
elif key in _REQUEST_PIPELINE_OVERRIDE_FIELDS or _is_supported_as_default_only(key, value):
|
||||
continue
|
||||
else:
|
||||
raise ValueError(f"Request field {key!r} is not supported by sampling params for {model_path}")
|
||||
|
||||
sampling_param.__post_init__()
|
||||
sampling_param.check_sampling_param()
|
||||
return sampling_param
|
||||
|
||||
|
||||
def expand_request_prompt_batch(request: GenerationRequest, ) -> list[GenerationRequest]:
|
||||
if not isinstance(request.prompt, list):
|
||||
return [request]
|
||||
|
||||
requests: list[GenerationRequest] = []
|
||||
for index, prompt in enumerate(request.prompt):
|
||||
single_request = deepcopy(request)
|
||||
single_request.prompt = prompt
|
||||
_fan_out_batched_input_value(request, single_request, "image_path", index)
|
||||
_fan_out_batched_input_value(request, single_request, "video_path", index)
|
||||
_fan_out_explicit_request_metadata(request, single_request, index, prompt)
|
||||
requests.append(single_request)
|
||||
return requests
|
||||
|
||||
|
||||
def _looks_like_run_or_serve_config(raw: Mapping[str, Any]) -> bool:
|
||||
return isinstance(raw.get("generator"), Mapping)
|
||||
|
||||
|
||||
def _normalize_overrides(overrides: list[str] | Mapping[str, Any] | None, ) -> dict[str, Any] | None:
|
||||
if not overrides:
|
||||
return None
|
||||
if isinstance(overrides, list):
|
||||
return parse_cli_overrides(overrides)
|
||||
return dict(overrides)
|
||||
|
||||
|
||||
def _sampling_param_to_request_raw(sampling_param: SamplingParam | None, ) -> dict[str, Any]:
|
||||
if sampling_param is None:
|
||||
return {}
|
||||
|
||||
raw: dict[str, Any] = {}
|
||||
for key, value in shallow_asdict(sampling_param).items():
|
||||
if key == "prompt":
|
||||
continue
|
||||
_apply_request_field(raw, key, deepcopy(value))
|
||||
return raw
|
||||
|
||||
|
||||
def _apply_request_field(
|
||||
raw: dict[str, Any],
|
||||
key: str,
|
||||
value: Any,
|
||||
) -> None:
|
||||
key = _LEGACY_REQUEST_ALIASES.get(key, key)
|
||||
if key == "negative_prompt":
|
||||
raw["negative_prompt"] = value
|
||||
return
|
||||
if key in _INPUT_FIELD_NAMES:
|
||||
raw.setdefault("inputs", {})[key] = value
|
||||
return
|
||||
if key in _SAMPLING_FIELD_NAMES:
|
||||
raw.setdefault("sampling", {})[key] = value
|
||||
return
|
||||
if key in _RUNTIME_FIELD_NAMES:
|
||||
raw.setdefault("runtime", {})[key] = value
|
||||
return
|
||||
if key in _OUTPUT_FIELD_NAMES:
|
||||
raw.setdefault("output", {})[key] = value
|
||||
return
|
||||
raw.setdefault("extensions", {})[key] = value
|
||||
|
||||
|
||||
def request_to_pipeline_overrides(request: GenerationRequest) -> dict[str, Any]:
|
||||
overrides: dict[str, Any] = {}
|
||||
for key, value in _explicit_request_updates(request).items():
|
||||
if key in _REQUEST_PIPELINE_OVERRIDE_FIELDS:
|
||||
overrides[key] = deepcopy(value)
|
||||
return overrides
|
||||
|
||||
|
||||
def _explicit_request_updates(request: GenerationRequest) -> dict[str, Any]:
|
||||
raw = getattr(request, _EXPLICIT_REQUEST_ATTR, None)
|
||||
if raw is None:
|
||||
raw = _serialize_generation_request(request)
|
||||
|
||||
return _extract_request_updates(raw)
|
||||
|
||||
|
||||
def _extract_request_updates(raw: Mapping[str, Any]) -> dict[str, Any]:
|
||||
updates: dict[str, Any] = {}
|
||||
if "negative_prompt" in raw:
|
||||
updates["negative_prompt"] = deepcopy(raw["negative_prompt"])
|
||||
|
||||
for section_name in ("inputs", "sampling", "runtime", "output"):
|
||||
section = raw.get(section_name)
|
||||
if not isinstance(section, Mapping):
|
||||
continue
|
||||
for key, value in section.items():
|
||||
updates[key] = deepcopy(value)
|
||||
|
||||
stage_overrides = raw.get("stage_overrides")
|
||||
if stage_overrides:
|
||||
updates.update(_flatten_stage_overrides(stage_overrides))
|
||||
|
||||
extensions = raw.get("extensions")
|
||||
if isinstance(extensions, Mapping):
|
||||
for key, value in extensions.items():
|
||||
updates[key] = deepcopy(value)
|
||||
|
||||
return updates
|
||||
|
||||
|
||||
def _flatten_stage_overrides(stage_overrides: Any) -> dict[str, Any]:
|
||||
if not isinstance(stage_overrides, Mapping):
|
||||
raise ValueError("GenerationRequest.stage_overrides must be a mapping")
|
||||
|
||||
flattened: dict[str, Any] = {}
|
||||
for stage_name, overrides in stage_overrides.items():
|
||||
if not isinstance(overrides, Mapping):
|
||||
raise ValueError(f"GenerationRequest.stage_overrides.{stage_name} must be a mapping")
|
||||
for key, value in overrides.items():
|
||||
if key in flattened and flattened[key] != value:
|
||||
raise ValueError(f"Conflicting stage override for {key!r} across stages")
|
||||
flattened[key] = deepcopy(value)
|
||||
return flattened
|
||||
|
||||
|
||||
def _serialize_generation_request(request: GenerationRequest) -> dict[str, Any]:
|
||||
return deepcopy(config_to_dict(request))
|
||||
|
||||
|
||||
def _fan_out_batched_input_value(
|
||||
source_request: GenerationRequest,
|
||||
target_request: GenerationRequest,
|
||||
field_name: str,
|
||||
index: int,
|
||||
) -> None:
|
||||
value = getattr(source_request.inputs, field_name)
|
||||
if not isinstance(value, list):
|
||||
return
|
||||
_validate_batched_input_length(source_request.prompt, value, field_name)
|
||||
setattr(target_request.inputs, field_name, deepcopy(value[index]))
|
||||
|
||||
|
||||
def _fan_out_explicit_request_metadata(
|
||||
source_request: GenerationRequest,
|
||||
target_request: GenerationRequest,
|
||||
index: int,
|
||||
prompt: str,
|
||||
) -> None:
|
||||
raw = getattr(source_request, _EXPLICIT_REQUEST_ATTR, None)
|
||||
if raw is None:
|
||||
return
|
||||
|
||||
raw = deepcopy(raw)
|
||||
raw["prompt"] = prompt
|
||||
inputs = raw.get("inputs")
|
||||
if isinstance(inputs, dict):
|
||||
for field_name in ("image_path", "video_path"):
|
||||
value = inputs.get(field_name)
|
||||
if isinstance(value, list):
|
||||
_validate_batched_input_length(source_request.prompt, value, field_name)
|
||||
inputs[field_name] = deepcopy(value[index])
|
||||
|
||||
setattr(target_request, _EXPLICIT_REQUEST_ATTR, raw)
|
||||
|
||||
|
||||
def _validate_batched_input_length(
|
||||
prompts: str | list[str] | None,
|
||||
values: list[Any],
|
||||
field_name: str,
|
||||
) -> None:
|
||||
if not isinstance(prompts, list):
|
||||
return
|
||||
if len(values) != len(prompts):
|
||||
raise ValueError(f"GenerationRequest.inputs.{field_name} must have the same length as request.prompt")
|
||||
|
||||
|
||||
def _is_supported_as_default_only(key: str, value: Any) -> bool:
|
||||
default_value = _DEFAULT_REQUEST_UPDATES.get(key, _MISSING)
|
||||
return default_value is not _MISSING and _values_equal(value, default_value)
|
||||
|
||||
|
||||
def _collect_non_default_fields(
|
||||
value: Any,
|
||||
default: Any,
|
||||
) -> dict[str, Any]:
|
||||
if not (is_dataclass(value) and is_dataclass(default)):
|
||||
return {}
|
||||
|
||||
result: dict[str, Any] = {}
|
||||
for field in fields(value):
|
||||
current = getattr(value, field.name)
|
||||
default_value = getattr(default, field.name)
|
||||
if is_dataclass(current) and is_dataclass(default_value):
|
||||
nested = _collect_non_default_fields(current, default_value)
|
||||
if nested:
|
||||
result[field.name] = nested
|
||||
continue
|
||||
if not _values_equal(current, default_value):
|
||||
result[field.name] = deepcopy(current)
|
||||
return result
|
||||
|
||||
|
||||
def _values_equal(left: Any, right: Any) -> bool:
|
||||
if left is right:
|
||||
return True
|
||||
try:
|
||||
return bool(left == right)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
_DEFAULT_REQUEST_UPDATES = _extract_request_updates(config_to_dict(GenerationRequest()))
|
||||
|
||||
__all__ = [
|
||||
"generator_config_to_fastvideo_args",
|
||||
"legacy_from_pretrained_to_config",
|
||||
"legacy_generate_call_to_request",
|
||||
"load_generator_config_from_file",
|
||||
"normalize_generation_request",
|
||||
"normalize_generator_config",
|
||||
"request_to_pipeline_overrides",
|
||||
"request_to_sampling_param",
|
||||
]
|
||||
@@ -1,16 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
class ConfigValidationError(ValueError):
|
||||
"""Validation error that keeps track of the nested config path."""
|
||||
|
||||
def __init__(self, path: str, message: str):
|
||||
self.path = path
|
||||
self.message = message
|
||||
super().__init__(str(self))
|
||||
|
||||
def __str__(self) -> str:
|
||||
if self.path:
|
||||
return f"{self.path}: {self.message}"
|
||||
return self.message
|
||||
@@ -1,97 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
from typing import Any
|
||||
from collections.abc import Mapping
|
||||
|
||||
import yaml
|
||||
|
||||
from fastvideo.api.errors import ConfigValidationError
|
||||
|
||||
|
||||
def parse_cli_overrides(overrides: list[str]) -> dict[str, Any]:
|
||||
"""Parse ``--dotted.key value`` style overrides into a flat mapping."""
|
||||
parsed: dict[str, Any] = {}
|
||||
index = 0
|
||||
while index < len(overrides):
|
||||
token = overrides[index]
|
||||
if not token.startswith("--"):
|
||||
raise ValueError(f"Expected --dotted.key, got {token!r}")
|
||||
|
||||
key = token[2:]
|
||||
if not key:
|
||||
raise ValueError("Override key cannot be empty")
|
||||
|
||||
if "=" in key:
|
||||
key, raw_value = key.split("=", 1)
|
||||
else:
|
||||
index += 1
|
||||
if index >= len(overrides):
|
||||
raise ValueError(f"Missing value for override {token!r}")
|
||||
raw_value = overrides[index]
|
||||
|
||||
parsed[key] = _cast_override_value(raw_value)
|
||||
index += 1
|
||||
|
||||
return parsed
|
||||
|
||||
|
||||
def apply_overrides(config: Mapping[str, Any], overrides: Mapping[str, Any]) -> dict[str, Any]:
|
||||
"""Return a copy of ``config`` with dotted-key overrides applied."""
|
||||
merged = deepcopy(dict(config))
|
||||
for dotted_key, value in overrides.items():
|
||||
_apply_single_override(merged, dotted_key, value)
|
||||
return merged
|
||||
|
||||
|
||||
def _apply_single_override(config: dict[str, Any], dotted_key: str, value: Any) -> None:
|
||||
parts = dotted_key.split(".")
|
||||
if not all(parts):
|
||||
raise ValueError(f"Invalid override path {dotted_key!r}")
|
||||
|
||||
cursor = config
|
||||
for depth, part in enumerate(parts[:-1]):
|
||||
existing = cursor.get(part)
|
||||
if existing is None:
|
||||
existing = {}
|
||||
cursor[part] = existing
|
||||
elif not isinstance(existing, dict):
|
||||
raise ConfigValidationError(
|
||||
".".join(parts[:depth + 1]),
|
||||
"cannot apply nested override through a non-mapping value",
|
||||
)
|
||||
cursor = existing
|
||||
|
||||
cursor[parts[-1]] = value
|
||||
|
||||
|
||||
def _cast_override_value(raw: str) -> Any:
|
||||
lowered = raw.lower()
|
||||
if lowered == "true":
|
||||
return True
|
||||
if lowered == "false":
|
||||
return False
|
||||
if lowered in {"none", "null"}:
|
||||
return None
|
||||
|
||||
try:
|
||||
return int(raw)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
try:
|
||||
return float(raw)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
if raw.startswith("[") or raw.startswith("{"):
|
||||
try:
|
||||
return yaml.safe_load(raw)
|
||||
except yaml.YAMLError:
|
||||
pass
|
||||
|
||||
return raw
|
||||
|
||||
|
||||
__all__ = ["apply_overrides", "parse_cli_overrides"]
|
||||
@@ -1,320 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
import json
|
||||
import types
|
||||
from pathlib import Path
|
||||
from collections.abc import Mapping
|
||||
from typing import Any, Literal, TypeVar, Union, get_args, get_origin, get_type_hints
|
||||
|
||||
import yaml
|
||||
|
||||
from fastvideo.api.errors import ConfigValidationError
|
||||
from fastvideo.api.overrides import apply_overrides, parse_cli_overrides
|
||||
from fastvideo.api.schema import RunConfig, ServeConfig
|
||||
|
||||
T = TypeVar("T")
|
||||
_UNION_ORIGINS = {types.UnionType, Union}
|
||||
|
||||
|
||||
@dataclasses.dataclass(frozen=True)
|
||||
class _DataclassSpec:
|
||||
cls: type[Any]
|
||||
type_hints: dict[str, Any]
|
||||
fields_by_name: dict[str, dataclasses.Field[Any]]
|
||||
|
||||
|
||||
def parse_config(config_type: type[T], raw: Mapping[str, Any] | T) -> T:
|
||||
"""Parse a nested mapping into a typed inference config object."""
|
||||
if isinstance(raw, config_type):
|
||||
return raw
|
||||
if not isinstance(raw, Mapping):
|
||||
raise ConfigValidationError("", f"expected mapping for {config_type.__name__}")
|
||||
return _SchemaParser().parse_dataclass(config_type, raw, "")
|
||||
|
||||
|
||||
def config_to_dict(config: Any) -> Any:
|
||||
"""Serialize a typed config object into plain Python containers."""
|
||||
if dataclasses.is_dataclass(config) and not isinstance(config, type):
|
||||
return {field.name: config_to_dict(getattr(config, field.name)) for field in dataclasses.fields(config)}
|
||||
if isinstance(config, list):
|
||||
return [config_to_dict(item) for item in config]
|
||||
if isinstance(config, dict):
|
||||
return {key: config_to_dict(value) for key, value in config.items()}
|
||||
return config
|
||||
|
||||
|
||||
def load_config(
|
||||
config_type: type[T],
|
||||
path: str | Path,
|
||||
overrides: list[str] | Mapping[str, Any] | None = None,
|
||||
) -> T:
|
||||
"""Load a typed config object from YAML or JSON."""
|
||||
raw = load_raw_config(path)
|
||||
normalized_overrides = _normalize_overrides(overrides)
|
||||
if normalized_overrides:
|
||||
raw = apply_overrides(raw, normalized_overrides)
|
||||
return parse_config(config_type, raw)
|
||||
|
||||
|
||||
def load_run_config(
|
||||
path: str | Path,
|
||||
overrides: list[str] | Mapping[str, Any] | None = None,
|
||||
) -> RunConfig:
|
||||
return load_config(RunConfig, path, overrides)
|
||||
|
||||
|
||||
def load_serve_config(
|
||||
path: str | Path,
|
||||
overrides: list[str] | Mapping[str, Any] | None = None,
|
||||
) -> ServeConfig:
|
||||
return load_config(ServeConfig, path, overrides)
|
||||
|
||||
|
||||
def load_raw_config(path: str | Path) -> dict[str, Any]:
|
||||
config_path = Path(path)
|
||||
if not config_path.exists():
|
||||
raise FileNotFoundError(f"Config file not found: {config_path}")
|
||||
|
||||
with config_path.open(encoding="utf-8") as handle:
|
||||
raw = _load_raw_mapping(handle, config_path)
|
||||
|
||||
if raw is None:
|
||||
return {}
|
||||
if not isinstance(raw, Mapping):
|
||||
raise ConfigValidationError("", f"{config_path} must contain a top-level mapping")
|
||||
return dict(raw)
|
||||
|
||||
|
||||
def _load_raw_mapping(handle: Any, config_path: Path) -> Any:
|
||||
suffix = config_path.suffix.lower()
|
||||
if suffix in {".yaml", ".yml"}:
|
||||
return yaml.safe_load(handle)
|
||||
if suffix == ".json":
|
||||
return json.load(handle)
|
||||
raise ValueError(f"Unsupported config file format: {config_path}")
|
||||
|
||||
|
||||
def _normalize_overrides(overrides: list[str] | Mapping[str, Any] | None, ) -> dict[str, Any] | None:
|
||||
if not overrides:
|
||||
return None
|
||||
if isinstance(overrides, list):
|
||||
return parse_cli_overrides(overrides)
|
||||
return dict(overrides)
|
||||
|
||||
|
||||
class _SchemaParser:
|
||||
|
||||
def parse_dataclass(
|
||||
self,
|
||||
config_type: type[T],
|
||||
raw: Mapping[str, Any],
|
||||
path: str,
|
||||
) -> T:
|
||||
if not isinstance(raw, Mapping):
|
||||
raise ConfigValidationError(path, f"expected mapping for {config_type.__name__}")
|
||||
|
||||
spec = _get_dataclass_spec(config_type)
|
||||
self._validate_keys(raw, spec, path)
|
||||
|
||||
values: dict[str, Any] = {}
|
||||
for name, field in spec.fields_by_name.items():
|
||||
field_path = _join_path(path, name)
|
||||
if name in raw:
|
||||
values[name] = self.parse_value(spec.type_hints[name], raw[name], field_path)
|
||||
continue
|
||||
if _field_is_required(field):
|
||||
raise ConfigValidationError(field_path, "missing required field")
|
||||
|
||||
return config_type(**values)
|
||||
|
||||
def parse_value(self, annotation: Any, value: Any, path: str) -> Any:
|
||||
if annotation is Any:
|
||||
return value
|
||||
|
||||
origin = get_origin(annotation)
|
||||
if origin in _UNION_ORIGINS:
|
||||
return self._parse_union(annotation, value, path)
|
||||
if origin is Literal:
|
||||
return self._parse_literal(annotation, value, path)
|
||||
if origin is list:
|
||||
return self._parse_list(annotation, value, path)
|
||||
if origin is dict:
|
||||
return self._parse_dict(annotation, value, path)
|
||||
if origin is tuple:
|
||||
return self._parse_tuple(annotation, value, path)
|
||||
if isinstance(annotation, type) and dataclasses.is_dataclass(annotation):
|
||||
return self.parse_dataclass(annotation, value, path)
|
||||
|
||||
scalar_parser = _SCALAR_PARSERS.get(annotation)
|
||||
if scalar_parser is not None:
|
||||
return scalar_parser(value, path)
|
||||
|
||||
return self._parse_instance(annotation, value, path)
|
||||
|
||||
def _validate_keys(
|
||||
self,
|
||||
raw: Mapping[str, Any],
|
||||
spec: _DataclassSpec,
|
||||
path: str,
|
||||
) -> None:
|
||||
for key in raw:
|
||||
if not isinstance(key, str):
|
||||
raise ConfigValidationError(path, "expected mapping keys to be strings")
|
||||
if key not in spec.fields_by_name:
|
||||
raise ConfigValidationError(_join_path(path, key), "unknown field")
|
||||
|
||||
def _parse_union(self, annotation: Any, value: Any, path: str) -> Any:
|
||||
candidates = [candidate for candidate in get_args(annotation) if candidate is not type(None)]
|
||||
if value is None and len(candidates) != len(get_args(annotation)):
|
||||
return None
|
||||
if len(candidates) == 1:
|
||||
return self.parse_value(candidates[0], value, path)
|
||||
|
||||
errors: list[str] = []
|
||||
for candidate in candidates:
|
||||
try:
|
||||
return self.parse_value(candidate, value, path)
|
||||
except ConfigValidationError as exc:
|
||||
errors.append(exc.message)
|
||||
|
||||
expected = ", ".join(_type_name(candidate) for candidate in candidates)
|
||||
detail = errors[0] if errors else f"expected one of ({expected})"
|
||||
raise ConfigValidationError(path, detail)
|
||||
|
||||
def _parse_literal(self, annotation: Any, value: Any, path: str) -> Any:
|
||||
allowed = get_args(annotation)
|
||||
if value not in allowed:
|
||||
raise ConfigValidationError(path, f"expected one of {sorted(allowed)!r}")
|
||||
return value
|
||||
|
||||
def _parse_list(self, annotation: Any, value: Any, path: str) -> list[Any]:
|
||||
if not isinstance(value, list):
|
||||
raise ConfigValidationError(path, "expected list")
|
||||
item_type = get_args(annotation)[0] if get_args(annotation) else Any
|
||||
return [self.parse_value(item_type, item, f"{path}[{index}]") for index, item in enumerate(value)]
|
||||
|
||||
def _parse_dict(self, annotation: Any, value: Any, path: str) -> dict[Any, Any]:
|
||||
if not isinstance(value, Mapping):
|
||||
raise ConfigValidationError(path, "expected mapping")
|
||||
|
||||
key_type, value_type = (get_args(annotation) + (Any, Any))[:2]
|
||||
parsed: dict[Any, Any] = {}
|
||||
for key, item in value.items():
|
||||
parsed_key = self._parse_dict_key(key_type, key, path)
|
||||
item_path = _join_path(path, str(key))
|
||||
parsed[parsed_key] = self.parse_value(value_type, item, item_path)
|
||||
return parsed
|
||||
|
||||
def _parse_tuple(self, annotation: Any, value: Any, path: str) -> tuple[Any, ...]:
|
||||
if not isinstance(value, list | tuple):
|
||||
raise ConfigValidationError(path, "expected tuple")
|
||||
|
||||
item_types = get_args(annotation)
|
||||
if len(item_types) == 2 and item_types[1] is Ellipsis:
|
||||
return tuple(self.parse_value(item_types[0], item, f"{path}[{index}]") for index, item in enumerate(value))
|
||||
|
||||
if len(value) != len(item_types):
|
||||
raise ConfigValidationError(path, f"expected tuple of length {len(item_types)}")
|
||||
|
||||
return tuple(
|
||||
self.parse_value(item_type, item, f"{path}[{index}]")
|
||||
for index, (item_type, item) in enumerate(zip(item_types, value, strict=True)))
|
||||
|
||||
def _parse_dict_key(self, annotation: Any, value: Any, path: str) -> Any:
|
||||
if annotation is Any:
|
||||
return value
|
||||
if annotation is str:
|
||||
if not isinstance(value, str):
|
||||
raise ConfigValidationError(path, "expected string dictionary keys")
|
||||
return value
|
||||
if annotation is int:
|
||||
if not isinstance(value, int) or isinstance(value, bool):
|
||||
raise ConfigValidationError(path, "expected integer dictionary keys")
|
||||
return value
|
||||
return value
|
||||
|
||||
def _parse_instance(self, annotation: Any, value: Any, path: str) -> Any:
|
||||
if isinstance(annotation, type) and not isinstance(value, annotation):
|
||||
raise ConfigValidationError(path, f"expected {annotation.__name__}")
|
||||
return value
|
||||
|
||||
|
||||
def _parse_bool(value: Any, path: str) -> bool:
|
||||
if type(value) is not bool:
|
||||
raise ConfigValidationError(path, "expected bool")
|
||||
return value
|
||||
|
||||
|
||||
def _parse_int(value: Any, path: str) -> int:
|
||||
if not isinstance(value, int) or isinstance(value, bool):
|
||||
raise ConfigValidationError(path, "expected int")
|
||||
return value
|
||||
|
||||
|
||||
def _parse_float(value: Any, path: str) -> float:
|
||||
if not isinstance(value, int | float) or isinstance(value, bool):
|
||||
raise ConfigValidationError(path, "expected float")
|
||||
return float(value)
|
||||
|
||||
|
||||
def _parse_str(value: Any, path: str) -> str:
|
||||
if not isinstance(value, str):
|
||||
raise ConfigValidationError(path, "expected str")
|
||||
return value
|
||||
|
||||
|
||||
_SCALAR_PARSERS: dict[Any, Any] = {
|
||||
bool: _parse_bool,
|
||||
int: _parse_int,
|
||||
float: _parse_float,
|
||||
str: _parse_str,
|
||||
}
|
||||
|
||||
|
||||
def _field_is_required(field: dataclasses.Field[Any]) -> bool:
|
||||
return (field.default is dataclasses.MISSING and field.default_factory is dataclasses.MISSING)
|
||||
|
||||
|
||||
def _get_dataclass_spec(config_type: type[Any]) -> _DataclassSpec:
|
||||
spec = _DATACLASS_SPEC_CACHE.get(config_type)
|
||||
if spec is not None:
|
||||
return spec
|
||||
|
||||
spec = _DataclassSpec(
|
||||
cls=config_type,
|
||||
type_hints=get_type_hints(config_type),
|
||||
fields_by_name={field.name: field
|
||||
for field in dataclasses.fields(config_type)},
|
||||
)
|
||||
_DATACLASS_SPEC_CACHE[config_type] = spec
|
||||
return spec
|
||||
|
||||
|
||||
_DATACLASS_SPEC_CACHE: dict[type[Any], _DataclassSpec] = {}
|
||||
|
||||
|
||||
def _join_path(prefix: str, suffix: str) -> str:
|
||||
if not prefix:
|
||||
return suffix
|
||||
return f"{prefix}.{suffix}"
|
||||
|
||||
|
||||
def _type_name(annotation: Any) -> str:
|
||||
origin = get_origin(annotation)
|
||||
if origin is not None:
|
||||
return str(annotation)
|
||||
if hasattr(annotation, "__name__"):
|
||||
return annotation.__name__
|
||||
return str(annotation)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"config_to_dict",
|
||||
"load_config",
|
||||
"load_raw_config",
|
||||
"load_run_config",
|
||||
"load_serve_config",
|
||||
"parse_config",
|
||||
]
|
||||
@@ -1,101 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
from collections.abc import Mapping
|
||||
|
||||
from fastvideo.api.schema import ContinuationState
|
||||
|
||||
|
||||
@dataclass
|
||||
class GenerationResult:
|
||||
prompt: str | None = None
|
||||
prompt_index: int | None = None
|
||||
samples: Any | None = None
|
||||
frames: Any | None = None
|
||||
audio: Any | None = None
|
||||
size: tuple[int, int, int] | None = None
|
||||
generation_time: float | None = None
|
||||
logging_info: Any | None = None
|
||||
trajectory: Any | None = None
|
||||
trajectory_timesteps: Any | None = None
|
||||
trajectory_decoded: Any | None = None
|
||||
video_path: str | None = None
|
||||
peak_memory_mb: float | None = None
|
||||
state: ContinuationState | None = None
|
||||
extra: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
@classmethod
|
||||
def from_legacy_result(
|
||||
cls,
|
||||
result: Mapping[str, Any],
|
||||
) -> GenerationResult:
|
||||
prompt = result.get("prompt")
|
||||
if prompt is None:
|
||||
prompt = result.get("prompts")
|
||||
|
||||
extra = {
|
||||
key: value
|
||||
for key, value in result.items() if key not in {
|
||||
"prompt",
|
||||
"prompt_index",
|
||||
"prompts",
|
||||
"samples",
|
||||
"frames",
|
||||
"audio",
|
||||
"size",
|
||||
"generation_time",
|
||||
"logging_info",
|
||||
"trajectory",
|
||||
"trajectory_timesteps",
|
||||
"trajectory_decoded",
|
||||
"video_path",
|
||||
"peak_memory_mb",
|
||||
"state",
|
||||
}
|
||||
}
|
||||
|
||||
return cls(
|
||||
prompt=prompt,
|
||||
prompt_index=result.get("prompt_index"),
|
||||
samples=result.get("samples"),
|
||||
frames=result.get("frames"),
|
||||
audio=result.get("audio"),
|
||||
size=result.get("size"),
|
||||
generation_time=result.get("generation_time"),
|
||||
logging_info=result.get("logging_info"),
|
||||
trajectory=result.get("trajectory"),
|
||||
trajectory_timesteps=result.get("trajectory_timesteps"),
|
||||
trajectory_decoded=result.get("trajectory_decoded"),
|
||||
video_path=result.get("video_path"),
|
||||
peak_memory_mb=result.get("peak_memory_mb"),
|
||||
state=result.get("state"),
|
||||
extra=extra,
|
||||
)
|
||||
|
||||
def to_legacy_dict(self) -> dict[str, Any]:
|
||||
result = {
|
||||
"prompts": self.prompt,
|
||||
"samples": self.samples,
|
||||
"frames": self.frames,
|
||||
"audio": self.audio,
|
||||
"size": self.size,
|
||||
"generation_time": self.generation_time,
|
||||
"logging_info": self.logging_info,
|
||||
"trajectory": self.trajectory,
|
||||
"trajectory_timesteps": self.trajectory_timesteps,
|
||||
"trajectory_decoded": self.trajectory_decoded,
|
||||
"video_path": self.video_path,
|
||||
"peak_memory_mb": self.peak_memory_mb,
|
||||
}
|
||||
if self.prompt_index is not None:
|
||||
result["prompt_index"] = self.prompt_index
|
||||
result["prompt"] = self.prompt
|
||||
if self.state is not None:
|
||||
result["state"] = self.state
|
||||
result.update(self.extra)
|
||||
return result
|
||||
|
||||
|
||||
__all__ = ["GenerationResult"]
|
||||
@@ -1,210 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Literal
|
||||
|
||||
|
||||
@dataclass
|
||||
class ServerConfig:
|
||||
host: str = "0.0.0.0"
|
||||
port: int = 8000
|
||||
output_dir: str = "outputs/"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParallelismConfig:
|
||||
tp_size: int = -1
|
||||
sp_size: int = -1
|
||||
hsdp_replicate_dim: int = 1
|
||||
hsdp_shard_dim: int = -1
|
||||
dist_timeout: int | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class OffloadConfig:
|
||||
dit: bool = True
|
||||
dit_layerwise: bool = True
|
||||
text_encoder: bool = True
|
||||
image_encoder: bool = True
|
||||
vae: bool = True
|
||||
pin_cpu_memory: bool = True
|
||||
|
||||
|
||||
@dataclass
|
||||
class CompileConfig:
|
||||
enabled: bool = False
|
||||
kwargs: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class QuantizationConfig:
|
||||
text_encoder_quant: str | None = None
|
||||
transformer_quant: str | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class EngineConfig:
|
||||
num_gpus: int = 1
|
||||
execution_backend: Literal["mp", "ray"] = "mp"
|
||||
parallelism: ParallelismConfig = field(default_factory=ParallelismConfig)
|
||||
offload: OffloadConfig = field(default_factory=OffloadConfig)
|
||||
compile: CompileConfig = field(default_factory=CompileConfig)
|
||||
enable_stage_verification: bool = True
|
||||
use_fsdp_inference: bool = False
|
||||
disable_autocast: bool = False
|
||||
quantization: QuantizationConfig | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class ComponentConfig:
|
||||
config_root: str | None = None
|
||||
pipeline_config_path: str | None = None
|
||||
text_encoder_weights: str | None = None
|
||||
transformer_weights: str | None = None
|
||||
transformer_2_weights: str | None = None
|
||||
vae_weights: str | None = None
|
||||
upsampler_weights: str | None = None
|
||||
lora_path: str | None = None
|
||||
override_pipeline_cls_name: str | None = None
|
||||
override_transformer_cls_name: str | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class PipelineSelection:
|
||||
workload_type: Literal["t2v", "i2v", "t2i", "i2i"] | None = None
|
||||
profile: str | None = None
|
||||
profile_version: str | None = None
|
||||
components: ComponentConfig = field(default_factory=ComponentConfig)
|
||||
profile_overrides: dict[str, Any] = field(default_factory=dict)
|
||||
experimental: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class GeneratorConfig:
|
||||
model_path: str
|
||||
revision: str | None = None
|
||||
trust_remote_code: bool = False
|
||||
engine: EngineConfig = field(default_factory=EngineConfig)
|
||||
pipeline: PipelineSelection = field(default_factory=PipelineSelection)
|
||||
|
||||
|
||||
@dataclass
|
||||
class InputConfig:
|
||||
prompt_path: str | None = None
|
||||
image_path: str | list[str] | None = None
|
||||
video_path: str | list[str] | None = None
|
||||
pil_image: Any | None = None
|
||||
pose: str | None = None
|
||||
mouse_cond: Any | None = None
|
||||
keyboard_cond: Any | None = None
|
||||
grid_sizes: Any | None = None
|
||||
c2ws_plucker_emb: Any | None = None
|
||||
refine_from: str | None = None
|
||||
stage1_video: Any | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class SamplingConfig:
|
||||
num_videos_per_prompt: int = 1
|
||||
seed: int = 1024
|
||||
num_frames: int = 125
|
||||
height: int = 720
|
||||
width: int = 1280
|
||||
height_sr: int = 1072
|
||||
width_sr: int = 1920
|
||||
fps: int = 24
|
||||
num_inference_steps: int = 50
|
||||
num_inference_steps_sr: int = 50
|
||||
guidance_scale: float = 1.0
|
||||
guidance_scale_2: float | None = None
|
||||
guidance_rescale: float = 0.0
|
||||
true_cfg_scale: float | None = None
|
||||
boundary_ratio: float | None = None
|
||||
sigmas: list[float] | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class RequestRuntimeConfig:
|
||||
enable_teacache: bool = False
|
||||
return_trajectory_latents: bool = False
|
||||
return_trajectory_decoded: bool = False
|
||||
|
||||
|
||||
@dataclass
|
||||
class OutputConfig:
|
||||
output_path: str = "outputs/"
|
||||
output_video_name: str | None = None
|
||||
save_video: bool = True
|
||||
return_frames: bool = True
|
||||
return_state: bool = False
|
||||
|
||||
|
||||
@dataclass
|
||||
class ContinuationState:
|
||||
kind: str
|
||||
payload: dict[str, Any]
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlannedStage:
|
||||
name: str
|
||||
kind: str
|
||||
source: str | None = None
|
||||
overrides: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class GenerationPlan:
|
||||
stages: list[PlannedStage]
|
||||
final_stage: str | None = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class GenerationRequest:
|
||||
prompt: str | list[str] | None = None
|
||||
negative_prompt: str | None = None
|
||||
inputs: InputConfig = field(default_factory=InputConfig)
|
||||
sampling: SamplingConfig = field(default_factory=SamplingConfig)
|
||||
runtime: RequestRuntimeConfig = field(default_factory=RequestRuntimeConfig)
|
||||
output: OutputConfig = field(default_factory=OutputConfig)
|
||||
stage_overrides: dict[str, Any] = field(default_factory=dict)
|
||||
state: ContinuationState | None = None
|
||||
plan: GenerationPlan | None = None
|
||||
extensions: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class RunConfig:
|
||||
generator: GeneratorConfig
|
||||
request: GenerationRequest
|
||||
|
||||
|
||||
@dataclass
|
||||
class ServeConfig:
|
||||
generator: GeneratorConfig
|
||||
server: ServerConfig = field(default_factory=ServerConfig)
|
||||
default_request: GenerationRequest = field(default_factory=GenerationRequest)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"CompileConfig",
|
||||
"ComponentConfig",
|
||||
"ContinuationState",
|
||||
"EngineConfig",
|
||||
"GenerationPlan",
|
||||
"GenerationRequest",
|
||||
"GeneratorConfig",
|
||||
"InputConfig",
|
||||
"OffloadConfig",
|
||||
"OutputConfig",
|
||||
"ParallelismConfig",
|
||||
"PipelineSelection",
|
||||
"PlannedStage",
|
||||
"QuantizationConfig",
|
||||
"RequestRuntimeConfig",
|
||||
"RunConfig",
|
||||
"SamplingConfig",
|
||||
"ServeConfig",
|
||||
"ServerConfig",
|
||||
]
|
||||
@@ -2,7 +2,7 @@
|
||||
# Adapted from vllm: https://github.com/vllm-project/vllm/blob/v0.7.3/vllm/attention/backends/abstract.py
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field, fields
|
||||
from dataclasses import dataclass, fields
|
||||
from typing import TYPE_CHECKING, Any, Generic, Protocol, TypeVar
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -53,10 +53,6 @@ class AttentionMetadata:
|
||||
"""Attention metadata for prefill and decode batched together."""
|
||||
# Current step of diffusion process
|
||||
current_timestep: int
|
||||
VSA_sparsity: float = field(default=0.0, kw_only=True)
|
||||
|
||||
def __getattr__(self, name: str) -> Any:
|
||||
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
|
||||
|
||||
def asdict_zerocopy(self, skip_fields: set[str] | None = None) -> dict[str, Any]:
|
||||
"""Similar to dataclasses.asdict, but avoids deepcopying."""
|
||||
@@ -86,7 +82,7 @@ class AttentionMetadataBuilder(ABC, Generic[T]):
|
||||
@abstractmethod
|
||||
def build(
|
||||
self,
|
||||
**kwargs: Any,
|
||||
**kwargs: dict[str, Any],
|
||||
) -> AttentionMetadata:
|
||||
"""Build attention metadata with on-device tensors."""
|
||||
raise NotImplementedError
|
||||
|
||||
@@ -1,121 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import importlib
|
||||
import sys
|
||||
from collections.abc import Callable
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
|
||||
from fastvideo.attention.backends.abstract import (
|
||||
AttentionBackend,
|
||||
AttentionImpl,
|
||||
AttentionMetadata,
|
||||
AttentionMetadataBuilder,
|
||||
)
|
||||
from fastvideo.logger import init_logger
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
_project_root = Path(__file__).resolve().parent.parent.parent.parent
|
||||
_kernel_root = _project_root / "fastvideo-kernel"
|
||||
_kernel_python_root = _kernel_root / "python"
|
||||
_attn_qat_infer: Callable[..., torch.Tensor] | None = None
|
||||
_attn_qat_infer_import_attempted = False
|
||||
|
||||
|
||||
def _ensure_kernel_paths() -> None:
|
||||
for path in (_project_root, _kernel_root, _kernel_python_root):
|
||||
path_str = str(path)
|
||||
if path_str not in sys.path:
|
||||
sys.path.insert(0, path_str)
|
||||
|
||||
|
||||
def _get_attn_qat_infer() -> Callable[..., torch.Tensor] | None:
|
||||
global _attn_qat_infer
|
||||
global _attn_qat_infer_import_attempted
|
||||
|
||||
if _attn_qat_infer_import_attempted:
|
||||
return _attn_qat_infer
|
||||
|
||||
_attn_qat_infer_import_attempted = True
|
||||
_ensure_kernel_paths()
|
||||
|
||||
try:
|
||||
# Prefer the in-repo kernel implementation during local development.
|
||||
_attn_qat_infer = importlib.import_module("attn_qat_infer").sageattn_blackwell
|
||||
except ImportError:
|
||||
_attn_qat_infer = None
|
||||
|
||||
return _attn_qat_infer
|
||||
|
||||
|
||||
def is_attn_qat_infer_available() -> bool:
|
||||
return _get_attn_qat_infer() is not None
|
||||
|
||||
|
||||
class AttnQatInferBackend(AttentionBackend):
|
||||
|
||||
accept_output_buffer: bool = True
|
||||
|
||||
@staticmethod
|
||||
def get_supported_head_sizes() -> list[int]:
|
||||
return [64, 128]
|
||||
|
||||
@staticmethod
|
||||
def get_name() -> str:
|
||||
return "ATTN_QAT_INFER"
|
||||
|
||||
@staticmethod
|
||||
def get_impl_cls() -> type["AttnQatInferImpl"]:
|
||||
return AttnQatInferImpl
|
||||
|
||||
@staticmethod
|
||||
def get_metadata_cls() -> type["AttentionMetadata"]:
|
||||
raise NotImplementedError
|
||||
|
||||
@staticmethod
|
||||
def get_builder_cls() -> type["AttentionMetadataBuilder"]:
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
class AttnQatInferImpl(AttentionImpl):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
num_heads: int,
|
||||
head_size: int,
|
||||
causal: bool,
|
||||
softmax_scale: float,
|
||||
num_kv_heads: int | None = None,
|
||||
prefix: str = "",
|
||||
**extra_impl_args,
|
||||
) -> None:
|
||||
self.causal = causal
|
||||
self.softmax_scale = softmax_scale
|
||||
self.dropout = extra_impl_args.get("dropout_p", 0.0)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
query: torch.Tensor,
|
||||
key: torch.Tensor,
|
||||
value: torch.Tensor,
|
||||
attn_metadata: AttentionMetadata,
|
||||
) -> torch.Tensor:
|
||||
attn_qat_infer = _get_attn_qat_infer()
|
||||
if attn_qat_infer is None:
|
||||
raise ImportError("attn_qat_infer is not available. Please ensure the "
|
||||
"attn_qat_infer kernel package is installed.")
|
||||
|
||||
query = query.transpose(1, 2).contiguous()
|
||||
key = key.transpose(1, 2).contiguous()
|
||||
value = value.transpose(1, 2).contiguous()
|
||||
|
||||
output = attn_qat_infer(
|
||||
query,
|
||||
key,
|
||||
value,
|
||||
attn_mask=None,
|
||||
is_causal=self.causal,
|
||||
)
|
||||
return output.transpose(1, 2).contiguous()
|
||||
@@ -1,145 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
import importlib
|
||||
import sys
|
||||
from collections.abc import Callable
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
|
||||
from fastvideo.attention.backends.abstract import (
|
||||
AttentionBackend,
|
||||
AttentionImpl,
|
||||
AttentionMetadata,
|
||||
AttentionMetadataBuilder,
|
||||
)
|
||||
from fastvideo.logger import init_logger
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
_project_root = Path(__file__).resolve().parent.parent.parent.parent
|
||||
_kernel_root = _project_root / "fastvideo-kernel"
|
||||
_kernel_python_root = _kernel_root / "python"
|
||||
_attn_qat_train_attention: Callable[..., torch.Tensor] | None = None
|
||||
_attn_qat_train_import_attempted = False
|
||||
|
||||
|
||||
def _ensure_kernel_paths() -> None:
|
||||
for path in (_project_root, _kernel_root, _kernel_python_root):
|
||||
path_str = str(path)
|
||||
if path_str not in sys.path:
|
||||
sys.path.insert(0, path_str)
|
||||
|
||||
|
||||
def _get_attn_qat_train_attention() -> Callable[..., torch.Tensor] | None:
|
||||
global _attn_qat_train_attention
|
||||
global _attn_qat_train_import_attempted
|
||||
|
||||
if _attn_qat_train_import_attempted:
|
||||
return _attn_qat_train_attention
|
||||
|
||||
_attn_qat_train_import_attempted = True
|
||||
_ensure_kernel_paths()
|
||||
|
||||
try:
|
||||
_attn_qat_train_attention = importlib.import_module("fastvideo_kernel.triton_kernels.attn_qat_train").attention
|
||||
except ImportError:
|
||||
_attn_qat_train_attention = None
|
||||
|
||||
return _attn_qat_train_attention
|
||||
|
||||
|
||||
def attn_qat_train(q_BLHD: torch.Tensor,
|
||||
k_BLHD: torch.Tensor,
|
||||
v_BLHD: torch.Tensor,
|
||||
is_causal: bool = False) -> torch.Tensor:
|
||||
attention = _get_attn_qat_train_attention()
|
||||
if attention is None:
|
||||
raise ImportError("fastvideo_kernel.triton_kernels.attn_qat_train is not available. "
|
||||
"Please ensure the FastVideo kernel package is installed.")
|
||||
|
||||
q_BHLD = q_BLHD.permute(0, 2, 1, 3).contiguous()
|
||||
k_BHLD = k_BLHD.permute(0, 2, 1, 3).contiguous()
|
||||
v_BHLD = v_BLHD.permute(0, 2, 1, 3).contiguous()
|
||||
|
||||
use_qat_qkv_backward = True
|
||||
smooth_k = False
|
||||
warp_specialize = True
|
||||
is_qat = True
|
||||
two_level_quant_p_sage3 = False
|
||||
fake_quant_p_bwd = True
|
||||
use_high_prec_o = True
|
||||
smooth_q = False
|
||||
sm_scale = 1.0 / (q_BHLD.shape[-1]**0.5)
|
||||
use_global_sf_qkv = False
|
||||
use_global_sf_p = False
|
||||
|
||||
o_BHLD = attention(
|
||||
q_BHLD,
|
||||
k_BHLD,
|
||||
v_BHLD,
|
||||
is_causal,
|
||||
sm_scale,
|
||||
use_qat_qkv_backward,
|
||||
smooth_k,
|
||||
warp_specialize,
|
||||
is_qat,
|
||||
two_level_quant_p_sage3,
|
||||
fake_quant_p_bwd,
|
||||
use_high_prec_o,
|
||||
smooth_q,
|
||||
use_global_sf_p,
|
||||
use_global_sf_qkv,
|
||||
)
|
||||
return o_BHLD.permute(0, 2, 1, 3).contiguous()
|
||||
|
||||
|
||||
class AttnQatTrainBackend(AttentionBackend):
|
||||
|
||||
accept_output_buffer: bool = True
|
||||
|
||||
@staticmethod
|
||||
def get_supported_head_sizes() -> list[int]:
|
||||
return [64, 96, 128, 160, 192, 224, 256]
|
||||
|
||||
@staticmethod
|
||||
def get_name() -> str:
|
||||
return "ATTN_QAT_TRAIN"
|
||||
|
||||
@staticmethod
|
||||
def get_impl_cls() -> type["AttnQatTrainImpl"]:
|
||||
return AttnQatTrainImpl
|
||||
|
||||
@staticmethod
|
||||
def get_metadata_cls() -> type["AttentionMetadata"]:
|
||||
raise NotImplementedError
|
||||
|
||||
@staticmethod
|
||||
def get_builder_cls() -> type["AttentionMetadataBuilder"]:
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
class AttnQatTrainImpl(AttentionImpl):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
num_heads: int,
|
||||
head_size: int,
|
||||
causal: bool,
|
||||
softmax_scale: float,
|
||||
num_kv_heads: int | None = None,
|
||||
prefix: str = "",
|
||||
**extra_impl_args,
|
||||
) -> None:
|
||||
self.causal = causal
|
||||
self.softmax_scale = softmax_scale
|
||||
self.dropout = extra_impl_args.get("dropout_p", 0.0)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
query: torch.Tensor,
|
||||
key: torch.Tensor,
|
||||
value: torch.Tensor,
|
||||
attn_metadata: AttentionMetadata,
|
||||
) -> torch.Tensor:
|
||||
return attn_qat_train(query, key, value, is_causal=self.causal)
|
||||
@@ -1,736 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""
|
||||
Bidirectional Sparse Attention (BSA) backend for FastVideo.
|
||||
|
||||
Pure-PyTorch reference implementation from:
|
||||
"Bidirectional Sparse Attention for Faster Video Diffusion Training"
|
||||
(arXiv:2509.01085)
|
||||
|
||||
BSA sparsifies both queries (pruning redundant tokens per block) and
|
||||
key-value pairs (keeping only relevant KV blocks per query block).
|
||||
|
||||
This is a training-free inference backend: it works with any model
|
||||
trained with full attention by applying BSA sparsity at inference time.
|
||||
"""
|
||||
|
||||
import functools
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
|
||||
from fastvideo.attention.backends.abstract import (
|
||||
AttentionBackend,
|
||||
AttentionImpl,
|
||||
AttentionMetadata,
|
||||
AttentionMetadataBuilder,
|
||||
)
|
||||
from fastvideo.distributed import get_sp_group
|
||||
from fastvideo.logger import init_logger
|
||||
|
||||
try:
|
||||
from fastvideo.attention.utils.flash_attn_no_pad import (
|
||||
flash_attn_varlen_func_impl, )
|
||||
|
||||
FLASH_ATTN_AVAILABLE = True
|
||||
except ImportError:
|
||||
flash_attn_varlen_func_impl = None
|
||||
FLASH_ATTN_AVAILABLE = False
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
BSA_TILE_SIZE = (4, 4, 4)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Cached index helpers (same pattern as VSA)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@functools.lru_cache(maxsize=10)
|
||||
def get_tile_partition_indices(
|
||||
dit_seq_shape: tuple[int, int, int],
|
||||
tile_size: tuple[int, int, int],
|
||||
device: torch.device,
|
||||
) -> torch.LongTensor:
|
||||
"""Map raster-order tokens to tile-contiguous order."""
|
||||
T, H, W = dit_seq_shape
|
||||
ts, hs, ws = tile_size
|
||||
indices = torch.arange(T * H * W, device=device, dtype=torch.long).reshape(T, H, W)
|
||||
ls = []
|
||||
for t in range(math.ceil(T / ts)):
|
||||
for h in range(math.ceil(H / hs)):
|
||||
for w in range(math.ceil(W / ws)):
|
||||
ls.append(indices[
|
||||
t * ts:min(t * ts + ts, T),
|
||||
h * hs:min(h * hs + hs, H),
|
||||
w * ws:min(w * ws + ws, W),
|
||||
].flatten())
|
||||
return torch.cat(ls, dim=0)
|
||||
|
||||
|
||||
@functools.lru_cache(maxsize=10)
|
||||
def get_reverse_tile_partition_indices(
|
||||
dit_seq_shape: tuple[int, int, int],
|
||||
tile_size: tuple[int, int, int],
|
||||
device: torch.device,
|
||||
) -> torch.LongTensor:
|
||||
"""Inverse mapping: tile-contiguous order back to raster order."""
|
||||
return torch.argsort(get_tile_partition_indices(dit_seq_shape, tile_size, device))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# BSA core operations
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _prune_queries(
|
||||
q_blocks: torch.Tensor,
|
||||
keep_ratio: float,
|
||||
) -> tuple[torch.Tensor, torch.Tensor, int]:
|
||||
"""
|
||||
Prune redundant query tokens within each block.
|
||||
|
||||
Scores tokens by cosine similarity to the block center.
|
||||
Keeps the LEAST similar (most informative) tokens.
|
||||
|
||||
Args:
|
||||
q_blocks: [B, N_heads, N_blocks, block_size, D]
|
||||
keep_ratio: fraction of tokens to keep
|
||||
|
||||
Returns:
|
||||
sparse_q: [B, N_heads, N_blocks, keep_size, D]
|
||||
keep_indices: [B, N_heads, N_blocks, keep_size]
|
||||
keep_size: int
|
||||
"""
|
||||
B, H, N, S, D = q_blocks.shape
|
||||
keep_size = max(1, int(S * keep_ratio))
|
||||
|
||||
if keep_size >= S:
|
||||
idx = torch.arange(S, device=q_blocks.device)
|
||||
idx = idx.view(1, 1, 1, S).expand(B, H, N, S)
|
||||
return q_blocks, idx, S
|
||||
|
||||
center_idx = S // 2
|
||||
center = q_blocks[:, :, :, center_idx:center_idx + 1, :]
|
||||
|
||||
q_norm = F.normalize(q_blocks, dim=-1)
|
||||
c_norm = F.normalize(center, dim=-1)
|
||||
similarity = (q_norm * c_norm).sum(dim=-1) # [B, H, N, S]
|
||||
|
||||
# lowest similarity = most distinctive = keep
|
||||
_, indices = similarity.topk(keep_size, dim=-1, largest=False)
|
||||
indices, _ = indices.sort(dim=-1)
|
||||
|
||||
idx_expand = indices.unsqueeze(-1).expand(-1, -1, -1, -1, D)
|
||||
sparse_q = torch.gather(q_blocks, 3, idx_expand)
|
||||
|
||||
return sparse_q, indices, keep_size
|
||||
|
||||
|
||||
def _select_kv_blocks(
|
||||
sparse_q: torch.Tensor,
|
||||
k_blocks: torch.Tensor,
|
||||
cumulative_threshold: float,
|
||||
min_kv_blocks: int,
|
||||
) -> torch.Tensor:
|
||||
"""
|
||||
Dynamically select KV blocks for each query block.
|
||||
|
||||
Mean-pools to block level, computes block attention scores,
|
||||
admits blocks in descending order until cumulative mass
|
||||
exceeds threshold.
|
||||
|
||||
Args:
|
||||
sparse_q: [B, H, N, Sq, D]
|
||||
k_blocks: [B, H, N, Sk, D]
|
||||
cumulative_threshold: e.g. 0.9
|
||||
min_kv_blocks: minimum blocks to keep
|
||||
|
||||
Returns:
|
||||
kv_mask: [B, H, N, N] boolean
|
||||
"""
|
||||
B, H, N, _, D = sparse_q.shape
|
||||
|
||||
q_repr = sparse_q.mean(dim=3)
|
||||
k_repr = k_blocks.mean(dim=3)
|
||||
|
||||
scores = torch.matmul(q_repr, k_repr.transpose(-1, -2)) / (D**0.5)
|
||||
block_attn = F.softmax(scores, dim=-1)
|
||||
|
||||
sorted_attn, sorted_idx = block_attn.sort(dim=-1, descending=True)
|
||||
cumsum = sorted_attn.cumsum(dim=-1)
|
||||
|
||||
keep_sorted = torch.ones_like(cumsum, dtype=torch.bool)
|
||||
keep_sorted[..., 1:] = cumsum[..., :-1] < cumulative_threshold
|
||||
|
||||
min_mask = torch.zeros_like(keep_sorted)
|
||||
min_mask[..., :min(min_kv_blocks, N)] = True
|
||||
keep_sorted = keep_sorted | min_mask
|
||||
|
||||
kv_mask = torch.zeros_like(block_attn, dtype=torch.bool)
|
||||
kv_mask.scatter_(-1, sorted_idx, keep_sorted)
|
||||
|
||||
return kv_mask
|
||||
|
||||
|
||||
def _compute_sparse_attention(
|
||||
sparse_q: torch.Tensor,
|
||||
k_blocks: torch.Tensor,
|
||||
v_blocks: torch.Tensor,
|
||||
kv_mask: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
"""
|
||||
Compute attention for each query block against selected KV blocks.
|
||||
|
||||
Handles per-batch and per-head KV masks correctly.
|
||||
Uses flash_attn_varlen_func when available on GPU.
|
||||
Falls back to pure-PyTorch reference on CPU.
|
||||
|
||||
Args:
|
||||
sparse_q: [B, H, N, Sq, D]
|
||||
k_blocks: [B, H, N, Sk, D]
|
||||
v_blocks: [B, H, N, Sk, D]
|
||||
kv_mask: [B, H, N, N] boolean (per-batch, per-head)
|
||||
|
||||
Returns:
|
||||
output: [B, H, N, Sq, D]
|
||||
"""
|
||||
if FLASH_ATTN_AVAILABLE and sparse_q.is_cuda:
|
||||
return _compute_sparse_attention_flash(sparse_q, k_blocks, v_blocks, kv_mask)
|
||||
else:
|
||||
return _compute_sparse_attention_reference(sparse_q, k_blocks, v_blocks, kv_mask)
|
||||
|
||||
|
||||
def _compute_sparse_attention_reference(
|
||||
sparse_q: torch.Tensor,
|
||||
k_blocks: torch.Tensor,
|
||||
v_blocks: torch.Tensor,
|
||||
kv_mask: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
"""Pure-PyTorch fallback with per-batch, per-head mask support."""
|
||||
B, H, N, Sq, D = sparse_q.shape
|
||||
output = torch.zeros_like(sparse_q)
|
||||
|
||||
for b in range(B):
|
||||
for h in range(H):
|
||||
for qb in range(N):
|
||||
selected = kv_mask[b, h, qb] # [N] boolean
|
||||
sel_idx = selected.nonzero(as_tuple=True)[0]
|
||||
|
||||
if sel_idx.shape[0] == 0:
|
||||
continue
|
||||
|
||||
# [num_sel * Sk, D]
|
||||
sel_k = k_blocks[b, h, sel_idx].reshape(-1, D)
|
||||
sel_v = v_blocks[b, h, sel_idx].reshape(-1, D)
|
||||
|
||||
q = sparse_q[b, h, qb] # [Sq, D]
|
||||
scores = torch.matmul(q, sel_k.transpose(-1, -2)) / (D**0.5)
|
||||
weights = F.softmax(scores, dim=-1)
|
||||
output[b, h, qb] = torch.matmul(weights, sel_v)
|
||||
|
||||
return output
|
||||
|
||||
|
||||
def _compute_sparse_attention_flash(
|
||||
sparse_q: torch.Tensor,
|
||||
k_blocks: torch.Tensor,
|
||||
v_blocks: torch.Tensor,
|
||||
kv_mask: torch.Tensor,
|
||||
) -> torch.Tensor:
|
||||
"""
|
||||
FlashAttention implementation with per-batch, per-head mask support.
|
||||
|
||||
Strategy: check if all heads share the same mask. If so, use a single
|
||||
FlashAttention call per batch (fast path). If not, process each head
|
||||
separately (correct path).
|
||||
|
||||
Args:
|
||||
sparse_q: [B, H, N, Sq, D]
|
||||
k_blocks: [B, H, N, Sk, D]
|
||||
v_blocks: [B, H, N, Sk, D]
|
||||
kv_mask: [B, H, N, N] boolean
|
||||
|
||||
Returns:
|
||||
output: [B, H, N, Sq, D]
|
||||
"""
|
||||
B, H, N, Sq, D = sparse_q.shape
|
||||
Sk = k_blocks.shape[3]
|
||||
device = sparse_q.device
|
||||
output = torch.zeros_like(sparse_q)
|
||||
|
||||
for b in range(B):
|
||||
# Check if all heads share the same mask for this batch element
|
||||
# Compare each head's mask to head 0's mask
|
||||
head0_mask = kv_mask[b, 0] # [N, N]
|
||||
all_heads_same = all(torch.equal(kv_mask[b, h], head0_mask) for h in range(1, H))
|
||||
|
||||
if all_heads_same:
|
||||
# Fast path: all heads share the same mask, single FA call
|
||||
_flash_attn_single_mask(
|
||||
sparse_q[b],
|
||||
k_blocks[b],
|
||||
v_blocks[b],
|
||||
head0_mask,
|
||||
output[b],
|
||||
H,
|
||||
N,
|
||||
Sq,
|
||||
Sk,
|
||||
D,
|
||||
device,
|
||||
)
|
||||
else:
|
||||
# Per-head path: process each head individually
|
||||
for h in range(H):
|
||||
head_mask = kv_mask[b, h] # [N, N]
|
||||
# Process single head: squeeze head dim, run FA, put back
|
||||
_flash_attn_single_head(
|
||||
sparse_q[b, h],
|
||||
k_blocks[b, h],
|
||||
v_blocks[b, h],
|
||||
head_mask,
|
||||
output,
|
||||
b,
|
||||
h,
|
||||
N,
|
||||
Sq,
|
||||
Sk,
|
||||
D,
|
||||
device,
|
||||
)
|
||||
|
||||
return output
|
||||
|
||||
|
||||
def _flash_attn_single_mask(
|
||||
sparse_q_b: torch.Tensor, # [H, N, Sq, D]
|
||||
k_blocks_b: torch.Tensor, # [H, N, Sk, D]
|
||||
v_blocks_b: torch.Tensor, # [H, N, Sk, D]
|
||||
mask: torch.Tensor, # [N, N] boolean
|
||||
output_b: torch.Tensor, # [H, N, Sq, D] (modified in-place)
|
||||
H: int,
|
||||
N: int,
|
||||
Sq: int,
|
||||
Sk: int,
|
||||
D: int,
|
||||
device: torch.device,
|
||||
) -> None:
|
||||
"""Run FlashAttention for all heads sharing the same KV mask."""
|
||||
q_list = []
|
||||
k_list = []
|
||||
v_list = []
|
||||
cu_seqlens_q = [0]
|
||||
cu_seqlens_k = [0]
|
||||
active_blocks = []
|
||||
|
||||
for qb in range(N):
|
||||
selected = mask[qb] # [N] boolean
|
||||
sel_idx = selected.nonzero(as_tuple=True)[0]
|
||||
|
||||
if sel_idx.shape[0] == 0:
|
||||
continue
|
||||
|
||||
active_blocks.append(qb)
|
||||
num_kv_tokens = sel_idx.shape[0] * Sk
|
||||
|
||||
# [H, Sq, D] -> [Sq, H, D]
|
||||
q_block = sparse_q_b[:, qb].permute(1, 0, 2)
|
||||
q_list.append(q_block)
|
||||
|
||||
# [H, num_sel, Sk, D] -> [num_kv_tokens, H, D]
|
||||
sel_k = k_blocks_b[:, sel_idx].permute(1, 2, 0, 3).reshape(num_kv_tokens, H, D)
|
||||
sel_v = v_blocks_b[:, sel_idx].permute(1, 2, 0, 3).reshape(num_kv_tokens, H, D)
|
||||
k_list.append(sel_k)
|
||||
v_list.append(sel_v)
|
||||
|
||||
cu_seqlens_q.append(cu_seqlens_q[-1] + Sq)
|
||||
cu_seqlens_k.append(cu_seqlens_k[-1] + num_kv_tokens)
|
||||
|
||||
if not q_list:
|
||||
return
|
||||
|
||||
flat_q = torch.cat(q_list, dim=0)
|
||||
flat_k = torch.cat(k_list, dim=0)
|
||||
flat_v = torch.cat(v_list, dim=0)
|
||||
|
||||
cu_seqlens_q_t = torch.tensor(cu_seqlens_q, dtype=torch.int32, device=device)
|
||||
cu_seqlens_k_t = torch.tensor(cu_seqlens_k, dtype=torch.int32, device=device)
|
||||
|
||||
max_seqlen_q = Sq
|
||||
max_seqlen_k = int((cu_seqlens_k_t[1:] - cu_seqlens_k_t[:-1]).max().item())
|
||||
|
||||
orig_dtype = flat_q.dtype
|
||||
compute_dtype = orig_dtype
|
||||
if compute_dtype not in (torch.float16, torch.bfloat16):
|
||||
compute_dtype = torch.bfloat16
|
||||
flat_q = flat_q.to(compute_dtype)
|
||||
flat_k = flat_k.to(compute_dtype)
|
||||
flat_v = flat_v.to(compute_dtype)
|
||||
|
||||
flat_out = flash_attn_varlen_func_impl(
|
||||
flat_q,
|
||||
flat_k,
|
||||
flat_v,
|
||||
cu_seqlens_q_t,
|
||||
cu_seqlens_k_t,
|
||||
max_seqlen_q,
|
||||
max_seqlen_k,
|
||||
causal=False,
|
||||
)
|
||||
|
||||
if compute_dtype != orig_dtype:
|
||||
flat_out = flat_out.to(orig_dtype)
|
||||
|
||||
idx = 0
|
||||
for qb in active_blocks:
|
||||
block_out = flat_out[idx:idx + Sq] # [Sq, H, D]
|
||||
output_b[:, qb] = block_out.permute(1, 0, 2) # [H, Sq, D]
|
||||
idx += Sq
|
||||
|
||||
|
||||
def _flash_attn_single_head(
|
||||
sparse_q_bh: torch.Tensor, # [N, Sq, D]
|
||||
k_blocks_bh: torch.Tensor, # [N, Sk, D]
|
||||
v_blocks_bh: torch.Tensor, # [N, Sk, D]
|
||||
mask: torch.Tensor, # [N, N] boolean
|
||||
output: torch.Tensor, # [B, H, N, Sq, D] (modified in-place)
|
||||
b: int,
|
||||
h: int,
|
||||
N: int,
|
||||
Sq: int,
|
||||
Sk: int,
|
||||
D: int,
|
||||
device: torch.device,
|
||||
) -> None:
|
||||
"""Run FlashAttention for a single head with its own KV mask."""
|
||||
q_list = []
|
||||
k_list = []
|
||||
v_list = []
|
||||
cu_seqlens_q = [0]
|
||||
cu_seqlens_k = [0]
|
||||
active_blocks = []
|
||||
|
||||
for qb in range(N):
|
||||
selected = mask[qb]
|
||||
sel_idx = selected.nonzero(as_tuple=True)[0]
|
||||
|
||||
if sel_idx.shape[0] == 0:
|
||||
continue
|
||||
|
||||
active_blocks.append(qb)
|
||||
num_kv_tokens = sel_idx.shape[0] * Sk
|
||||
|
||||
# [Sq, D] -> [Sq, 1, D] (single head)
|
||||
q_block = sparse_q_bh[qb].unsqueeze(1)
|
||||
q_list.append(q_block)
|
||||
|
||||
# [num_sel, Sk, D] -> [num_kv_tokens, 1, D]
|
||||
sel_k = k_blocks_bh[sel_idx].reshape(num_kv_tokens, 1, D)
|
||||
sel_v = v_blocks_bh[sel_idx].reshape(num_kv_tokens, 1, D)
|
||||
k_list.append(sel_k)
|
||||
v_list.append(sel_v)
|
||||
|
||||
cu_seqlens_q.append(cu_seqlens_q[-1] + Sq)
|
||||
cu_seqlens_k.append(cu_seqlens_k[-1] + num_kv_tokens)
|
||||
|
||||
if not q_list:
|
||||
return
|
||||
|
||||
flat_q = torch.cat(q_list, dim=0)
|
||||
flat_k = torch.cat(k_list, dim=0)
|
||||
flat_v = torch.cat(v_list, dim=0)
|
||||
|
||||
cu_seqlens_q_t = torch.tensor(cu_seqlens_q, dtype=torch.int32, device=device)
|
||||
cu_seqlens_k_t = torch.tensor(cu_seqlens_k, dtype=torch.int32, device=device)
|
||||
|
||||
max_seqlen_q = Sq
|
||||
max_seqlen_k = int((cu_seqlens_k_t[1:] - cu_seqlens_k_t[:-1]).max().item())
|
||||
|
||||
orig_dtype = flat_q.dtype
|
||||
compute_dtype = orig_dtype
|
||||
if compute_dtype not in (torch.float16, torch.bfloat16):
|
||||
compute_dtype = torch.bfloat16
|
||||
flat_q = flat_q.to(compute_dtype)
|
||||
flat_k = flat_k.to(compute_dtype)
|
||||
flat_v = flat_v.to(compute_dtype)
|
||||
|
||||
flat_out = flash_attn_varlen_func_impl(
|
||||
flat_q,
|
||||
flat_k,
|
||||
flat_v,
|
||||
cu_seqlens_q_t,
|
||||
cu_seqlens_k_t,
|
||||
max_seqlen_q,
|
||||
max_seqlen_k,
|
||||
causal=False,
|
||||
)
|
||||
|
||||
if compute_dtype != orig_dtype:
|
||||
flat_out = flat_out.to(orig_dtype)
|
||||
|
||||
idx = 0
|
||||
for qb in active_blocks:
|
||||
block_out = flat_out[idx:idx + Sq] # [Sq, 1, D]
|
||||
output[b, h, qb] = block_out.squeeze(1) # [Sq, D]
|
||||
idx += Sq
|
||||
|
||||
|
||||
def _reconstruct_pruned(
|
||||
sparse_output: torch.Tensor,
|
||||
keep_indices: torch.Tensor,
|
||||
block_size: int,
|
||||
) -> torch.Tensor:
|
||||
"""
|
||||
Scatter sparse output back to full block size.
|
||||
Pruned positions get nearest kept token's output.
|
||||
|
||||
Handles per-batch, per-head indices correctly.
|
||||
|
||||
Args:
|
||||
sparse_output: [B, H, N, keep_size, D]
|
||||
keep_indices: [B, H, N, keep_size]
|
||||
block_size: original tokens per block
|
||||
|
||||
Returns:
|
||||
full_output: [B, H, N, block_size, D]
|
||||
"""
|
||||
B, H, N, keep_size, D = sparse_output.shape
|
||||
device = sparse_output.device
|
||||
|
||||
if keep_size >= block_size:
|
||||
return sparse_output
|
||||
|
||||
full_output = torch.zeros(B, H, N, block_size, D, device=device, dtype=sparse_output.dtype)
|
||||
|
||||
# Scatter kept tokens
|
||||
idx_expand = keep_indices.unsqueeze(-1).expand(-1, -1, -1, -1, D)
|
||||
full_output.scatter_(3, idx_expand, sparse_output)
|
||||
|
||||
# Fill pruned positions with nearest kept token (vectorized)
|
||||
all_pos = torch.arange(block_size, device=device)
|
||||
|
||||
for b in range(B):
|
||||
for h in range(H):
|
||||
for n in range(N):
|
||||
kept = keep_indices[b, h, n] # [keep_size]
|
||||
|
||||
# Distance from every position to every kept position
|
||||
dists = (all_pos.view(-1, 1) - kept.view(1, -1)).abs()
|
||||
nearest_local_idx = dists.argmin(dim=1) # [block_size]
|
||||
|
||||
# Identify pruned positions
|
||||
is_pruned = torch.ones(block_size, dtype=torch.bool, device=device)
|
||||
is_pruned[kept] = False
|
||||
pruned_indices = is_pruned.nonzero(as_tuple=True)[0]
|
||||
|
||||
if pruned_indices.numel() > 0:
|
||||
src_indices = nearest_local_idx[pruned_indices]
|
||||
full_output[b, h, n, pruned_indices] = sparse_output[b, h, n, src_indices]
|
||||
|
||||
return full_output
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FastVideo backend classes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class BSAAttentionBackend(AttentionBackend):
|
||||
|
||||
accept_output_buffer: bool = False
|
||||
|
||||
@staticmethod
|
||||
def get_supported_head_sizes() -> list[int]:
|
||||
return [64, 128]
|
||||
|
||||
@staticmethod
|
||||
def get_name() -> str:
|
||||
return "BSA_ATTN"
|
||||
|
||||
@staticmethod
|
||||
def get_impl_cls() -> type["BSAAttentionImpl"]:
|
||||
return BSAAttentionImpl
|
||||
|
||||
@staticmethod
|
||||
def get_metadata_cls() -> type["BSAAttentionMetadata"]:
|
||||
return BSAAttentionMetadata
|
||||
|
||||
@staticmethod
|
||||
def get_builder_cls() -> type["BSAAttentionMetadataBuilder"]:
|
||||
return BSAAttentionMetadataBuilder
|
||||
|
||||
|
||||
@dataclass
|
||||
class BSAAttentionMetadata(AttentionMetadata):
|
||||
current_timestep: int
|
||||
dit_seq_shape: tuple[int, int, int]
|
||||
total_seq_length: int
|
||||
num_blocks: int
|
||||
block_size: int
|
||||
tile_partition_indices: torch.LongTensor
|
||||
reverse_tile_partition_indices: torch.LongTensor
|
||||
# BSA-specific config
|
||||
query_keep_ratio: float
|
||||
kv_cumulative_threshold: float
|
||||
min_kv_blocks: int
|
||||
|
||||
|
||||
class BSAAttentionMetadataBuilder(AttentionMetadataBuilder):
|
||||
|
||||
def __init__(self):
|
||||
pass
|
||||
|
||||
def prepare(self):
|
||||
pass
|
||||
|
||||
def build(
|
||||
self,
|
||||
current_timestep: int,
|
||||
raw_latent_shape: tuple[int, int, int],
|
||||
patch_size: tuple[int, int, int],
|
||||
device: torch.device,
|
||||
bsa_query_keep_ratio: float = 0.5,
|
||||
bsa_kv_cumulative_threshold: float = 0.9,
|
||||
bsa_min_kv_blocks: int = 4,
|
||||
**kwargs: dict[str, Any],
|
||||
) -> "BSAAttentionMetadata":
|
||||
# Ensure patching does not drop tokens silently.
|
||||
assert all(r % p == 0 for r, p in zip(raw_latent_shape, patch_size, strict=False)), (
|
||||
"raw_latent_shape must be divisible by patch_size for BSA", )
|
||||
|
||||
dit_seq_shape = (
|
||||
raw_latent_shape[0] // patch_size[0],
|
||||
raw_latent_shape[1] // patch_size[1],
|
||||
raw_latent_shape[2] // patch_size[2],
|
||||
)
|
||||
|
||||
total_seq_length = math.prod(dit_seq_shape)
|
||||
block_size = math.prod(BSA_TILE_SIZE)
|
||||
# Require exact tiling to avoid reshape failures later.
|
||||
assert all(d % t == 0 for d, t in zip(dit_seq_shape, BSA_TILE_SIZE, strict=False)), (
|
||||
"dit_seq_shape must be divisible by BSA_TILE_SIZE", )
|
||||
num_blocks = total_seq_length // block_size
|
||||
|
||||
tile_partition_indices = get_tile_partition_indices(dit_seq_shape, BSA_TILE_SIZE, device)
|
||||
reverse_tile_partition_indices = get_reverse_tile_partition_indices(dit_seq_shape, BSA_TILE_SIZE, device)
|
||||
|
||||
return BSAAttentionMetadata(
|
||||
current_timestep=current_timestep,
|
||||
dit_seq_shape=dit_seq_shape,
|
||||
total_seq_length=total_seq_length,
|
||||
num_blocks=num_blocks,
|
||||
block_size=block_size,
|
||||
tile_partition_indices=tile_partition_indices,
|
||||
reverse_tile_partition_indices=reverse_tile_partition_indices,
|
||||
query_keep_ratio=bsa_query_keep_ratio,
|
||||
kv_cumulative_threshold=bsa_kv_cumulative_threshold,
|
||||
min_kv_blocks=bsa_min_kv_blocks,
|
||||
)
|
||||
|
||||
|
||||
class BSAAttentionImpl(AttentionImpl):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
num_heads: int,
|
||||
head_size: int,
|
||||
causal: bool,
|
||||
softmax_scale: float,
|
||||
num_kv_heads: int | None = None,
|
||||
prefix: str = "",
|
||||
**extra_impl_args,
|
||||
) -> None:
|
||||
self.prefix = prefix
|
||||
self.num_heads = num_heads
|
||||
self.head_size = head_size
|
||||
if num_kv_heads is not None and num_kv_heads != num_heads:
|
||||
raise ValueError("BSA backend does not support grouped-query attention")
|
||||
if causal:
|
||||
raise ValueError("BSA backend is bidirectional; causal=True is unsupported")
|
||||
if softmax_scale is not None:
|
||||
expected_scale = 1.0 / math.sqrt(self.head_size)
|
||||
if not math.isclose(softmax_scale, expected_scale, rel_tol=1e-4, abs_tol=1e-5):
|
||||
raise ValueError("softmax_scale must be default (1/sqrt(d)) for BSA")
|
||||
try:
|
||||
sp_group = get_sp_group()
|
||||
self.sp_size = sp_group.world_size
|
||||
except (AssertionError, RuntimeError):
|
||||
self.sp_size = 1
|
||||
|
||||
def preprocess_qkv(
|
||||
self,
|
||||
qkv: torch.Tensor,
|
||||
attn_metadata: BSAAttentionMetadata,
|
||||
) -> torch.Tensor:
|
||||
"""Reorder tokens from raster order to tile-contiguous order."""
|
||||
# qkv: [B, L, num_heads, D]
|
||||
return qkv[:, attn_metadata.tile_partition_indices]
|
||||
|
||||
def postprocess_output(
|
||||
self,
|
||||
output: torch.Tensor,
|
||||
attn_metadata: BSAAttentionMetadata,
|
||||
) -> torch.Tensor:
|
||||
"""Reorder tokens from tile-contiguous order back to raster order."""
|
||||
return output[:, attn_metadata.reverse_tile_partition_indices]
|
||||
|
||||
def forward(
|
||||
self,
|
||||
query: torch.Tensor,
|
||||
key: torch.Tensor,
|
||||
value: torch.Tensor,
|
||||
attn_metadata: BSAAttentionMetadata,
|
||||
) -> torch.Tensor:
|
||||
"""
|
||||
BSA attention forward pass.
|
||||
|
||||
Input tensors are already in tile-contiguous order from preprocess_qkv.
|
||||
|
||||
Args:
|
||||
query: [B, L, num_heads, D] (tile-ordered)
|
||||
key: [B, L, num_heads, D] (tile-ordered)
|
||||
value: [B, L, num_heads, D] (tile-ordered)
|
||||
attn_metadata: BSA metadata
|
||||
|
||||
Returns:
|
||||
output: [B, L, num_heads, D] (tile-ordered)
|
||||
"""
|
||||
B, L, H, D = query.shape
|
||||
block_size = attn_metadata.block_size
|
||||
num_blocks = attn_metadata.num_blocks
|
||||
assert num_blocks * block_size == L, "Sequence length must match tiling"
|
||||
|
||||
# Reshape to [B, H, L, D] for attention computation
|
||||
q = query.transpose(1, 2).contiguous() # [B, H, L, D]
|
||||
k = key.transpose(1, 2).contiguous()
|
||||
v = value.transpose(1, 2).contiguous()
|
||||
|
||||
# Reshape into blocks: [B, H, num_blocks, block_size, D]
|
||||
q_blocks = q.view(B, H, num_blocks, block_size, D)
|
||||
k_blocks = k.view(B, H, num_blocks, block_size, D)
|
||||
v_blocks = v.view(B, H, num_blocks, block_size, D)
|
||||
|
||||
# --- Query sparsification ---
|
||||
sparse_q, keep_indices, keep_size = _prune_queries(q_blocks, attn_metadata.query_keep_ratio)
|
||||
|
||||
# --- KV block selection ---
|
||||
kv_mask = _select_kv_blocks(
|
||||
sparse_q,
|
||||
k_blocks,
|
||||
attn_metadata.kv_cumulative_threshold,
|
||||
attn_metadata.min_kv_blocks,
|
||||
)
|
||||
|
||||
# --- Sparse attention ---
|
||||
sparse_output = _compute_sparse_attention(sparse_q, k_blocks, v_blocks, kv_mask)
|
||||
|
||||
# --- Reconstruct pruned positions ---
|
||||
full_output = _reconstruct_pruned(sparse_output, keep_indices, block_size)
|
||||
|
||||
# Reshape back: [B, H, num_blocks, block_size, D] -> [B, H, L, D] -> [B, L, H, D]
|
||||
hidden_states = full_output.view(B, H, L, D).transpose(1, 2)
|
||||
|
||||
return hidden_states
|
||||
@@ -16,7 +16,7 @@ class SageAttention3Backend(AttentionBackend):
|
||||
|
||||
@staticmethod
|
||||
def get_supported_head_sizes() -> list[int]:
|
||||
return [64, 128]
|
||||
return [64, 128, 256]
|
||||
|
||||
@staticmethod
|
||||
def get_name() -> str:
|
||||
|
||||
@@ -133,6 +133,7 @@ class VideoSparseAttentionBackend(AttentionBackend):
|
||||
class VideoSparseAttentionMetadata(AttentionMetadata):
|
||||
current_timestep: int
|
||||
dit_seq_shape: list[int]
|
||||
VSA_sparsity: float
|
||||
num_tiles: list[int]
|
||||
total_seq_length: int
|
||||
tile_partition_indices: torch.LongTensor
|
||||
@@ -143,10 +144,10 @@ class VideoSparseAttentionMetadata(AttentionMetadata):
|
||||
|
||||
class VideoSparseAttentionMetadataBuilder(AttentionMetadataBuilder):
|
||||
|
||||
def __init__(self) -> None:
|
||||
def __init__(self):
|
||||
pass
|
||||
|
||||
def prepare(self) -> None:
|
||||
def prepare(self):
|
||||
pass
|
||||
|
||||
def build( # type: ignore
|
||||
|
||||
@@ -61,10 +61,10 @@ class VideoMobaAttentionMetadata(AttentionMetadata):
|
||||
|
||||
class VideoMobaAttentionMetadataBuilder(AttentionMetadataBuilder):
|
||||
|
||||
def __init__(self) -> None:
|
||||
def __init__(self):
|
||||
pass
|
||||
|
||||
def prepare(self) -> None:
|
||||
def prepare(self):
|
||||
pass
|
||||
|
||||
def build( # type: ignore
|
||||
@@ -81,7 +81,7 @@ class VideoMobaAttentionMetadataBuilder(AttentionMetadataBuilder):
|
||||
moba_select_mode: str = 'threshold',
|
||||
moba_threshold: float = 0.25,
|
||||
moba_threshold_type: str = 'query_head',
|
||||
device: torch.device | None = None,
|
||||
device: torch.device = None,
|
||||
first_full_layer: int = 0,
|
||||
first_full_step: int = 12,
|
||||
temporal_layer: int = 1,
|
||||
@@ -142,7 +142,7 @@ class VMOBAAttentionImpl(AttentionImpl):
|
||||
query: torch.Tensor,
|
||||
key: torch.Tensor,
|
||||
value: torch.Tensor,
|
||||
attn_metadata: VideoMobaAttentionMetadata,
|
||||
attn_metadata: AttentionMetadata,
|
||||
) -> torch.Tensor:
|
||||
"""
|
||||
query: [B, L, H, D]
|
||||
@@ -154,9 +154,7 @@ class VMOBAAttentionImpl(AttentionImpl):
|
||||
|
||||
# select chunk type according to layer idx:
|
||||
loop_layer_num = attn_metadata.temporal_layer + attn_metadata.spatial_layer + attn_metadata.st_layer
|
||||
assert self.layer_idx is not None, "VMoBA attention requires layer_idx to be set"
|
||||
moba_layer = self.layer_idx - attn_metadata.first_full_layer
|
||||
moba_chunk_size: int | tuple[int, int] | tuple[int, int, int]
|
||||
if moba_layer % loop_layer_num < attn_metadata.temporal_layer:
|
||||
moba_chunk_size = attn_metadata.temporal_chunk_size
|
||||
moba_topk = attn_metadata.temporal_topk
|
||||
@@ -166,8 +164,6 @@ class VMOBAAttentionImpl(AttentionImpl):
|
||||
elif moba_layer % loop_layer_num < attn_metadata.temporal_layer + attn_metadata.spatial_layer + attn_metadata.st_layer:
|
||||
moba_chunk_size = attn_metadata.st_chunk_size
|
||||
moba_topk = attn_metadata.st_topk
|
||||
else:
|
||||
raise ValueError(f"Invalid MoBA layer selection for layer {moba_layer}")
|
||||
|
||||
query, chunk_size = process_moba_input(query, attn_metadata.patch_resolution, moba_chunk_size)
|
||||
key, chunk_size = process_moba_input(key, attn_metadata.patch_resolution, moba_chunk_size)
|
||||
|
||||
@@ -158,7 +158,6 @@ class DistributedAttention_VSA(DistributedAttention):
|
||||
replicated_v: torch.Tensor | None = None,
|
||||
gate_compress: torch.Tensor | None = None,
|
||||
freqs_cis: tuple[torch.Tensor, torch.Tensor] | None = None,
|
||||
attention_mask: torch.Tensor | None = None,
|
||||
) -> tuple[torch.Tensor, torch.Tensor | None]:
|
||||
"""Forward pass for distributed attention.
|
||||
|
||||
@@ -171,7 +170,6 @@ class DistributedAttention_VSA(DistributedAttention):
|
||||
replicated_q (Optional[torch.Tensor]): Replicated query tensor, typically for text tokens
|
||||
replicated_k (Optional[torch.Tensor]): Replicated key tensor
|
||||
replicated_v (Optional[torch.Tensor]): Replicated value tensor
|
||||
attention_mask (Optional[torch.Tensor]): Attention mask [batch_size, seq_len]
|
||||
|
||||
Returns:
|
||||
Tuple[torch.Tensor, Optional[torch.Tensor]]: A tuple containing:
|
||||
|
||||
@@ -85,26 +85,7 @@ def get_attn_backend(
|
||||
supported_attention_backends: tuple[AttentionBackendEnum, ...]
|
||||
| None = None,
|
||||
) -> type[AttentionBackend]:
|
||||
selected_backend, is_forced = _resolve_backend_override()
|
||||
return _cached_get_attn_backend(
|
||||
head_size,
|
||||
dtype,
|
||||
supported_attention_backends,
|
||||
selected_backend,
|
||||
is_forced,
|
||||
)
|
||||
|
||||
|
||||
def _resolve_backend_override() -> tuple[AttentionBackendEnum | None, bool]:
|
||||
backend_by_global_setting = get_global_forced_attn_backend()
|
||||
if backend_by_global_setting is not None:
|
||||
return backend_by_global_setting, True
|
||||
|
||||
backend_by_env_var: str | None = envs.FASTVIDEO_ATTENTION_BACKEND
|
||||
if backend_by_env_var is not None:
|
||||
return backend_name_to_enum(backend_by_env_var), False
|
||||
|
||||
return None, False
|
||||
return _cached_get_attn_backend(head_size, dtype, supported_attention_backends)
|
||||
|
||||
|
||||
@cache
|
||||
@@ -113,16 +94,28 @@ def _cached_get_attn_backend(
|
||||
dtype: torch.dtype,
|
||||
supported_attention_backends: tuple[AttentionBackendEnum, ...]
|
||||
| None = None,
|
||||
selected_backend: AttentionBackendEnum | None = None,
|
||||
is_forced_backend: bool = False,
|
||||
) -> type[AttentionBackend]:
|
||||
# Check whether a particular choice of backend was
|
||||
# previously forced.
|
||||
#
|
||||
# THIS SELECTION OVERRIDES THE FASTVIDEO_ATTENTION_BACKEND
|
||||
# ENVIRONMENT VARIABLE.
|
||||
if not supported_attention_backends:
|
||||
raise ValueError("supported_attention_backends is empty")
|
||||
selected_backend = None
|
||||
backend_by_global_setting: AttentionBackendEnum | None = (get_global_forced_attn_backend())
|
||||
if backend_by_global_setting is not None:
|
||||
selected_backend = backend_by_global_setting
|
||||
else:
|
||||
# Check the environment variable and override if specified
|
||||
backend_by_env_var: str | None = envs.FASTVIDEO_ATTENTION_BACKEND
|
||||
if backend_by_env_var is not None:
|
||||
selected_backend = backend_name_to_enum(backend_by_env_var)
|
||||
|
||||
# get device-specific attn_backend
|
||||
from fastvideo.platforms import current_platform
|
||||
|
||||
if not is_forced_backend and selected_backend not in supported_attention_backends:
|
||||
if selected_backend not in supported_attention_backends:
|
||||
selected_backend = None
|
||||
attention_cls = current_platform.get_attn_backend_cls(selected_backend, head_size, dtype)
|
||||
if not attention_cls:
|
||||
|
||||
@@ -14,44 +14,30 @@
|
||||
# of rights and permissions under this agreement.
|
||||
# See the License for the specific language governing permissions and limitations under the License.
|
||||
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
from einops import rearrange
|
||||
from flash_attn import flash_attn_varlen_qkvpacked_func
|
||||
from flash_attn.bert_padding import pad_input, unpad_input
|
||||
from flash_attn import flash_attn_varlen_qkvpacked_func
|
||||
|
||||
|
||||
def _resolve_flash_attn_varlen_func() -> Any:
|
||||
try:
|
||||
from fastvideo.attention.utils.flash_attn_cute import (
|
||||
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
|
||||
except ImportError:
|
||||
try:
|
||||
from fastvideo.attention.utils.flash_attn_cute import (
|
||||
flash_attn_varlen_func as flash_attn_varlen_func_cute, )
|
||||
|
||||
return flash_attn_varlen_func_cute
|
||||
from flash_attn_interface import (
|
||||
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
|
||||
except ImportError:
|
||||
try:
|
||||
from flash_attn_interface import (
|
||||
flash_attn_varlen_func as flash_attn_varlen_func_interface, )
|
||||
|
||||
return flash_attn_varlen_func_interface
|
||||
except ImportError:
|
||||
from flash_attn import (
|
||||
flash_attn_varlen_func as flash_attn_varlen_func_flash, )
|
||||
|
||||
return flash_attn_varlen_func_flash
|
||||
|
||||
|
||||
flash_attn_varlen_func_impl = _resolve_flash_attn_varlen_func()
|
||||
from flash_attn import (
|
||||
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
|
||||
|
||||
|
||||
def flash_attn_no_pad(
|
||||
qkv: torch.Tensor,
|
||||
key_padding_mask: torch.Tensor,
|
||||
causal: bool = False,
|
||||
dropout_p: float = 0.0,
|
||||
softmax_scale: float | None = None,
|
||||
deterministic: bool = False,
|
||||
) -> torch.Tensor:
|
||||
qkv,
|
||||
key_padding_mask,
|
||||
causal=False,
|
||||
dropout_p=0.0,
|
||||
softmax_scale=None,
|
||||
deterministic=False,
|
||||
):
|
||||
batch_size = qkv.shape[0]
|
||||
seqlen = qkv.shape[1]
|
||||
nheads = qkv.shape[-2]
|
||||
@@ -82,13 +68,13 @@ def flash_attn_no_pad(
|
||||
|
||||
|
||||
def flash_attn_no_pad_v3(
|
||||
qkv: torch.Tensor,
|
||||
key_padding_mask: torch.Tensor,
|
||||
causal: bool = False,
|
||||
dropout_p: float = 0.0,
|
||||
softmax_scale: float | None = None,
|
||||
deterministic: bool = False,
|
||||
) -> torch.Tensor:
|
||||
qkv,
|
||||
key_padding_mask,
|
||||
causal=False,
|
||||
dropout_p=0.0,
|
||||
softmax_scale=None,
|
||||
deterministic=False,
|
||||
):
|
||||
from flash_attn_interface import (
|
||||
flash_attn_varlen_func as flash_attn_varlen_func_v3, )
|
||||
|
||||
@@ -134,16 +120,16 @@ def flash_attn_no_pad_v3(
|
||||
|
||||
|
||||
def flash_attn_varlen_qk_no_pad(
|
||||
query: torch.Tensor,
|
||||
key: torch.Tensor,
|
||||
value: torch.Tensor,
|
||||
query_padding_mask: torch.Tensor,
|
||||
key_padding_mask: torch.Tensor,
|
||||
causal: bool = False,
|
||||
dropout_p: float = 0.0,
|
||||
softmax_scale: float | None = None,
|
||||
deterministic: bool = False,
|
||||
) -> torch.Tensor:
|
||||
query,
|
||||
key,
|
||||
value,
|
||||
query_padding_mask,
|
||||
key_padding_mask,
|
||||
causal=False,
|
||||
dropout_p=0.0,
|
||||
softmax_scale=None,
|
||||
deterministic=False,
|
||||
):
|
||||
batch_size, q_seqlen, nheads, _ = query.shape
|
||||
|
||||
query_unpad, q_indices, cu_seqlens_q, max_seqlen_q, _ = unpad_input(rearrange(query, "b s h d -> b s (h d)"),
|
||||
|
||||
@@ -12,15 +12,8 @@ logger = init_logger(__name__)
|
||||
# 3. Any field in ArchConfig is fixed upon initialization, and should be hidden away from users
|
||||
@dataclass
|
||||
class ArchConfig:
|
||||
stacked_params_mapping: list[tuple[str, str, str | int]] = field(
|
||||
stacked_params_mapping: list[tuple[str, str, str]] = field(
|
||||
default_factory=list) # mapping from huggingface weight names to custom names
|
||||
output_hidden_states: bool = False
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
pass
|
||||
|
||||
def __getattr__(self, name: str) -> Any:
|
||||
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -31,22 +24,19 @@ class ModelConfig:
|
||||
|
||||
# FastVideo-specific parameters here
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
pass
|
||||
|
||||
def __getattr__(self, name: str) -> Any:
|
||||
def __getattr__(self, name):
|
||||
# Only called if 'name' is not found in ModelConfig directly
|
||||
if hasattr(self.arch_config, name):
|
||||
return getattr(self.arch_config, name)
|
||||
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
|
||||
|
||||
def __getstate__(self) -> dict[str, Any]:
|
||||
def __getstate__(self):
|
||||
# Return a dictionary of attributes to pickle
|
||||
# Convert to dict and exclude any problematic attributes
|
||||
state = self.__dict__.copy()
|
||||
return state
|
||||
|
||||
def __setstate__(self, state: dict[str, Any]) -> None:
|
||||
def __setstate__(self, state):
|
||||
# Restore instance attributes from the unpickled state
|
||||
self.__dict__.update(state)
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user